diff --git a/.github/workflows/build_container.yml b/.github/workflows/build_container.yml new file mode 100644 index 0000000..1b97a6a --- /dev/null +++ b/.github/workflows/build_container.yml @@ -0,0 +1,102 @@ +name: Build R container + +# The image tag is read from conf/containers.config, so that single line is both what the +# pipeline pulls and what this workflow publishes -- the two cannot drift. +# Runs on pushes to a feature branch, whenever the container changes. + +on: + push: + paths: + - 'containers/r/**' + - 'conf/containers.config' + - '.github/workflows/build_container.yml' + workflow_dispatch: + inputs: + force: + description: 'Rebuild and overwrite the tag even on main/devel' + type: boolean + default: false + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + +jobs: + build: + name: Build and publish + runs-on: ubuntu-latest + permissions: + contents: read + packages: write + + steps: + - uses: actions/checkout@v4 + + - name: Resolve image from conf/containers.config + id: image + run: | + IMAGE=$(sed -n "s|.*withLabel: *'r'.*container *= *\"\([^\"]*\)\".*|\1|p" conf/containers.config) + if [ -z "$IMAGE" ]; then + echo "::error::could not parse the 'r' container from conf/containers.config" + exit 1 + fi + case "$IMAGE" in + ghcr.io/goekelab/*) ;; + *) echo "::error::refusing to push outside ghcr.io/goekelab: $IMAGE"; exit 1 ;; + esac + echo "image=$IMAGE" >> "$GITHUB_OUTPUT" + echo "Resolved container: $IMAGE" + + - name: Refuse to overwrite a tag main or devel is pinned to + if: github.ref_name != 'main' && github.ref_name != 'devel' + run: | + IMAGE="${{ steps.image.outputs.image }}" + for BASE in main devel; do + git fetch --no-tags --depth=1 origin "$BASE" 2>/dev/null || continue + PINNED=$(git show "FETCH_HEAD:conf/containers.config" 2>/dev/null \ + | sed -n "s|.*withLabel: *'r'.*container *= *\"\([^\"]*\)\".*|\1|p") + if [ "$IMAGE" = "$PINNED" ]; then + echo "::error::$BASE is pinned to $IMAGE -- bump the tag in conf/containers.config before rebuilding it" + exit 1 + fi + done + + - uses: docker/login-action@v3 + with: + registry: ghcr.io + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + + - name: Decide whether to build + id: decide + run: | + IMAGE="${{ steps.image.outputs.image }}" + if docker manifest inspect "$IMAGE" >/dev/null 2>&1; then EXISTS=true; else EXISTS=false; fi + + if [ "${{ inputs.force }}" = "true" ]; then + BUILD=true + elif [ "${{ github.ref_name }}" = "main" ] || [ "${{ github.ref_name }}" = "devel" ]; then + # Released tags are immutable: the merge that brought the bump in has already + # published it from the feature branch, so there is nothing left to do + [ "$EXISTS" = "false" ] && BUILD=true || BUILD=false + else + # Feature branch: the tag is unreleased (guarded above), so rebuild freely + BUILD=true + fi + + echo "build=$BUILD" >> "$GITHUB_OUTPUT" + if [ "$BUILD" = "false" ]; then + echo "::notice::$IMAGE is already published -- skipping build" + fi + + - uses: docker/setup-buildx-action@v3 + if: steps.decide.outputs.build == 'true' + + - uses: docker/build-push-action@v6 + if: steps.decide.outputs.build == 'true' + with: + context: containers/r + push: true + tags: ${{ steps.image.outputs.image }} + cache-from: type=gha + cache-to: type=gha,mode=max diff --git a/.github/workflows/smoke_test.yml b/.github/workflows/smoke_test.yml index 8351ebb..cfea629 100644 --- a/.github/workflows/smoke_test.yml +++ b/.github/workflows/smoke_test.yml @@ -1,10 +1,13 @@ name: Smoke Tests on: - push: - branches: [main, devel] pull_request: branches: [main, devel] + workflow_dispatch: + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true jobs: smoke_test: diff --git a/conf/containers.config b/conf/containers.config index 60465f9..3cac435 100644 --- a/conf/containers.config +++ b/conf/containers.config @@ -1,5 +1,5 @@ process { - withLabel: 'r' { container = "ghcr.io/ch99l/bambu-pipe-r:latest" } + withLabel: 'r' { container = "ghcr.io/goekelab/bambu-pipe-r:1.0.0" } withLabel: 'spaceranger' { container = "quay.io/nf-core/spaceranger:9c5e7dc93c32448e" } withLabel: 'minimap2_samtools' { container = "community.wave.seqera.io/library/minimap2_samtools:b09096fc890429ce" } withLabel: 'preprocess' { container = "community.wave.seqera.io/library/chopper_cutadapt_flexiplex_pigz:077c3bc67452482c" } diff --git a/containers/r/Dockerfile b/containers/r/Dockerfile index 294f264..fcdd656 100644 --- a/containers/r/Dockerfile +++ b/containers/r/Dockerfile @@ -7,6 +7,7 @@ RUN apt-get update && apt-get install -y libcurl4-openssl-dev procps git \ ARG MAMBA_DOCKERFILE_ACTIVATE=1 # Install bioconductor version of bambu for its dependencies +# TODO: data.table 1.17.8 installation is a temp fix for clustered_EM RUN micromamba install -y -n base -c conda-forge -c bioconda \ r-base=4.5.3 \ r-devtools=2.5.2 \ @@ -15,11 +16,12 @@ RUN micromamba install -y -n base -c conda-forge -c bioconda \ bioconductor-dropletutils=1.30.0 \ r-seurat=5.4.0 \ r-harmony=2.0.2 \ + r-data.table=1.17.8 \ gxx=15.2.0 \ && micromamba clean --all --yes # Clone bambu single cell feature branch -RUN git clone --branch exportNewFunctions \ +RUN git clone --branch devel_pre_v4 \ https://github.com/GoekeLab/bambu.git /opt/bambu \ && R -e 'devtools::install("/opt/bambu")' diff --git a/main.nf b/main.nf index b121aaa..2b45d93 100644 --- a/main.nf +++ b/main.nf @@ -88,7 +88,7 @@ workflow { if (params.quantification_mode != 'no_quant') { if (params.quantification_mode == 'EM_clusters') { - CLUSTERING(BAMBU_TRANSCRIPT_DISCOVERY.out.se_gene_counts, BAMBU_TRANSCRIPT_DISCOVERY.out.sample_names, ch_n_samples) + CLUSTERING(BAMBU_TRANSCRIPT_DISCOVERY.out.se_gene_counts, ch_n_samples) ch_clusters = CLUSTERING.out.clusters.map { clusters -> [true, clusters] } // flag to indicate that clustering was performed } else { ch_clusters = channel.value([false, []]) // flag to indicate that clustering was not performed diff --git a/modules/bambu/EM_quant.nf b/modules/bambu/EM_quant.nf index 09abaa4..b10b28e 100644 --- a/modules/bambu/EM_quant.nf +++ b/modules/bambu/EM_quant.nf @@ -32,14 +32,11 @@ process BAMBU_EM{ degBias <- !is.null(clusters) se <- bambu.singlecell( - reads = NULL, + reads = quantData, + output = if (is.null(clusters)) "EM" else "clusteredEM", annotations = extendedAnno, genome = "$genome", - quantData = quantData, - assignDist = FALSE, ncore = $task.cpus, - discovery = FALSE, - quant = TRUE, verbose = FALSE, opt.em = list(degradationBias = degBias), clusters = clusters diff --git a/modules/bambu/construct_read_class.nf b/modules/bambu/construct_read_class.nf index 2f423bb..8a7270c 100644 --- a/modules/bambu/construct_read_class.nf +++ b/modules/bambu/construct_read_class.nf @@ -22,8 +22,8 @@ process BAMBU_CONSTRUCT_READ_CLASS{ annotation <- readRDS("$bambu_annotation") # Rename BAM to val(sample) so bambu derives sampleName from val(sample), regardless of the original filename file.symlink("$bam", "${sample}.bam") - readClassFile <- bambu.singlecell(reads = "${sample}.bam", annotations = annotation, genome = "$genome", - ncore = $task.cpus, discovery = FALSE, quant = FALSE, verbose = FALSE, assignDist = FALSE, + readClassFile <- bambu.singlecell(reads = "${sample}.bam", output = "readClasses", + annotations = annotation, genome = "$genome", ncore = $task.cpus, verbose = FALSE, processByChromosome = as.logical("$params.process_by_chromosome"), yieldSize = 10000000) saveRDS(readClassFile[[1]], "${sample}_read_class.rds") diff --git a/modules/bambu/transcript_discovery.nf b/modules/bambu/transcript_discovery.nf index dba495f..8214740 100644 --- a/modules/bambu/transcript_discovery.nf +++ b/modules/bambu/transcript_discovery.nf @@ -20,7 +20,6 @@ process BAMBU_TRANSCRIPT_DISCOVERY{ path ('gene_counts/se_gene_counts.rds'), emit: se_gene_counts path ('extended_annotations.rds'), emit: extended_annotations path ('extended_annotations.gtf') - path ('sample_names.rds'), emit: sample_names path ('unique_counts') path ('gene_counts') path "versions.yml", topic: 'versions' @@ -33,26 +32,26 @@ process BAMBU_TRANSCRIPT_DISCOVERY{ source(Sys.which("save_counts.R")) annotation <- readRDS("$bambu_annotation") - readClassFile <- strsplit("${rds_files.join(',')}", ",")[[1]] sampleNames <- strsplit("${sample.join(',')}", ",")[[1]] + readClassFile <- setNames(strsplit("${rds_files.join(',')}", ",")[[1]], sampleNames) sampleData <- strsplit("${spatial_metadata_files.join(',')}", ",")[[1]] chemistry <- setNames(strsplit("${meta.collect { m -> m.chemistry }.join(',')}", ",")[[1]], sampleNames) technology <- setNames(strsplit("${meta.collect { m -> m.technology }.join(',')}", ",")[[1]], sampleNames) # Transcript discovery - extendedAnno <- bambu.singlecell(reads = readClassFile, annotations = annotation, genome = "$genome", ncore = $task.cpus, - discovery = TRUE, quant = FALSE, verbose = FALSE, assignDist = FALSE, NDR = $ndr) + extendedAnno <- bambu.singlecell(reads = readClassFile, output = "extendedAnnotations", + annotations = annotation, genome = "$genome", ncore = $task.cpus, verbose = FALSE, NDR = $ndr) saveRDS(extendedAnno, "extended_annotations.rds") writeToGTF(extendedAnno, "extended_annotations.gtf") - # Quantification without EM + # Read to transcript assignment sampleData <- if (any(startsWith(chemistry, "visium-v"))) sampleData else NULL # Add spatial metadata for visium samples - quantData <- bambu.singlecell(reads = readClassFile, annotations = extendedAnno, genome = "$genome", ncore = $task.cpus, - discovery = FALSE, quant = FALSE, verbose = FALSE, opt.em = list(degradationBias = FALSE), assignDist = TRUE, sampleData = sampleData) + quantData <- bambu.singlecell(reads = readClassFile, output = "quantData", + annotations = extendedAnno, genome = "$genome", ncore = $task.cpus, verbose = FALSE, sampleData = sampleData) saveRDS(quantData, "quant_data.rds") - # Generate unique counts SE from quantData - seDiscovery <- generateUniqueCountsSEFromQuantData(quantData, extendedAnno) + # Quantification without EM + seDiscovery <- bambu.singlecell(reads = quantData, output = "uniqueCounts", annotations = extendedAnno) colData(seDiscovery)\$chemistry <- unname(chemistry[colData(seDiscovery)\$sampleName]) # Add chemistry into colData (for subsequent batch correction) colData(seDiscovery)\$technology <- unname(technology[colData(seDiscovery)\$sampleName]) # Add technology into colData (for subsequent batch correction) save_counts(seDiscovery, "unique_counts", "Transcript Expression") @@ -61,9 +60,6 @@ process BAMBU_TRANSCRIPT_DISCOVERY{ seDiscovery.gene <- transcriptToGeneExpression(seDiscovery) save_counts(seDiscovery.gene, "gene_counts") - # Save sampleNames (required for multi-sample Seurat clustering) - saveRDS(sampleNames, "sample_names.rds") - writeLines(c('"${task.process}":', paste0(' R: ', R.Version()\$version.string), paste0(' bambu: ', as.character(packageVersion("bambu")))), "versions.yml") """ } \ No newline at end of file diff --git a/modules/seurat/multi_sample.nf b/modules/seurat/multi_sample.nf index 40f609b..2316cfb 100644 --- a/modules/seurat/multi_sample.nf +++ b/modules/seurat/multi_sample.nf @@ -8,7 +8,6 @@ process SEURAT_MULTI_SAMPLE { input: path(se) - path(sample_names) output: path ('clusters.rds'), emit: clusters @@ -19,7 +18,6 @@ process SEURAT_MULTI_SAMPLE { """ #!/usr/bin/env Rscript library(SummarizedExperiment) - library(IRanges) library(Seurat) # Extract gene count matrix and colData metadata @@ -55,24 +53,13 @@ process SEURAT_MULTI_SAMPLE { cellMix <- FindClusters(cellMix, resolution = $params.resolution, cluster.name = "harmony_clusters") saveRDS(cellMix, "seurat_obj.rds") - # Build ordered list of CompressedCharacterLists, one per sample, in quantData order - allBarcodes <- names(cellMix\$harmony_clusters) - clusterLabels <- paste0("cluster", as.character(cellMix\$harmony_clusters)) - sampleNames <- readRDS("$sample_names") # sampleNames contain the order of the samples in quantData - - clusters <- setNames(lapply(sampleNames, function(s) { - idx <- which(cellMix\$sample == s) - sampleBarcode <- allBarcodes[idx] - clusterLabel <- paste0(s, "_", clusterLabels[idx]) - splitAsList(sampleBarcode, clusterLabel) - }), sampleNames) + clusters <- setNames(paste0("cluster_", cellMix\$harmony_clusters), names(cellMix\$harmony_clusters)) saveRDS(clusters, "clusters.rds") writeLines(c( '"${task.process}":', paste0(' R: ', R.Version()\$version.string), paste0(' seurat: ', as.character(packageVersion("Seurat"))), - paste0(' IRanges: ', as.character(packageVersion("IRanges"))), paste0(' SummarizedExperiment: ', as.character(packageVersion("SummarizedExperiment"))) ), "versions.yml") """ diff --git a/modules/seurat/single_sample.nf b/modules/seurat/single_sample.nf index 14a61f3..283374c 100644 --- a/modules/seurat/single_sample.nf +++ b/modules/seurat/single_sample.nf @@ -18,7 +18,6 @@ process SEURAT_SINGLE_SAMPLE { """ #!/usr/bin/env Rscript library(SummarizedExperiment) - library(IRanges) library(Seurat) se <- readRDS("$se") @@ -34,19 +33,17 @@ process SEURAT_SINGLE_SAMPLE { cellMix <- RunPCA(cellMix, features = VariableFeatures(object = cellMix), npcs = npcs) dim <- ifelse(dim >= dim(cellMix@reductions\$pca)[2], dim(cellMix@reductions\$pca)[2], dim) cellMix <- FindNeighbors(cellMix, dims = 1:dim) - cellMix <- FindClusters(cellMix, resolution = $params.resolution) + cellMix <- FindClusters(cellMix, resolution = $params.resolution, cluster.name = "clusters") + saveRDS(cellMix, "seurat_obj.rds") - x <- setNames(names(cellMix@active.ident), cellMix@active.ident) - clusters <- list(splitAsList(unname(x), paste0("cluster", names(x)))) - - saveRDS(cellMix, "cell_mix.rds") + clusters <- setNames(paste0("cluster_", cellMix\$clusters), names(cellMix\$clusters)) saveRDS(clusters, "clusters.rds") + writeLines(c( '"${task.process}":', paste0(' R: ', R.Version()\$version.string), paste0(' seurat: ', as.character(packageVersion("Seurat"))), - paste0(' IRanges: ', as.character(packageVersion("IRanges"))), paste0(' SummarizedExperiment: ', as.character(packageVersion("SummarizedExperiment"))) ), "versions.yml") """ diff --git a/subworkflows/clustering.nf b/subworkflows/clustering.nf index 2e4b127..08a3979 100644 --- a/subworkflows/clustering.nf +++ b/subworkflows/clustering.nf @@ -4,7 +4,6 @@ include { SEURAT_MULTI_SAMPLE } from '../modules/seurat/multi_sample.nf' workflow CLUSTERING { take: ch_se_gene_counts - ch_sample_names ch_n_samples main: @@ -16,7 +15,7 @@ workflow CLUSTERING { } SEURAT_SINGLE_SAMPLE(ch_se_branched.single.map { se, _n -> se }) - SEURAT_MULTI_SAMPLE(ch_se_branched.multi.map { se, _n -> se }, ch_sample_names) + SEURAT_MULTI_SAMPLE(ch_se_branched.multi.map { se, _n -> se }) emit: clusters = SEURAT_SINGLE_SAMPLE.out.clusters.mix(SEURAT_MULTI_SAMPLE.out.clusters)