diff --git a/CHANGELOG.md b/CHANGELOG.md index 79e461a..babaf4c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,7 @@ +## 0.4.0 + +1. Added a new worksheet field: `samplesheets_setting`. This field can be used to change the behaviour of the samplesheet creation. Currently this field only contains one options called `split_by` which can be used to specify a field to split on (for example on analysis tag) + ## 0.3.0 1. Removed the hardcoded setup of samplesheet generation and migrated to a worksheet implementation to generate samplesheets. See the [worksheets](docs/worksheets.md) documentation for more information diff --git a/README.md b/README.md index 32097fe..63dea36 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ To use the plugin add the following to your `nextflow.config`: ```groovy plugins { - id 'nf-cmgg@0.3.0' + id 'nf-cmgg@0.4.0' } ``` diff --git a/build.gradle b/build.gradle index 7c62fae..19408a2 100644 --- a/build.gradle +++ b/build.gradle @@ -6,7 +6,7 @@ dependencies { implementation 'com.squareup.okhttp3:okhttp:5.3.2' } -version = '0.3.0' +version = '0.4.0' nextflowPlugin { nextflowVersion = '25.10.0' diff --git a/docs/worksheets.md b/docs/worksheets.md index 72fb23b..45a8b6a 100644 --- a/docs/worksheets.md +++ b/docs/worksheets.md @@ -260,6 +260,14 @@ With `filter_func`, samples that fail are written to e.g. `nfcore_rnafusion_samp In `include_func` and `filter_func`, `data` exposes everything defined in `input`, `values`, `output`, and `metrics`. Additionally it also has access to all parameters using the `params` structure. +## `samplesheet_settings` + +A map containing settings on how to generate the samplesheets. + +| Option | Meaning | +| ---------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `split_by` | A worksheet field name to split the samplesheets on. Whenever this option is used, a directory with the value of the worksheet field will be created and each samplesheet will be published in that location (only containing samplesheet entries with the same value in the specific worksheet field) | + ## End-to-end example The built-in worksheet for **nf-cmgg/preprocessing** ([`nfcmgg_preprocessing.yml`](../src/main/resources/worksheets/nfcmgg_preprocessing.yml)) shows the full pattern: diff --git a/src/main/groovy/nfcmgg/plugin/worksheet/Worksheet.groovy b/src/main/groovy/nfcmgg/plugin/worksheet/Worksheet.groovy index 7a488ee..7120975 100644 --- a/src/main/groovy/nfcmgg/plugin/worksheet/Worksheet.groovy +++ b/src/main/groovy/nfcmgg/plugin/worksheet/Worksheet.groovy @@ -99,8 +99,11 @@ class Worksheet { dataFields.addAll(parsedMetrics.fields.keySet()) } try { + final WorksheetSamplesheetsSettings samplesheetSettings = new WorksheetSamplesheetsSettings( + worksheetMap.samplesheets_settings as Map, dataFields + ) parsedSamplesheets = new WorksheetSamplesheets( - worksheetMap.samplesheets as List, dataFields + worksheetMap.samplesheets as List, dataFields, samplesheetSettings ) } catch (WorksheetException e) { errors.record(e.message) diff --git a/src/main/groovy/nfcmgg/plugin/worksheet/WorksheetSamplesheets.groovy b/src/main/groovy/nfcmgg/plugin/worksheet/WorksheetSamplesheets.groovy index f01aa8a..943618b 100644 --- a/src/main/groovy/nfcmgg/plugin/worksheet/WorksheetSamplesheets.groovy +++ b/src/main/groovy/nfcmgg/plugin/worksheet/WorksheetSamplesheets.groovy @@ -24,17 +24,20 @@ class WorksheetSamplesheets { final SamplesheetCreator creator = new SamplesheetCreator() + final WorksheetSamplesheetsSettings settings + /** * Samplesheet definitions in declaration order */ final List samplesheets - WorksheetSamplesheets(List samplesheets, Set dataFields) { + WorksheetSamplesheets(List samplesheets, Set dataFields, WorksheetSamplesheetsSettings settings) { if (samplesheets == null || samplesheets.isEmpty()) { final WorksheetErrors errors = new WorksheetErrors() errors.error('Worksheet samplesheets is missing or empty') errors.throwIfAny('Invalid worksheet samplesheets') } + this.settings = settings final List parsed = [] samplesheets.each { rawEntry -> try { @@ -60,7 +63,37 @@ class WorksheetSamplesheets { this.samplesheets = parsed.asImmutable() } - void publishSamplesheets(Map entries, Path location, Map params) { + void publishSamplesheets( + Map entries, + Path location, + Map params + ) { + if (settings.splitBy) { + Map> splitEntries = [:] + entries + .each { String key, OutputEntry entry -> + String splitOption = entry.values[settings.splitBy] as String ?: 'undefined' + if (!splitEntries.containsKey(splitOption)) { + splitEntries[splitOption] = [:] + } + splitEntries[splitOption][key] = entry + } + splitEntries + .each { String splitOption, Map newEntries -> + Path newLocation = location.resolve(splitOption) + pushSamplesheets(newEntries, newLocation, creator, params) + } + } else { + pushSamplesheets(entries, location, creator, params) + } + } + + private void pushSamplesheets( + Map entries, + Path location, + SamplesheetCreator creator, + Map params + ) { samplesheets.each { Samplesheet samplesheet -> try { log.info("Publishing samplesheet '${samplesheet.name}'") diff --git a/src/main/groovy/nfcmgg/plugin/worksheet/WorksheetSamplesheetsSettings.groovy b/src/main/groovy/nfcmgg/plugin/worksheet/WorksheetSamplesheetsSettings.groovy new file mode 100644 index 0000000..798080e --- /dev/null +++ b/src/main/groovy/nfcmgg/plugin/worksheet/WorksheetSamplesheetsSettings.groovy @@ -0,0 +1,31 @@ +package nfcmgg.plugin.worksheet + +import groovy.transform.CompileStatic +import groovy.util.logging.Slf4j + +/** + * A class used to define which samplesheets should be generated and what their content is + */ +@CompileStatic +@Slf4j +class WorksheetSamplesheetsSettings { + + /** + * Samplesheet definitions in declaration order + */ + final String splitBy = null + + WorksheetSamplesheetsSettings(Map samplesheetsSettings, Set dataFields) { + final WorksheetErrors errors = new WorksheetErrors() + + if (samplesheetsSettings?.containsKey('split_by')) { + final String splitByField = samplesheetsSettings['split_by'] as String + if (dataFields.contains(splitByField)) { + this.splitBy = splitByField + } else { + errors.error("${splitByField} is not a valid `split_by` field, use one of: ${dataFields.join(', ')}") + } + } + } + +} diff --git a/src/main/resources/worksheets/nfcmgg_preprocessing.yml b/src/main/resources/worksheets/nfcmgg_preprocessing.yml index f1f4bfa..1216fa7 100644 --- a/src/main/resources/worksheets/nfcmgg_preprocessing.yml +++ b/src/main/resources/worksheets/nfcmgg_preprocessing.yml @@ -189,3 +189,5 @@ samplesheets: source: family_number cram: crai: +samplesheets_settings: + split_by: tag diff --git a/src/test/groovy/nfcmgg/plugin/worksheet/WorksheetTest.groovy b/src/test/groovy/nfcmgg/plugin/worksheet/WorksheetTest.groovy index 2873493..ab48353 100644 --- a/src/test/groovy/nfcmgg/plugin/worksheet/WorksheetTest.groovy +++ b/src/test/groovy/nfcmgg/plugin/worksheet/WorksheetTest.groovy @@ -102,6 +102,7 @@ samplesheets: void 'invalid samplesheet entries are skipped and valid ones are kept'() { when: + Set dataFields = ['sample'] as Set WorksheetSamplesheets sheets = new WorksheetSamplesheets([ [fields: [sample: [:]]], [ @@ -113,7 +114,7 @@ samplesheets: include_func: 'data.unknown == true', fields: [sample: [:]] ] - ], ['sample'] as Set) + ], dataFields, new WorksheetSamplesheetsSettings([:], dataFields)) then: sheets.samplesheets.size() == 1 @@ -122,7 +123,8 @@ samplesheets: void 'empty samplesheets block aborts because the block itself is required'() { when: - new WorksheetSamplesheets([], ['sample'] as Set) + Set dataFields = ['sample'] as Set + new WorksheetSamplesheets([], dataFields, new WorksheetSamplesheetsSettings([:], dataFields)) then: thrown(WorksheetException) diff --git a/tests/lib/Utils.groovy b/tests/lib/Utils.groovy new file mode 100644 index 0000000..43820ab --- /dev/null +++ b/tests/lib/Utils.groovy @@ -0,0 +1,16 @@ +import groovy.transform.CompileDynamic + +import java.nio.file.Path + +/** + * A series of utility functions for testing. + */ +@CompileDynamic +class Utils { + + /* groovylint-disable-next-line FactoryMethodName */ + static List create_yaml_snap(Path f, String outputDir) { + return [f.toString().tokenize('/')[-1], f.text.replaceAll("${outputDir}/", '').tokenize('\n')] + } + +} diff --git a/tests/preprocessing_mock/main.nf b/tests/preprocessing_mock/main.nf index a79c64d..345c5de 100644 --- a/tests/preprocessing_mock/main.nf +++ b/tests/preprocessing_mock/main.nf @@ -57,6 +57,7 @@ process MOCK_OUTPUT { .collect { entry -> "echo '' | gzip > ${entry.samplename}.per-base.bed.gz && echo '' | gzip > ${entry.samplename}.per-base.bed.gz.csi"} .join("\n ") def sav_data = input_list + .findAll { entry -> entry.samplename } .collect { entry -> "echo '${entry.samplename}\t${entry.get('reads_to_use_in_test', '-1')}' >> multiqc_SAV_data/multiqc_bclconvert_bysample.txt"} .join("\n ") """ diff --git a/tests/preprocessing_mock/main.nf.test b/tests/preprocessing_mock/main.nf.test index a0d3465..10573a8 100644 --- a/tests/preprocessing_mock/main.nf.test +++ b/tests/preprocessing_mock/main.nf.test @@ -16,7 +16,11 @@ nextflow_pipeline { then { assert workflow.success assert snapshot(path("${outputDir}/samplesheets/").list().collectEntries { f -> - [f.toString().tokenize("/")[-1], f.text.replaceAll("${outputDir}/", '').tokenize('\n')] + if (!f.toString().endsWith('.yaml') && !f.toString().endsWith('.yml')) { + [f.toString().tokenize("/")[-1], f.list().collect { f2 -> Utils.create_yaml_snap(f2, outputDir) }] + } else { + Utils.create_yaml_snap(f, outputDir) + } }).match() } diff --git a/tests/preprocessing_mock/main.nf.test.snap b/tests/preprocessing_mock/main.nf.test.snap index 3d6448d..c9e6b12 100644 --- a/tests/preprocessing_mock/main.nf.test.snap +++ b/tests/preprocessing_mock/main.nf.test.snap @@ -2,138 +2,240 @@ "Test preprocessing mock": { "content": [ { - "nfcmgg_exomecnv_samplesheet.yaml": [ - "- sample: D123456WES", - " batch: WES_prep_F", - " family: Proband_123456", - " cram: D123456WES.cram", - " crai: D123456WES.cram.crai", - " bed: D123456WES.per-base.bed.gz", - " bed_index: D123456WES.per-base.bed.gz.csi", - "- sample: mtD123456", - " batch: mito_prep_U", - " family: mito_family", - " cram: mtD123456.cram", - " crai: mtD123456.cram.crai", - " bed: mtD123456.per-base.bed.gz", - " bed_index: mtD123456.per-base.bed.gz.csi" + "RNAseqMDG": [ + [ + "nfcmgg_exomecnv_samplesheet.yaml", + [ + "[", + " ]" + ] + ], + [ + "nfcmgg_sampletracking_samplesheet.yaml", + [ + "[", + " ]" + ] + ], + [ + "nfcmgg_smallvariants_samplesheet.yaml", + [ + "[", + " ]" + ] + ], + [ + "nfcmgg_vivar_samplesheet.yaml", + [ + "[", + " ]" + ] + ], + [ + "nfcore_rnafusion_samplesheet.yaml", + [ + "- sample: R123456", + " fastq_1: R123456_R1_001.fastq.gz", + " fastq_2: R123456_R2_001.fastq.gz", + " strandedness: unknown", + " reads: '1200000'", + "- sample: R56789", + " fastq_1: R56789_R1_001.fastq.gz", + " fastq_2: R56789_R2_001.fastq.gz", + " strandedness: unknown", + " reads: '1200000'", + "- sample: R56789", + " fastq_1: R56789_R1_002.fastq.gz", + " fastq_2: R56789_R2_002.fastq.gz", + " strandedness: unknown", + " reads: '1200000'", + "- sample: R56789", + " fastq_1: R56789_R1_003.fastq.gz", + " fastq_2: R56789_R2_003.fastq.gz", + " strandedness: unknown", + " reads: '1200000'" + ] + ], + [ + "nfcore_rnafusion_samplesheet_failed.yaml", + [ + "- sample: R654321", + " fastq_1: R654321_R1_001.fastq.gz", + " fastq_2: R654321_R2_001.fastq.gz", + " strandedness: unknown", + " reads: '150'", + "- sample: R654322", + " fastq_1: R654322_R1_001.fastq.gz", + " fastq_2: R654322_R2_001.fastq.gz", + " strandedness: unknown", + " reads: '150'", + "- sample: R654322", + " fastq_1: R654322_R1_002.fastq.gz", + " fastq_2: R654322_R2_002.fastq.gz", + " strandedness: unknown", + " reads: '150'", + "- sample: R654322", + " fastq_1: R654322_R1_003.fastq.gz", + " fastq_2: R654322_R2_003.fastq.gz", + " strandedness: unknown", + " reads: '150'" + ] + ] ], - "nfcmgg_sampletracking_samplesheet.yaml": [ - "- sample: D123456WES", - " pool: WES_prep", - " sample_bam: D123456WES.cram", - " sample_bam_index: D123456WES.cram.crai", - " snp_bam: snp_D123456WES.cram", - " snp_bam_index: snp_D123456WES.cram.crai", - " sex: F", - "- sample: D123456WGS", - " pool: WGS_prep", - " sample_bam: D123456WGS.cram", - " sample_bam_index: D123456WGS.cram.crai", - " snp_bam: snp_D123456WGS.cram", - " snp_bam_index: snp_D123456WGS.cram.crai", - " sex: M", - "- sample: mtD123456", - " pool: mito_prep", - " sample_bam: mtD123456.cram", - " sample_bam_index: mtD123456.cram.crai", - " snp_bam: snp_mtD123456.cram", - " snp_bam_index: snp_mtD123456.cram.crai", - " sex: U" + "WES": [ + [ + "nfcmgg_exomecnv_samplesheet.yaml", + [ + "- sample: D123456WES", + " batch: WES_prep_F", + " family: Proband_123456", + " cram: D123456WES.cram", + " crai: D123456WES.cram.crai", + " bed: D123456WES.per-base.bed.gz", + " bed_index: D123456WES.per-base.bed.gz.csi", + "- sample: mtD123456", + " batch: mito_prep_U", + " family: mito_family", + " cram: mtD123456.cram", + " crai: mtD123456.cram.crai", + " bed: mtD123456.per-base.bed.gz", + " bed_index: mtD123456.per-base.bed.gz.csi" + ] + ], + [ + "nfcmgg_sampletracking_samplesheet.yaml", + [ + "- sample: D123456WES", + " pool: WES_prep", + " sample_bam: D123456WES.cram", + " sample_bam_index: D123456WES.cram.crai", + " snp_bam: snp_D123456WES.cram", + " snp_bam_index: snp_D123456WES.cram.crai", + " sex: F", + "- sample: mtD123456", + " pool: mito_prep", + " sample_bam: mtD123456.cram", + " sample_bam_index: mtD123456.cram.crai", + " snp_bam: snp_mtD123456.cram", + " snp_bam_index: snp_mtD123456.cram.crai", + " sex: U" + ] + ], + [ + "nfcmgg_smallvariants_samplesheet.yaml", + [ + "- sample: D123456WES", + " family: Proband_123456", + " cram: D123456WES.cram", + " crai: D123456WES.cram.crai", + "- sample: mtD123456", + " family: mito_family", + " cram: mtD123456.cram", + " crai: mtD123456.cram.crai" + ] + ], + [ + "nfcmgg_vivar_samplesheet.yaml", + [ + "- id: D123456WES", + " organism: Homo sapiens", + " tag: WES", + " binsize: 15", + " project: test_project", + " normdup: false", + " nipt: false", + " reads: D123456WES.cram", + " reads_index: D123456WES.cram.crai", + "- id: D123mouse", + " organism: Mus musculus", + " tag: WES", + " binsize: 100", + " project: test_project", + " normdup: false", + " nipt: false", + " reads: D123mouse.cram", + " reads_index: D123mouse.cram.crai" + ] + ], + [ + "nfcore_rnafusion_samplesheet.yaml", + [ + "[", + " ]" + ] + ], + [ + "nfcore_rnafusion_samplesheet_failed.yaml", + [ + "[", + " ]" + ] + ] ], - "nfcmgg_smallvariants_samplesheet.yaml": [ - "- sample: D123456WES", - " family: Proband_123456", - " cram: D123456WES.cram", - " crai: D123456WES.cram.crai", - "- sample: D123456WGS", - " family: Proband_123456", - " cram: D123456WGS.cram", - " crai: D123456WGS.cram.crai", - "- sample: mtD123456", - " family: mito_family", - " cram: mtD123456.cram", - " crai: mtD123456.cram.crai" - ], - "nfcmgg_vivar_samplesheet.yaml": [ - "- id: D123456WES", - " organism: Homo sapiens", - " tag: WES", - " binsize: 15", - " project: test_project", - " normdup: false", - " nipt: false", - " reads: D123456WES.cram", - " reads_index: D123456WES.cram.crai", - "- id: D123456WGS", - " organism: Homo sapiens", - " tag: WGS", - " binsize: 10", - " project: test_project", - " normdup: false", - " nipt: false", - " reads: D123456WGS.cram", - " reads_index: D123456WGS.cram.crai", - "- id: D123mouse", - " organism: Mus musculus", - " tag: WES", - " binsize: 100", - " project: test_project", - " normdup: false", - " nipt: false", - " reads: D123mouse.cram", - " reads_index: D123mouse.cram.crai" - ], - "nfcore_rnafusion_samplesheet.yaml": [ - "- sample: R123456", - " fastq_1: R123456_R1_001.fastq.gz", - " fastq_2: R123456_R2_001.fastq.gz", - " strandedness: unknown", - " reads: '1200000'", - "- sample: R56789", - " fastq_1: R56789_R1_001.fastq.gz", - " fastq_2: R56789_R2_001.fastq.gz", - " strandedness: unknown", - " reads: '1200000'", - "- sample: R56789", - " fastq_1: R56789_R1_002.fastq.gz", - " fastq_2: R56789_R2_002.fastq.gz", - " strandedness: unknown", - " reads: '1200000'", - "- sample: R56789", - " fastq_1: R56789_R1_003.fastq.gz", - " fastq_2: R56789_R2_003.fastq.gz", - " strandedness: unknown", - " reads: '1200000'" - ], - "nfcore_rnafusion_samplesheet_failed.yaml": [ - "- sample: R654321", - " fastq_1: R654321_R1_001.fastq.gz", - " fastq_2: R654321_R2_001.fastq.gz", - " strandedness: unknown", - " reads: '150'", - "- sample: R654322", - " fastq_1: R654322_R1_001.fastq.gz", - " fastq_2: R654322_R2_001.fastq.gz", - " strandedness: unknown", - " reads: '150'", - "- sample: R654322", - " fastq_1: R654322_R1_002.fastq.gz", - " fastq_2: R654322_R2_002.fastq.gz", - " strandedness: unknown", - " reads: '150'", - "- sample: R654322", - " fastq_1: R654322_R1_003.fastq.gz", - " fastq_2: R654322_R2_003.fastq.gz", - " strandedness: unknown", - " reads: '150'" + "WGS": [ + [ + "nfcmgg_exomecnv_samplesheet.yaml", + [ + "[", + " ]" + ] + ], + [ + "nfcmgg_sampletracking_samplesheet.yaml", + [ + "- sample: D123456WGS", + " pool: WGS_prep", + " sample_bam: D123456WGS.cram", + " sample_bam_index: D123456WGS.cram.crai", + " snp_bam: snp_D123456WGS.cram", + " snp_bam_index: snp_D123456WGS.cram.crai", + " sex: M" + ] + ], + [ + "nfcmgg_smallvariants_samplesheet.yaml", + [ + "- sample: D123456WGS", + " family: Proband_123456", + " cram: D123456WGS.cram", + " crai: D123456WGS.cram.crai" + ] + ], + [ + "nfcmgg_vivar_samplesheet.yaml", + [ + "- id: D123456WGS", + " organism: Homo sapiens", + " tag: WGS", + " binsize: 10", + " project: test_project", + " normdup: false", + " nipt: false", + " reads: D123456WGS.cram", + " reads_index: D123456WGS.cram.crai" + ] + ], + [ + "nfcore_rnafusion_samplesheet.yaml", + [ + "[", + " ]" + ] + ], + [ + "nfcore_rnafusion_samplesheet_failed.yaml", + [ + "[", + " ]" + ] + ] ] } ], - "timestamp": "2026-09-04T14:50:53.833600948", + "timestamp": "2026-09-24T14:47:18.311821057", "meta": { "nf-test": "0.9.5", - "nextflow": "26.04.0" + "nextflow": "26.04.6" } } } \ No newline at end of file diff --git a/worksheet-schema.json b/worksheet-schema.json index d0bb752..0116af9 100644 --- a/worksheet-schema.json +++ b/worksheet-schema.json @@ -196,6 +196,16 @@ } } } + }, + "samplesheets_settings": { + "type": "object", + "description": "Settings related to how samplesheets should be generated.", + "properties": { + "split_by": { + "type": "string", + "description": "The field by which the samplesheet should be split. An additional directory will be created which will consist of this value. All samples for which the specified value is empty will be in the `other` directory." + } + } } } }