Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
@@ -1,3 +1,7 @@
## 0.4.0

1. Added a new worksheet field: `samplesheets_setting`. This field can be used to change the behaviour of the samplesheet creation. Currently this field only contains one options called `split_by` which can be used to specify a field to split on (for example on analysis tag)

## 0.3.0

1. Removed the hardcoded setup of samplesheet generation and migrated to a worksheet implementation to generate samplesheets. See the [worksheets](docs/worksheets.md) documentation for more information
Expand Down
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ To use the plugin add the following to your `nextflow.config`:

```groovy
plugins {
id 'nf-cmgg@0.3.0'
id 'nf-cmgg@0.4.0'
}
```

Expand Down
2 changes: 1 addition & 1 deletion build.gradle
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ dependencies {
implementation 'com.squareup.okhttp3:okhttp:5.3.2'
}

version = '0.3.0'
version = '0.4.0'

nextflowPlugin {
nextflowVersion = '25.10.0'
Expand Down
8 changes: 8 additions & 0 deletions docs/worksheets.md
Original file line number Diff line number Diff line change
Expand Up @@ -260,6 +260,14 @@ With `filter_func`, samples that fail are written to e.g. `nfcore_rnafusion_samp

In `include_func` and `filter_func`, `data` exposes everything defined in `input`, `values`, `output`, and `metrics`. Additionally it also has access to all parameters using the `params` structure.

## `samplesheet_settings`

A map containing settings on how to generate the samplesheets.

| Option | Meaning |
| ---------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
| `split_by` | A worksheet field name to split the samplesheets on. Whenever this option is used, a directory with the value of the worksheet field will be created and each samplesheet will be published in that location (only containing samplesheet entries with the same value in the specific worksheet field) |

## End-to-end example

The built-in worksheet for **nf-cmgg/preprocessing** ([`nfcmgg_preprocessing.yml`](../src/main/resources/worksheets/nfcmgg_preprocessing.yml)) shows the full pattern:
Expand Down
5 changes: 4 additions & 1 deletion src/main/groovy/nfcmgg/plugin/worksheet/Worksheet.groovy
Original file line number Diff line number Diff line change
Expand Up @@ -99,8 +99,11 @@ class Worksheet {
dataFields.addAll(parsedMetrics.fields.keySet())
}
try {
final WorksheetSamplesheetsSettings samplesheetSettings = new WorksheetSamplesheetsSettings(
worksheetMap.samplesheets_settings as Map, dataFields
)
parsedSamplesheets = new WorksheetSamplesheets(
worksheetMap.samplesheets as List<Map>, dataFields
worksheetMap.samplesheets as List<Map>, dataFields, samplesheetSettings
)
} catch (WorksheetException e) {
errors.record(e.message)
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -24,17 +24,20 @@ class WorksheetSamplesheets {

final SamplesheetCreator creator = new SamplesheetCreator()

final WorksheetSamplesheetsSettings settings

/**
* Samplesheet definitions in declaration order
*/
final List<Samplesheet> samplesheets

WorksheetSamplesheets(List<Map> samplesheets, Set<String> dataFields) {
WorksheetSamplesheets(List<Map> samplesheets, Set<String> dataFields, WorksheetSamplesheetsSettings settings) {
if (samplesheets == null || samplesheets.isEmpty()) {
final WorksheetErrors errors = new WorksheetErrors()
errors.error('Worksheet samplesheets is missing or empty')
errors.throwIfAny('Invalid worksheet samplesheets')
}
this.settings = settings
final List<Samplesheet> parsed = []
samplesheets.each { rawEntry ->
try {
Expand All @@ -60,7 +63,37 @@ class WorksheetSamplesheets {
this.samplesheets = parsed.asImmutable()
}

void publishSamplesheets(Map<String, OutputEntry> entries, Path location, Map<String, Object> params) {
void publishSamplesheets(
Map<String, OutputEntry> entries,
Path location,
Map<String, Object> params
) {
if (settings.splitBy) {
Map<String, Map<String, OutputEntry>> splitEntries = [:]
entries
.each { String key, OutputEntry entry ->
String splitOption = entry.values[settings.splitBy] as String ?: 'undefined'
if (!splitEntries.containsKey(splitOption)) {
splitEntries[splitOption] = [:]
}
splitEntries[splitOption][key] = entry
}
splitEntries
.each { String splitOption, Map<String, OutputEntry> newEntries ->
Path newLocation = location.resolve(splitOption)
pushSamplesheets(newEntries, newLocation, creator, params)
}
} else {
pushSamplesheets(entries, location, creator, params)
}
}

private void pushSamplesheets(
Map<String, OutputEntry> entries,
Path location,
SamplesheetCreator creator,
Map<String, Object> params
) {
samplesheets.each { Samplesheet samplesheet ->
try {
log.info("Publishing samplesheet '${samplesheet.name}'")
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,31 @@
package nfcmgg.plugin.worksheet

import groovy.transform.CompileStatic
import groovy.util.logging.Slf4j

/**
* A class used to define which samplesheets should be generated and what their content is
*/
@CompileStatic
@Slf4j
class WorksheetSamplesheetsSettings {

/**
* Samplesheet definitions in declaration order
*/
final String splitBy = null

WorksheetSamplesheetsSettings(Map<String,Object> samplesheetsSettings, Set<String> dataFields) {
final WorksheetErrors errors = new WorksheetErrors()

if (samplesheetsSettings?.containsKey('split_by')) {
final String splitByField = samplesheetsSettings['split_by'] as String
if (dataFields.contains(splitByField)) {
this.splitBy = splitByField
} else {
errors.error("${splitByField} is not a valid `split_by` field, use one of: ${dataFields.join(', ')}")
}
}
}

}
2 changes: 2 additions & 0 deletions src/main/resources/worksheets/nfcmgg_preprocessing.yml
Original file line number Diff line number Diff line change
Expand Up @@ -189,3 +189,5 @@ samplesheets:
source: family_number
cram:
crai:
samplesheets_settings:
split_by: tag
6 changes: 4 additions & 2 deletions src/test/groovy/nfcmgg/plugin/worksheet/WorksheetTest.groovy
Original file line number Diff line number Diff line change
Expand Up @@ -102,6 +102,7 @@ samplesheets:

void 'invalid samplesheet entries are skipped and valid ones are kept'() {
when:
Set<String> dataFields = ['sample'] as Set
WorksheetSamplesheets sheets = new WorksheetSamplesheets([
[fields: [sample: [:]]],
[
Expand All @@ -113,7 +114,7 @@ samplesheets:
include_func: 'data.unknown == true',
fields: [sample: [:]]
]
], ['sample'] as Set)
], dataFields, new WorksheetSamplesheetsSettings([:], dataFields))

then:
sheets.samplesheets.size() == 1
Expand All @@ -122,7 +123,8 @@ samplesheets:

void 'empty samplesheets block aborts because the block itself is required'() {
when:
new WorksheetSamplesheets([], ['sample'] as Set)
Set<String> dataFields = ['sample'] as Set
new WorksheetSamplesheets([], dataFields, new WorksheetSamplesheetsSettings([:], dataFields))

then:
thrown(WorksheetException)
Expand Down
16 changes: 16 additions & 0 deletions tests/lib/Utils.groovy
Original file line number Diff line number Diff line change
@@ -0,0 +1,16 @@
import groovy.transform.CompileDynamic

import java.nio.file.Path

/**
* A series of utility functions for testing.
*/
@CompileDynamic
class Utils {

/* groovylint-disable-next-line FactoryMethodName */
static List<String> create_yaml_snap(Path f, String outputDir) {
return [f.toString().tokenize('/')[-1], f.text.replaceAll("${outputDir}/", '').tokenize('\n')]
}

}
1 change: 1 addition & 0 deletions tests/preprocessing_mock/main.nf
Original file line number Diff line number Diff line change
Expand Up @@ -57,6 +57,7 @@ process MOCK_OUTPUT {
.collect { entry -> "echo '' | gzip > ${entry.samplename}.per-base.bed.gz && echo '' | gzip > ${entry.samplename}.per-base.bed.gz.csi"}
.join("\n ")
def sav_data = input_list
.findAll { entry -> entry.samplename }
.collect { entry -> "echo '${entry.samplename}\t${entry.get('reads_to_use_in_test', '-1')}' >> multiqc_SAV_data/multiqc_bclconvert_bysample.txt"}
.join("\n ")
"""
Expand Down
6 changes: 5 additions & 1 deletion tests/preprocessing_mock/main.nf.test
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,11 @@ nextflow_pipeline {
then {
assert workflow.success
assert snapshot(path("${outputDir}/samplesheets/").list().collectEntries { f ->
[f.toString().tokenize("/")[-1], f.text.replaceAll("${outputDir}/", '').tokenize('\n')]
if (!f.toString().endsWith('.yaml') && !f.toString().endsWith('.yml')) {
[f.toString().tokenize("/")[-1], f.list().collect { f2 -> Utils.create_yaml_snap(f2, outputDir) }]
} else {
Utils.create_yaml_snap(f, outputDir)
}
}).match()
}

Expand Down
Loading
Loading