diff --git a/conf/test_full.config b/conf/test_full.config
index ea90cc16..35f705b0 100644
--- a/conf/test_full.config
+++ b/conf/test_full.config
@@ -10,15 +10,68 @@
----------------------------------------------------------------------------------------
*/
+process {
+ resourceLimits = [
+ cpus: 4,
+ memory: '15.GB',
+ time: '1.h'
+ ]
+
+ withName: FASTQC {
+ cpus = { 1 }
+ memory = { 15.GB * task.attempt }
+ }
+ withName: ADAPTER_REMOVAL {
+ cpus = { 8 }
+ memory = { 15.GB * task.attempt }
+ time = { 2.h * task.attempt }
+ }
+ withName: PICARD_CREATESEQUENCEDICTIONARY {
+ cpus = { 12 }
+ memory = { 15.GB * task.attempt }
+ time = { 8.h * task.attempt }
+ }
+ withName: PICARD_MARKDUPLICATES {
+ memory = { 15.GB }
+ }
+ withName: BWA_ALN {
+ cpus = { 8 }
+ memory = { 15.GB * task.attempt }
+ time = { 8.h * task.attempt }
+ }
+ withName: DEDUP {
+ cpus = { 8 }
+ memory = { 15.GB * task.attempt }
+ time = { 4.h * task.attempt }
+ }
+ withName: GENOTYPING_HC {
+ cpus = { 8 }
+ memory = { 15.GB * task.attempt }
+ time = { 8.h * task.attempt }
+ }
+}
+
params {
- config_profile_name = 'Full test profile'
- config_profile_description = 'Full test dataset to check pipeline function'
+ config_profile_name = 'Full test profile'
+ config_profile_description = 'Full test dataset to check pipeline function'
+ pipelines_testdata_base_path = 'https://raw.githubusercontent.com/TCLamnidis/test-datasets/'
// Input data for full size test
- // TODO nf-core: Specify the paths to your full test data ( on nf-core/test-datasets or directly in repositories, e.g. SRA)
- // TODO nf-core: Give any required params for the test so that command line flags are not needed
- input = params.pipelines_testdata_base_path + 'viralrecon/samplesheet/samplesheet_full_illumina_amplicon.csv'
+ input = params.pipelines_testdata_base_path + 'eager/testdata/Benchmarking/eager3_benchmarking_vikingfish.tsv'
// Genome references
- genome = 'R64-1-1'
+ // Genome reference
+ fasta = 'https://ftp.ncbi.nlm.nih.gov/genomes/refseq/vertebrate_other/Gadus_morhua/reference/GCF_902167405.1_gadMor3.0/GCF_902167405.1_gadMor3.0_rna.fna.gz'
+
+ bwaalnn = 0.04
+ bwaalnl = 1024
+
+ run_bam_filtering = true
+ bam_unmapped_type = 'discard'
+ bam_mapping_quality_threshold = 25
+
+ run_genotyping = true
+ genotyping_tool = 'hc'
+ genotyping_source = 'raw'
+ gatk_ploidy = 2
}
diff --git a/tests/test_full.nf.test b/tests/test_full.nf.test
new file mode 100644
index 00000000..9efdc314
--- /dev/null
+++ b/tests/test_full.nf.test
@@ -0,0 +1,152 @@
+nextflow_pipeline {
+
+ name "Test pipeline: NFCORE_EAGER"
+ script "main.nf"
+ tag "pipeline"
+ tag "nfcore_eager"
+ tag "test_full" // Tag containing the name of the profile to test. Should match the profile name below
+ profile "test_full" // The name of the profile used when testing
+
+ test("Test `test_full` profile:") {
+
+ when {
+ params {
+ outdir = "$outputDir"
+ }
+ }
+
+ then {
+
+ ///////////////////
+ // DOCUMENTATION //
+ ///////////////////
+
+ // The contents of each top level results directory should be tested with individually named snapshots.
+ // Within each snapshot, there should be two to three distinct variables, that contain the files to be tested.
+ // - stable_name_
is for files with variable md5sums (i.e. content) so only names will be compared
+ // - stable_content_ is for files with stable md5sums (i.e. content) so md5sums will be compared
+ // - bams_ is for BAM files, where the headerMD5 is checked for stability (since the content can be unstable)
+ // If a directory is fully stable, you can drop `stable_name_*`
+ // If a directory contains no BAMs, you can drop `bams_*`
+
+ // Due to the very long runtime of the full test, the snapshots were generated on the EVA computational cluster.
+ // Generate with: nf-test test --profile=+eva,archgen --tag test_full --update-snapshot
+ // Test with: nf-test test --profile=+eva,archgen --tag test_full
+ // NOTE: BAMs are always only stable in name, because:
+ // a) sharding breaks header since the shard that was first is named in the header (Fixed in https://github.com/nf-core/eager/pull/1112)
+ // b) the order of the reads in the BAMs is not stable (sorted, but reads that share a start position can be in any order)
+ // point b) also causes BAIs to be unstable.
+ // c) Merging of multiple BAMs with duplicate @RG / @PG tags can cause the header to be unstable (particularly in the case of shards/lanes)
+
+ //////////////////////
+ // DEFINE VARIABLES //
+ //////////////////////
+
+ // Define exclusion patterns for files with unstable contents
+ // NOTE: When a section needs more than a couple of small patterns, consider adding a variable to store the patterns here
+ // This is particularly important if the patterns excluded in the stable content section should be included in the stable name section
+ def unstable_patterns_auth = [
+ '**/mapped_reads_gc-content_distribution.txt',
+ '**/mapped_reads_nucleotide_content.txt',
+ '**/genome_gc_content_per_window.png',
+ '**/*.{svg,pdf,html,png}',
+ '**/DamageProfiler.log',
+ '**/3p_freq_misincorporations.txt',
+ '**/5p_freq_misincorporations.txt',
+ '**/DNA_comp_genome.txt',
+ '**/DNA_composition_sample.txt',
+ '**/misincorporation.txt',
+ '**/genome_results.txt',
+ '**/*command.log',
+ ]
+
+ // Check that no files are missing/added
+ // Command legend: Result directory to index , includeDir: include dirs?, ignore: exclude patterns , ignoreFile: exclude pattern list , include: include patterns
+ def stable_name_all = getAllFilesFromDir("$outputDir/" , includeDir: false , ignore: ['pipeline_info/*'] , ignoreFile: null , include: ['*', '**/*'] )
+
+ // Authentication
+ // def stable_content_authentication = getAllFilesFromDir("$outputDir/authentication" , includeDir: false , ignore: unstable_patterns_auth , ignoreFile: null , include: ['*', '**/*'] )
+ // def stable_name_authentication = getAllFilesFromDir("$outputDir/authentication" , includeDir: false , ignore: null , ignoreFile: null , include: unstable_patterns_auth)
+
+ // // Deduplication
+ // def stable_content_deduplication = getAllFilesFromDir("$outputDir/deduplication" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.flagstat'] )
+ // def stable_name_deduplication = getAllFilesFromDir("$outputDir/deduplication" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.{bam,bai}'] )
+
+ // // Final_bams
+ // def stable_content_final_bams = getAllFilesFromDir("$outputDir/final_bams" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.flagstat'] )
+ // def stable_name_final_bams = getAllFilesFromDir("$outputDir/final_bams" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.{bam,bai}'] )
+
+ // // Mapping (incl. bam_input flasgstat)
+ // def stable_content_mapping = getAllFilesFromDir("$outputDir/mapping" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.flagstat'] )
+ // def stable_name_mapping = getAllFilesFromDir("$outputDir/mapping" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.{bam,bai}'] )
+
+ // // Preprocessing
+ // // NOTE: FastQC html appears stable, but I worry it might just include a day timestamp instead of a full timestamp. To keep the expression simpler I removed both from checksum testing.
+ // def stable_content_preprocessing = getAllFilesFromDir("$outputDir/preprocessing" , includeDir: false , ignore: ['**/*.{zip,log,html}'], ignoreFile: null , include: ['**/*'] )
+ // def stable_name_preprocessing = getAllFilesFromDir("$outputDir/preprocessing" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.{zip,log,html}'] )
+
+ // // Read filtering
+ // def stable_content_readfiltering = getAllFilesFromDir("$outputDir/read_filtering" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.flagstat'] )
+ // def stable_name_readfiltering = getAllFilesFromDir("$outputDir/read_filtering" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.{bam,bai}'] )
+
+ // // Genotyping
+ // def stable_content_genotyping = getAllFilesFromDir("$outputDir/genotyping" , includeDir: false , ignore: ['**/*.{tbi,vcf.gz}'] , ignoreFile: null , include: ['**/*'] )
+ // def stable_name_genotyping = getAllFilesFromDir("$outputDir/genotyping" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.tbi'] )
+ // // We need to collect the vcfs separately to run more specific md5sum checks on the header (contnts are unstable due to same reasons as BAMs, explained above).
+ // def genotyping_vcfs = getAllFilesFromDir("$outputDir/genotyping" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.vcf.gz'] )
+
+ // // Metagenomics
+ // def stable_content_metagenomics = getAllFilesFromDir("$outputDir/metagenomics" , includeDir: false , ignore: ['**/*.biom', '**/*table.tsv'] , ignoreFile: null , include: ['**/*'] )
+ // def stable_name_metagenomics = getAllFilesFromDir("$outputDir/metagenomics" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.biom', '**/*table.tsv'] )
+
+ // MultiQC
+ // def stable_name_multiqc = getAllFilesFromDir("$outputDir/multiqc" , includeDir: false , ignore: null , ignoreFile: null , include: ['*', '**/*'] )
+
+ ///////////////////////
+ // DEFINE ASSERTIONS //
+ ///////////////////////
+
+ assertAll(
+ { assert workflow.success },
+ // This checks that there are no missing or additional output files.
+ // Also a good starting point to look at all the files in the output folder than need to be checked in subsequent sections.
+ { assert snapshot( stable_name_all*.name ).match("all_files") },
+
+ // Checking changes to contents of each section
+ // NOTE: Keep the order of the sections in the alphanumeric order of the output directories.
+ // Each section should first check stable_content, stable_name second (if applicable).
+ // { assert snapshot( stable_content_authentication , stable_name_authentication*.name ).match("authentication") },
+ // { assert snapshot( stable_content_deduplication , stable_name_deduplication*.name ).match("deduplication") },
+ // { assert snapshot( stable_content_final_bams , stable_name_final_bams*.name ).match("final_bams") },
+ // // NOTE: The snapshot section for mapping cannot be named 'mapping'. See https://github.com/askimed/nf-test/issues/279
+ // { assert snapshot( stable_content_mapping , stable_name_mapping*.name ).match("mapping_output") },
+ // { assert snapshot( stable_content_preprocessing , stable_name_preprocessing*.name ).match("preprocessing") },
+ // { assert snapshot( stable_content_readfiltering , stable_name_readfiltering*.name ).match("read_filtering") },
+ // { assert snapshot( stable_content_genotyping , stable_name_genotyping*.name ).match("genotyping") },
+ // // Additional checks on the genotyping VCFs for content. Specifically the md5sums of the header FORMAT, INFO, FILTER, CONTIG lines, and sample names
+ // { assert snapshot(
+ // genotyping_vcfs.collect {
+ // file ->
+ // def vcf_head = path(file.toString()).vcf.header
+ // // The header contains lines in the "OTHER" category, which contain a timestamp and/or work dir paths, so we need to filter those out, then calculate md5sums.
+ // def header_md5 = [
+ // vcf_head.getFormatHeaderLines().toString(),
+ // vcf_head.getInfoHeaderLines().toString(),
+ // vcf_head.getFilterLines().toString(),
+ // vcf_head.getIDHeaderLines().toString(),
+ // vcf_head.getGenotypeSamples().toString(),
+ // vcf_head.getContigLines().toString(),
+ // ].join(' ').md5()
+ // file.getName() + ":header_md5," + header_md5
+ // }
+ // ).match("genotyping_vcfs")},
+ // { assert snapshot( stable_content_metagenomics , stable_name_metagenomics*.name ).match("metagenomics") },
+ // { assert snapshot( stable_name_multiqc*.name ).match("multiqc") },
+
+ // Versions
+ { assert new File("$outputDir/pipeline_info/nf_core_eager_software_mqc_versions.yml").exists() },
+
+ )
+ }
+ }
+}
diff --git a/tests/test_full.nf.test.snap b/tests/test_full.nf.test.snap
new file mode 100644
index 00000000..38457193
--- /dev/null
+++ b/tests/test_full.nf.test.snap
@@ -0,0 +1,77 @@
+{
+ "all_files": {
+ "content": [
+ [
+ "COD076_COD076E1bL1_GCF_902167405.1_gadMor3.0_rna.c_curve.txt",
+ "COD076_COD076E1bL1_GCF_902167405.1_gadMor3.0_rna.command.log",
+ "COD092_COD092E1bL1i69_GCF_902167405.1_gadMor3.0_rna.c_curve.txt",
+ "COD092_COD092E1bL1i69_GCF_902167405.1_gadMor3.0_rna.command.log",
+ "COD076_COD076E1bL1_GCF_902167405.1_gadMor3.0_rna_sorted.bam",
+ "COD076_COD076E1bL1_GCF_902167405.1_gadMor3.0_rna_sorted.bam.bai",
+ "COD092_COD092E1bL1i69_GCF_902167405.1_gadMor3.0_rna_sorted.bam",
+ "COD092_COD092E1bL1i69_GCF_902167405.1_gadMor3.0_rna_sorted.bam.bai",
+ "COD076_COD076E1bL1_GCF_902167405.1_gadMor3.0_rna_sorted.flagstat",
+ "COD092_COD092E1bL1i69_GCF_902167405.1_gadMor3.0_rna_sorted.flagstat",
+ "COD076_COD076E1bL1_L1.fastp.html",
+ "COD076_COD076E1bL1_L1.fastp.json",
+ "COD076_COD076E1bL1_L1.fastp.log",
+ "COD076_COD076E1bL1_L6.fastp.html",
+ "COD076_COD076E1bL1_L6.fastp.json",
+ "COD076_COD076E1bL1_L6.fastp.log",
+ "COD076_COD076E1bL1_L8.fastp.html",
+ "COD076_COD076E1bL1_L8.fastp.json",
+ "COD076_COD076E1bL1_L8.fastp.log",
+ "COD092_COD092E1bL1i69_L6.fastp.html",
+ "COD092_COD092E1bL1i69_L6.fastp.json",
+ "COD092_COD092E1bL1i69_L6.fastp.log",
+ "COD092_COD092E1bL1i69_L7.fastp.html",
+ "COD092_COD092E1bL1i69_L7.fastp.json",
+ "COD092_COD092E1bL1i69_L7.fastp.log",
+ "COD092_COD092E1bL1i69_L8.fastp.html",
+ "COD092_COD092E1bL1i69_L8.fastp.json",
+ "COD092_COD092E1bL1i69_L8.fastp.log",
+ "COD076_COD076E1bL1_L1_fastqc.html",
+ "COD076_COD076E1bL1_L1_fastqc.zip",
+ "COD076_COD076E1bL1_L6_fastqc.html",
+ "COD076_COD076E1bL1_L6_fastqc.zip",
+ "COD076_COD076E1bL1_L8_fastqc.html",
+ "COD076_COD076E1bL1_L8_fastqc.zip",
+ "COD092_COD092E1bL1i69_L6_fastqc.html",
+ "COD092_COD092E1bL1i69_L6_fastqc.zip",
+ "COD092_COD092E1bL1i69_L7_fastqc.html",
+ "COD092_COD092E1bL1i69_L7_fastqc.zip",
+ "COD092_COD092E1bL1i69_L8_fastqc.html",
+ "COD092_COD092E1bL1i69_L8_fastqc.zip",
+ "COD076_COD076E1bL1_L1_1_fastqc.html",
+ "COD076_COD076E1bL1_L1_1_fastqc.zip",
+ "COD076_COD076E1bL1_L1_2_fastqc.html",
+ "COD076_COD076E1bL1_L1_2_fastqc.zip",
+ "COD076_COD076E1bL1_L6_1_fastqc.html",
+ "COD076_COD076E1bL1_L6_1_fastqc.zip",
+ "COD076_COD076E1bL1_L6_2_fastqc.html",
+ "COD076_COD076E1bL1_L6_2_fastqc.zip",
+ "COD076_COD076E1bL1_L8_1_fastqc.html",
+ "COD076_COD076E1bL1_L8_1_fastqc.zip",
+ "COD076_COD076E1bL1_L8_2_fastqc.html",
+ "COD076_COD076E1bL1_L8_2_fastqc.zip",
+ "COD092_COD092E1bL1i69_L6_1_fastqc.html",
+ "COD092_COD092E1bL1i69_L6_1_fastqc.zip",
+ "COD092_COD092E1bL1i69_L6_2_fastqc.html",
+ "COD092_COD092E1bL1i69_L6_2_fastqc.zip",
+ "COD092_COD092E1bL1i69_L7_1_fastqc.html",
+ "COD092_COD092E1bL1i69_L7_1_fastqc.zip",
+ "COD092_COD092E1bL1i69_L7_2_fastqc.html",
+ "COD092_COD092E1bL1i69_L7_2_fastqc.zip",
+ "COD092_COD092E1bL1i69_L8_1_fastqc.html",
+ "COD092_COD092E1bL1i69_L8_1_fastqc.zip",
+ "COD092_COD092E1bL1i69_L8_2_fastqc.html",
+ "COD092_COD092E1bL1i69_L8_2_fastqc.zip"
+ ]
+ ],
+ "meta": {
+ "nf-test": "0.9.3",
+ "nextflow": "25.10.2"
+ },
+ "timestamp": "2026-01-23T04:03:22.466351835"
+ }
+}
\ No newline at end of file