diff --git a/conf/test_full.config b/conf/test_full.config index ea90cc16..35f705b0 100644 --- a/conf/test_full.config +++ b/conf/test_full.config @@ -10,15 +10,68 @@ ---------------------------------------------------------------------------------------- */ +process { + resourceLimits = [ + cpus: 4, + memory: '15.GB', + time: '1.h' + ] + + withName: FASTQC { + cpus = { 1 } + memory = { 15.GB * task.attempt } + } + withName: ADAPTER_REMOVAL { + cpus = { 8 } + memory = { 15.GB * task.attempt } + time = { 2.h * task.attempt } + } + withName: PICARD_CREATESEQUENCEDICTIONARY { + cpus = { 12 } + memory = { 15.GB * task.attempt } + time = { 8.h * task.attempt } + } + withName: PICARD_MARKDUPLICATES { + memory = { 15.GB } + } + withName: BWA_ALN { + cpus = { 8 } + memory = { 15.GB * task.attempt } + time = { 8.h * task.attempt } + } + withName: DEDUP { + cpus = { 8 } + memory = { 15.GB * task.attempt } + time = { 4.h * task.attempt } + } + withName: GENOTYPING_HC { + cpus = { 8 } + memory = { 15.GB * task.attempt } + time = { 8.h * task.attempt } + } +} + params { - config_profile_name = 'Full test profile' - config_profile_description = 'Full test dataset to check pipeline function' + config_profile_name = 'Full test profile' + config_profile_description = 'Full test dataset to check pipeline function' + pipelines_testdata_base_path = 'https://raw.githubusercontent.com/TCLamnidis/test-datasets/' // Input data for full size test - // TODO nf-core: Specify the paths to your full test data ( on nf-core/test-datasets or directly in repositories, e.g. SRA) - // TODO nf-core: Give any required params for the test so that command line flags are not needed - input = params.pipelines_testdata_base_path + 'viralrecon/samplesheet/samplesheet_full_illumina_amplicon.csv' + input = params.pipelines_testdata_base_path + 'eager/testdata/Benchmarking/eager3_benchmarking_vikingfish.tsv' // Genome references - genome = 'R64-1-1' + // Genome reference + fasta = 'https://ftp.ncbi.nlm.nih.gov/genomes/refseq/vertebrate_other/Gadus_morhua/reference/GCF_902167405.1_gadMor3.0/GCF_902167405.1_gadMor3.0_rna.fna.gz' + + bwaalnn = 0.04 + bwaalnl = 1024 + + run_bam_filtering = true + bam_unmapped_type = 'discard' + bam_mapping_quality_threshold = 25 + + run_genotyping = true + genotyping_tool = 'hc' + genotyping_source = 'raw' + gatk_ploidy = 2 } diff --git a/tests/test_full.nf.test b/tests/test_full.nf.test new file mode 100644 index 00000000..9efdc314 --- /dev/null +++ b/tests/test_full.nf.test @@ -0,0 +1,152 @@ +nextflow_pipeline { + + name "Test pipeline: NFCORE_EAGER" + script "main.nf" + tag "pipeline" + tag "nfcore_eager" + tag "test_full" // Tag containing the name of the profile to test. Should match the profile name below + profile "test_full" // The name of the profile used when testing + + test("Test `test_full` profile:") { + + when { + params { + outdir = "$outputDir" + } + } + + then { + + /////////////////// + // DOCUMENTATION // + /////////////////// + + // The contents of each top level results directory should be tested with individually named snapshots. + // Within each snapshot, there should be two to three distinct variables, that contain the files to be tested. + // - stable_name_ is for files with variable md5sums (i.e. content) so only names will be compared + // - stable_content_ is for files with stable md5sums (i.e. content) so md5sums will be compared + // - bams_ is for BAM files, where the headerMD5 is checked for stability (since the content can be unstable) + // If a directory is fully stable, you can drop `stable_name_*` + // If a directory contains no BAMs, you can drop `bams_*` + + // Due to the very long runtime of the full test, the snapshots were generated on the EVA computational cluster. + // Generate with: nf-test test --profile=+eva,archgen --tag test_full --update-snapshot + // Test with: nf-test test --profile=+eva,archgen --tag test_full + // NOTE: BAMs are always only stable in name, because: + // a) sharding breaks header since the shard that was first is named in the header (Fixed in https://github.com/nf-core/eager/pull/1112) + // b) the order of the reads in the BAMs is not stable (sorted, but reads that share a start position can be in any order) + // point b) also causes BAIs to be unstable. + // c) Merging of multiple BAMs with duplicate @RG / @PG tags can cause the header to be unstable (particularly in the case of shards/lanes) + + ////////////////////// + // DEFINE VARIABLES // + ////////////////////// + + // Define exclusion patterns for files with unstable contents + // NOTE: When a section needs more than a couple of small patterns, consider adding a variable to store the patterns here + // This is particularly important if the patterns excluded in the stable content section should be included in the stable name section + def unstable_patterns_auth = [ + '**/mapped_reads_gc-content_distribution.txt', + '**/mapped_reads_nucleotide_content.txt', + '**/genome_gc_content_per_window.png', + '**/*.{svg,pdf,html,png}', + '**/DamageProfiler.log', + '**/3p_freq_misincorporations.txt', + '**/5p_freq_misincorporations.txt', + '**/DNA_comp_genome.txt', + '**/DNA_composition_sample.txt', + '**/misincorporation.txt', + '**/genome_results.txt', + '**/*command.log', + ] + + // Check that no files are missing/added + // Command legend: Result directory to index , includeDir: include dirs?, ignore: exclude patterns , ignoreFile: exclude pattern list , include: include patterns + def stable_name_all = getAllFilesFromDir("$outputDir/" , includeDir: false , ignore: ['pipeline_info/*'] , ignoreFile: null , include: ['*', '**/*'] ) + + // Authentication + // def stable_content_authentication = getAllFilesFromDir("$outputDir/authentication" , includeDir: false , ignore: unstable_patterns_auth , ignoreFile: null , include: ['*', '**/*'] ) + // def stable_name_authentication = getAllFilesFromDir("$outputDir/authentication" , includeDir: false , ignore: null , ignoreFile: null , include: unstable_patterns_auth) + + // // Deduplication + // def stable_content_deduplication = getAllFilesFromDir("$outputDir/deduplication" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.flagstat'] ) + // def stable_name_deduplication = getAllFilesFromDir("$outputDir/deduplication" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.{bam,bai}'] ) + + // // Final_bams + // def stable_content_final_bams = getAllFilesFromDir("$outputDir/final_bams" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.flagstat'] ) + // def stable_name_final_bams = getAllFilesFromDir("$outputDir/final_bams" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.{bam,bai}'] ) + + // // Mapping (incl. bam_input flasgstat) + // def stable_content_mapping = getAllFilesFromDir("$outputDir/mapping" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.flagstat'] ) + // def stable_name_mapping = getAllFilesFromDir("$outputDir/mapping" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.{bam,bai}'] ) + + // // Preprocessing + // // NOTE: FastQC html appears stable, but I worry it might just include a day timestamp instead of a full timestamp. To keep the expression simpler I removed both from checksum testing. + // def stable_content_preprocessing = getAllFilesFromDir("$outputDir/preprocessing" , includeDir: false , ignore: ['**/*.{zip,log,html}'], ignoreFile: null , include: ['**/*'] ) + // def stable_name_preprocessing = getAllFilesFromDir("$outputDir/preprocessing" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.{zip,log,html}'] ) + + // // Read filtering + // def stable_content_readfiltering = getAllFilesFromDir("$outputDir/read_filtering" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.flagstat'] ) + // def stable_name_readfiltering = getAllFilesFromDir("$outputDir/read_filtering" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.{bam,bai}'] ) + + // // Genotyping + // def stable_content_genotyping = getAllFilesFromDir("$outputDir/genotyping" , includeDir: false , ignore: ['**/*.{tbi,vcf.gz}'] , ignoreFile: null , include: ['**/*'] ) + // def stable_name_genotyping = getAllFilesFromDir("$outputDir/genotyping" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.tbi'] ) + // // We need to collect the vcfs separately to run more specific md5sum checks on the header (contnts are unstable due to same reasons as BAMs, explained above). + // def genotyping_vcfs = getAllFilesFromDir("$outputDir/genotyping" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.vcf.gz'] ) + + // // Metagenomics + // def stable_content_metagenomics = getAllFilesFromDir("$outputDir/metagenomics" , includeDir: false , ignore: ['**/*.biom', '**/*table.tsv'] , ignoreFile: null , include: ['**/*'] ) + // def stable_name_metagenomics = getAllFilesFromDir("$outputDir/metagenomics" , includeDir: false , ignore: null , ignoreFile: null , include: ['**/*.biom', '**/*table.tsv'] ) + + // MultiQC + // def stable_name_multiqc = getAllFilesFromDir("$outputDir/multiqc" , includeDir: false , ignore: null , ignoreFile: null , include: ['*', '**/*'] ) + + /////////////////////// + // DEFINE ASSERTIONS // + /////////////////////// + + assertAll( + { assert workflow.success }, + // This checks that there are no missing or additional output files. + // Also a good starting point to look at all the files in the output folder than need to be checked in subsequent sections. + { assert snapshot( stable_name_all*.name ).match("all_files") }, + + // Checking changes to contents of each section + // NOTE: Keep the order of the sections in the alphanumeric order of the output directories. + // Each section should first check stable_content, stable_name second (if applicable). + // { assert snapshot( stable_content_authentication , stable_name_authentication*.name ).match("authentication") }, + // { assert snapshot( stable_content_deduplication , stable_name_deduplication*.name ).match("deduplication") }, + // { assert snapshot( stable_content_final_bams , stable_name_final_bams*.name ).match("final_bams") }, + // // NOTE: The snapshot section for mapping cannot be named 'mapping'. See https://github.com/askimed/nf-test/issues/279 + // { assert snapshot( stable_content_mapping , stable_name_mapping*.name ).match("mapping_output") }, + // { assert snapshot( stable_content_preprocessing , stable_name_preprocessing*.name ).match("preprocessing") }, + // { assert snapshot( stable_content_readfiltering , stable_name_readfiltering*.name ).match("read_filtering") }, + // { assert snapshot( stable_content_genotyping , stable_name_genotyping*.name ).match("genotyping") }, + // // Additional checks on the genotyping VCFs for content. Specifically the md5sums of the header FORMAT, INFO, FILTER, CONTIG lines, and sample names + // { assert snapshot( + // genotyping_vcfs.collect { + // file -> + // def vcf_head = path(file.toString()).vcf.header + // // The header contains lines in the "OTHER" category, which contain a timestamp and/or work dir paths, so we need to filter those out, then calculate md5sums. + // def header_md5 = [ + // vcf_head.getFormatHeaderLines().toString(), + // vcf_head.getInfoHeaderLines().toString(), + // vcf_head.getFilterLines().toString(), + // vcf_head.getIDHeaderLines().toString(), + // vcf_head.getGenotypeSamples().toString(), + // vcf_head.getContigLines().toString(), + // ].join(' ').md5() + // file.getName() + ":header_md5," + header_md5 + // } + // ).match("genotyping_vcfs")}, + // { assert snapshot( stable_content_metagenomics , stable_name_metagenomics*.name ).match("metagenomics") }, + // { assert snapshot( stable_name_multiqc*.name ).match("multiqc") }, + + // Versions + { assert new File("$outputDir/pipeline_info/nf_core_eager_software_mqc_versions.yml").exists() }, + + ) + } + } +} diff --git a/tests/test_full.nf.test.snap b/tests/test_full.nf.test.snap new file mode 100644 index 00000000..38457193 --- /dev/null +++ b/tests/test_full.nf.test.snap @@ -0,0 +1,77 @@ +{ + "all_files": { + "content": [ + [ + "COD076_COD076E1bL1_GCF_902167405.1_gadMor3.0_rna.c_curve.txt", + "COD076_COD076E1bL1_GCF_902167405.1_gadMor3.0_rna.command.log", + "COD092_COD092E1bL1i69_GCF_902167405.1_gadMor3.0_rna.c_curve.txt", + "COD092_COD092E1bL1i69_GCF_902167405.1_gadMor3.0_rna.command.log", + "COD076_COD076E1bL1_GCF_902167405.1_gadMor3.0_rna_sorted.bam", + "COD076_COD076E1bL1_GCF_902167405.1_gadMor3.0_rna_sorted.bam.bai", + "COD092_COD092E1bL1i69_GCF_902167405.1_gadMor3.0_rna_sorted.bam", + "COD092_COD092E1bL1i69_GCF_902167405.1_gadMor3.0_rna_sorted.bam.bai", + "COD076_COD076E1bL1_GCF_902167405.1_gadMor3.0_rna_sorted.flagstat", + "COD092_COD092E1bL1i69_GCF_902167405.1_gadMor3.0_rna_sorted.flagstat", + "COD076_COD076E1bL1_L1.fastp.html", + "COD076_COD076E1bL1_L1.fastp.json", + "COD076_COD076E1bL1_L1.fastp.log", + "COD076_COD076E1bL1_L6.fastp.html", + "COD076_COD076E1bL1_L6.fastp.json", + "COD076_COD076E1bL1_L6.fastp.log", + "COD076_COD076E1bL1_L8.fastp.html", + "COD076_COD076E1bL1_L8.fastp.json", + "COD076_COD076E1bL1_L8.fastp.log", + "COD092_COD092E1bL1i69_L6.fastp.html", + "COD092_COD092E1bL1i69_L6.fastp.json", + "COD092_COD092E1bL1i69_L6.fastp.log", + "COD092_COD092E1bL1i69_L7.fastp.html", + "COD092_COD092E1bL1i69_L7.fastp.json", + "COD092_COD092E1bL1i69_L7.fastp.log", + "COD092_COD092E1bL1i69_L8.fastp.html", + "COD092_COD092E1bL1i69_L8.fastp.json", + "COD092_COD092E1bL1i69_L8.fastp.log", + "COD076_COD076E1bL1_L1_fastqc.html", + "COD076_COD076E1bL1_L1_fastqc.zip", + "COD076_COD076E1bL1_L6_fastqc.html", + "COD076_COD076E1bL1_L6_fastqc.zip", + "COD076_COD076E1bL1_L8_fastqc.html", + "COD076_COD076E1bL1_L8_fastqc.zip", + "COD092_COD092E1bL1i69_L6_fastqc.html", + "COD092_COD092E1bL1i69_L6_fastqc.zip", + "COD092_COD092E1bL1i69_L7_fastqc.html", + "COD092_COD092E1bL1i69_L7_fastqc.zip", + "COD092_COD092E1bL1i69_L8_fastqc.html", + "COD092_COD092E1bL1i69_L8_fastqc.zip", + "COD076_COD076E1bL1_L1_1_fastqc.html", + "COD076_COD076E1bL1_L1_1_fastqc.zip", + "COD076_COD076E1bL1_L1_2_fastqc.html", + "COD076_COD076E1bL1_L1_2_fastqc.zip", + "COD076_COD076E1bL1_L6_1_fastqc.html", + "COD076_COD076E1bL1_L6_1_fastqc.zip", + "COD076_COD076E1bL1_L6_2_fastqc.html", + "COD076_COD076E1bL1_L6_2_fastqc.zip", + "COD076_COD076E1bL1_L8_1_fastqc.html", + "COD076_COD076E1bL1_L8_1_fastqc.zip", + "COD076_COD076E1bL1_L8_2_fastqc.html", + "COD076_COD076E1bL1_L8_2_fastqc.zip", + "COD092_COD092E1bL1i69_L6_1_fastqc.html", + "COD092_COD092E1bL1i69_L6_1_fastqc.zip", + "COD092_COD092E1bL1i69_L6_2_fastqc.html", + "COD092_COD092E1bL1i69_L6_2_fastqc.zip", + "COD092_COD092E1bL1i69_L7_1_fastqc.html", + "COD092_COD092E1bL1i69_L7_1_fastqc.zip", + "COD092_COD092E1bL1i69_L7_2_fastqc.html", + "COD092_COD092E1bL1i69_L7_2_fastqc.zip", + "COD092_COD092E1bL1i69_L8_1_fastqc.html", + "COD092_COD092E1bL1i69_L8_1_fastqc.zip", + "COD092_COD092E1bL1i69_L8_2_fastqc.html", + "COD092_COD092E1bL1i69_L8_2_fastqc.zip" + ] + ], + "meta": { + "nf-test": "0.9.3", + "nextflow": "25.10.2" + }, + "timestamp": "2026-01-23T04:03:22.466351835" + } +} \ No newline at end of file