-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathparameters.config.template
More file actions
274 lines (244 loc) · 13.1 KB
/
Copy pathparameters.config.template
File metadata and controls
274 lines (244 loc) · 13.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
// PoolSeqFlow parameters. The manual explains every setting in this file:
// https://ozankiratli.github.io/PoolSeqFlow/
//
// Configure runs here and nowhere else. Nextflow passes command-line parameters as strings,
// so `--annotate false` sets the string "false", which evaluates as true.
params {
//main parameters
threads = 8 // cores per task
memory = '24 GB' // memory ceiling per task
// Your project directory, and the one you launch from. The faster volume. It may be
// neither the installation nor storageDir.
mainDir = "/path/to/project/directory"
// Permanent storage, and a slower volume will do. Called projectDir before 3.0;
// `./PoolSeqFlow migrate_config` carries the old value over.
storageDir = "/path/to/permanent/storage"
// The files you place, all relative to mainDir:
// mainDir/Data/ the reads, in it or in subfolders of it (dataSource names it)
// mainDir/Reference/ reference + annotation (referenceFile, gffFile)
// mainDir/metadata.csv what your samples ARE (metadataFile)
dataSource = 'Data'
readPattern = "*_R{1,2}.fq.gz"
referenceFile = 'reference.fasta.gz'
poolSize = 100
ploidy = 2
// One row per pair of FASTQ files. metadata.csv.template explains the columns; it
// replaces RGTags.csv.
metadataFile = 'metadata.csv'
annotate = true
gffFile = 'reference.gff.gz'
// Several runs from one invocation - the same reads against two references, say. true
// reads multiRunFile below: a RunID column, plus one column per parameter that differs
// from the values above. See multi-run.csv.example in the installation.
multiRun = false
multiRunFile = 'runs.csv'
referenceFa = "${params.referenceFile.replace('.gz', '')}"
// Core allocation, computed from `threads` and left commented out so that changing
// `threads` moves all of it together. Uncomment any one to pin it; a pinned value still
// feeds the ones computed from it.
cores {
// ladder = largest power of two <= threads, capped at 8
// trim = largest N whose N+4 fits in threads: 12+ -> 8, 8+ -> 4, 6+ -> 2, else 1
// trimTotal = trim > 1 ? trim + 4 : 1 threads Trim Galore actually spawns
// bwa = ladder bwa mem -t N
// cutadapt = ladder cutadapt --cores N
// fastqc = threads >= 2 ? 2 : 1 one thread per file
// samtools = threads >= 2 ? 1 : 0 -@ counts ADDITIONAL threads
// javaGc = samtools + 1 -XX:ParallelGCThreads is a total
}
// java parameters. -XX:ParallelGCThreads is set per process from task.cpus.
java {
heapSize = '-Xmx8g'
}
// fastqc parameters
fastqc {
memory = 2048 // megabytes, as a plain number - FastQC rejects "2G"
// Uncomment to take full control; `memory` is then unused. -t comes from task.cpus.
// options = "--memory ${params.fastqc.memory}"
}
// trim_galore parameters
trim_galore {
quality = 25
// true -> no adapter is passed; Trim Galore auto-detects it (Illumina/Nextera/smallRNA)
// and adapter1/adapter2 below are ignored.
// false -> BOTH adapter1 and adapter2 must be set to your provider's sequences.
autodetect = true
adapter1 = ''
adapter2 = ''
// Built from the values above; uncomment either to take full control of it.
// --cores and --fastqc_args come from task.cpus.
// adapterOptions = autodetect ? '' : "-a ${params.trim_galore.adapter1} -a2 ${params.trim_galore.adapter2}"
// options = "--fastqc --paired --retain_unpaired -q ${params.trim_galore.quality} ${params.trim_galore.adapterOptions}"
}
cutadapt {
at_gc_error = 0.025
min_length = 50 // inert on its own - see the options lines below
// --cores is supplied by ClipReads from task.cpus
options = ""
// To apply min_length, comment the line above and uncomment the one below. ClipReads
// also passes -l, a limit it computes from the FastQC report, and cutadapt applies -m
// AFTER -l: a min_length above that limit discards every read and still exits 0.
// options = "-m ${params.cutadapt.min_length}" // discard reads shorter than min_length after clipping
}
// BWA parameters
bwa {
minScoreOutput = 30
batchSize = 10000000
// Built from the two values above; uncomment to take full control. -t comes from
// task.cpus.
// options = "-K ${params.bwa.batchSize} -T ${params.bwa.minScoreOutput}"
}
// Step 4: what samtools keeps when it cleans each BAM.
cleanBAM {
filter = "0xF0C" // See `man samtools-flags` for more information
required = "0x2" // See `man samtools-flags` for more information
mapq = 30
}
// The depth ceiling decided at step 5 and applied to each BAM at step 6, before calling.
// -1 measures one per sample from its own depth histogram, a positive number caps every
// sample there, 0 caps nothing.
capBAM {
maxDepth = -1
// How deep step 5's depth histogram looks. A sample with positions deeper than this
// stops the run naming this parameter: a ceiling read from a truncated histogram is
// read from a partial picture. Raising it cannot change a result and is not compared
// against previous runs, so it never costs a reset.
histogramMax = 100000
}
// Step 6: how bcftools calls variants.
variantCall {
scaleMapQ = 50
baseQualMin = 30
varQualMin = 30
// mpileup's own ceiling, on top of whatever capBAM applied. Here 0 is no limit, and
// only capBAM.maxDepth takes -1.
maxDepth = 0
// Built from the four values above; uncomment to take full control of them, maxDepth
// included. -q is mapping quality, -Q is base quality.
// mpileupOptions = "-B -C ${params.variantCall.scaleMapQ} -q ${params.variantCall.varQualMin} -Q ${params.variantCall.baseQualMin} -d ${params.variantCall.maxDepth} -a AD,DP,SP,INFO/AD -Ou"
callOptions = "-m -A -v -Ov"
}
// VCF parameters
vcf {
fileName = 'Test'
}
// vcf Filtering parameters
vcffilter {
minDP = 20
minQUAL = 30
}
// Filter False Positives parameters
filterFalsePositives {
sensitivity = 1.0 / (2 * params.ploidy * params.poolSize)
sampleThreshold = 0.2
}
// SNPEff parameters
snpEff {
// Derived from the GFF file name ([reference].gff.gz -> [reference].gff).
db = "${params.gffFile.replace('.gz', '')}"
config = "snpEff.config"
buildOptions = "-gff3 -noCheckCds -noCheckProtein -v"
runOptions = "-v"
}
// Directories.
//
// Everything the run reads repeatedly is on mainDir; everything it has finished with is
// on storageDir.
dir {
// Placed by you, and never moved or deleted by the pipeline.
data = "${params.mainDir}/${params.dataSource}"
references = "${params.mainDir}/Reference"
// Built from your reference by step 1. Yours to keep, the pipeline's to rebuild.
dictionaries = "${params.dir.references}/Dictionaries"
// Outputs land here first and move to Output/ once whatever reads them has succeeded.
utilized = "${params.mainDir}/Utilized"
outputs = "${params.storageDir}/Output"
// Under multiRun, Output/ and Logs/ are divided into All_Runs/, Shared_<N>/ and
// <RunID>/. A single run sits at Output/ directly.
allOutputs = params.multiRun ? "${params.storageDir}/Output/All_Runs" : "${params.storageDir}/Output"
allLogs = params.multiRun ? "${params.storageDir}/Logs/All_Runs" : "${params.storageDir}/Logs"
// The installation's own, under neither storage root. `projectDir` is Nextflow's
// variable, not a parameter of ours.
bin = "${projectDir}/bin"
// Sourced by another script, never run, so it is not on PATH.
lib = "${projectDir}/lib"
logs = "${params.storageDir}/Logs"
snpEff = "${params.dir.dictionaries}/snpEff"
// Relative to EITHER root: Utilized/ mirrors Output/ exactly.
subpath {
aligned = "Aligned"
ready = "Ready"
trimmed = "Trimmed"
unpaired = "Unpaired"
vcf = "VCF"
freq = "Frequencies"
reports = "Reports"
report {
align = "${params.dir.subpath.reports}/Alignment"
coverage = "${params.dir.subpath.reports}/Coverage"
depth = "${params.dir.subpath.reports}/Depth"
fastqc = "${params.dir.subpath.reports}/Fastqc"
trim = "${params.dir.subpath.reports}/Trimming"
}
}
// Permanent storage: where an artifact ends up.
output {
aligned = "${params.dir.outputs}/${params.dir.subpath.aligned}"
ready = "${params.dir.outputs}/${params.dir.subpath.ready}"
trimmed = "${params.dir.outputs}/${params.dir.subpath.trimmed}"
unpaired = "${params.dir.outputs}/${params.dir.subpath.unpaired}"
vcf = "${params.dir.outputs}/${params.dir.subpath.vcf}"
freq = "${params.dir.outputs}/${params.dir.subpath.freq}"
reports = "${params.dir.outputs}/${params.dir.subpath.reports}"
report {
align = "${params.dir.outputs}/${params.dir.subpath.report.align}"
coverage = "${params.dir.outputs}/${params.dir.subpath.report.coverage}"
depth = "${params.dir.outputs}/${params.dir.subpath.report.depth}"
fastqc = "${params.dir.outputs}/${params.dir.subpath.report.fastqc}"
trim = "${params.dir.outputs}/${params.dir.subpath.report.trim}"
}
}
// Nextflow's own dag, trace, timeline and report. Last in this block, and it has to
// be: interpolation is eager and top to bottom, so a reference to `subpath` from above
// it reads null and the config fails to parse.
sessionReports = "${params.dir.allOutputs}/${params.dir.subpath.reports}"
}
// Full paths to the files you place. Below the dir block, not beside the file names
// above: interpolation is a single forward pass, and a forward reference fails to parse.
referencePath = "${params.dir.references}/${params.referenceFile}"
gffPath = "${params.dir.references}/${params.gffFile}"
metadataPath = "${params.mainDir}/${params.metadataFile}"
multiRunPath = "${params.mainDir}/${params.multiRunFile}"
// The decompressed reference, written by step 1. There is no `gff` beside it: snpEff
// builds its database from the file you placed, gzipped or not.
reference = "${params.dir.dictionaries}/${params.referenceFile.replace('.gz', '')}"
// The `**` is what lets the reads sit in subfolders of Data/ as well as directly in it -
// one folder per sample, one per sequencing run, or none at all. It matches across
// directories INCLUDING none, which `**/` does not: `**/` finds only the nested ones.
reads = "${params.dir.data}/**${params.readPattern}"
// How each tool is invoked. Replace a command with a full path to use a system binary
// instead of the conda environment's - at your own risk, as versions may conflict.
software {
java = 'java'
cutadapt = 'cutadapt'
fastqc = 'fastqc'
trim_galore = 'trim_galore'
samtools = 'samtools'
bamtools = 'bamtools'
bwa = 'bwa'
bcftools = 'bcftools'
vcftools = 'vcftools'
snpEff = 'snpEff'
unzip = 'unzip'
// Moving a finished artifact into permanent storage. bin/atomic_mv.sh copies it
// with rsync, compares the copy against its source with diff, and only then
// removes the source; find lists what a results folder holds.
//
// These three are verified at the start of a run like the rest, but they are the
// one group a full path above does not repoint: they are called by name from the
// helper scripts in bin/, which read no Nextflow settings.
rsync = 'rsync'
diff = 'diff'
find = 'find'
}
}