-
Notifications
You must be signed in to change notification settings - Fork 8
Expand file tree
/
Copy pathschema_input_genome.json
More file actions
215 lines (215 loc) · 10.7 KB
/
Copy pathschema_input_genome.json
File metadata and controls
215 lines (215 loc) · 10.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://raw.githubusercontent.com/nf-core/seqsubmit/main/assets/schema_input_genome.json",
"title": "nf-core/seqsubmit pipeline - params.input schema",
"description": "Schema for the file provided with params.input if params.mode is set to 'mags' or 'bins'",
"type": "array",
"items": {
"type": "object",
"properties": {
"id": {
"type": "string",
"pattern": "^\\S+$",
"description": "A unique identifier for this data entry. Must be globally unique within the input dataset",
"errorMessage": "ID must be provided and cannot contain spaces",
"meta": ["id"]
},
"fasta": {
"type": "string",
"format": "file-path",
"exists": true,
"pattern": "^([\\S\\s]*\\/)?[^\\s\\/]+\\.(fa|fasta|fna)\\.gz$",
"errorMessage": "Path to the FASTA file must be provided, cannot contain spaces and must have extension '.fa.gz', '.fasta.gz', or '.fna.gz'",
"description": "Path to MAG/bin sequences in FASTA format compressed with `gzip`. All names of the FASTA files must be unique to prevent pipeline errors. File should contain at least one sequence"
},
"accession": {
"type": "string",
"description": "ENA accession of the run or metagenomic assembly used to generate the MAG/bin. Reads or assembly must already be submitted to ENA",
"meta": ["accession"]
},
"fastq_1": {
"anyOf": [
{
"type": "string",
"format": "file-path",
"exists": true,
"pattern": "^\\S+\\.(fq|fastq)(\\.gz)?$"
},
{
"type": "string",
"maxLength": 0
}
],
"default": null,
"errorMessage": "FASTQ file must have extension '.fq' or '.fastq' (optionally gzipped)",
"description": "Path to single-end/paired-end forward reads (in FASTQ format, optionally gzipped) used to generate the source metagenomic assembly. Required if genome_coverage is not provided"
},
"fastq_2": {
"anyOf": [
{
"type": "string",
"format": "file-path",
"exists": true,
"pattern": "^\\S+\\.(fq|fastq)(\\.gz)?$"
},
{
"type": "string",
"maxLength": 0
}
],
"default": null,
"errorMessage": "FASTQ file for reverse reads must have extension '.fq' or '.fastq' (optionally gzipped)",
"description": "Path to reverse reads in FASTQ format for paired-end data (optionally gzipped) used to generate the source metagenomic assembly. Leave empty for single-end reads"
},
"assembly_software": {
"type": "string",
"description": "Tool name and version that was used to assemble data, e.g. MEGAHIT_v1.0, SPAdes_v4.0.0, metaSPAdes_v3.15.0",
"meta": ["assembly_software"]
},
"binning_software": {
"type": "string",
"description": "Tool name and version that was used to bin data, e.g. MetaBAT2_v1.0, VAMB_2.0, MaxBin2_v1, CONCOCT_v3.0, SemiBin2_v1, COMEBin_v2.3.4, etc",
"meta": ["binning_software"]
},
"binning_parameters": {
"type": "string",
"description": "Arguments used to bin data different from default, e.g. -min_contig_length 1000, otherwise use 'default'",
"meta": ["binning_parameters"]
},
"stats_generation_software": {
"type": "string",
"default": null,
"description": "Tool, including version, that was used to calculate completeness and contamination, e.g. CheckM2_v1.0.0",
"meta": ["stats_generation_software"]
},
"completeness": {
"anyOf": [
{
"type": "number",
"minimum": 0,
"maximum": 100
},
{
"type": "string",
"maxLength": 0
}
],
"default": null,
"description": "MAG/bin completeness score: the 0<= ratio <= 100 of observed single-copy marker genes to total single-copy marker genes in chosen marker gene set (%). ENA docs: https://ena-docs.readthedocs.io/en/latest/faq/metagenomes.html#how-is-the-quality-of-a-metagenomic-assembly-defined",
"meta": ["completeness"]
},
"contamination": {
"anyOf": [
{
"type": "number",
"minimum": 0,
"maximum": 100
},
{
"type": "string",
"maxLength": 0
}
],
"default": null,
"description": "MAG/bin contamination score: the 0<= ratio <= 100 of observed single-copy marker genes in ≥2 copies to total single-copy marker genes in chosen marker gene set (%). ENA docs: https://ena-docs.readthedocs.io/en/latest/faq/metagenomes.html#how-is-the-quality-of-a-metagenomic-assembly-defined",
"meta": ["contamination"]
},
"genome_coverage": {
"anyOf": [
{
"type": "number",
"exclusiveMinimum": 0
},
{
"type": "string",
"maxLength": 0
}
],
"default": null,
"description": "Estimated average sequencing depth (>0) across the genome. If the value is missing, it is computed automatically during pipeline execution when reads are provided",
"meta": ["genome_coverage"]
},
"metagenome": {
"type": "string",
"description": "Registered metagenome taxonomic identifier or name that matches an existing ENA taxonomy entry. It needs to be listed in the taxonomy tree https://www.ebi.ac.uk/ena/browser/view/408169?show=tax-tree (you might need to press \"Tax tree - Show\" in the right most section of the page). Full list can also be found https://github.com/EBI-Metagenomics/genome_uploader/blob/main/genomeuploader/constants.py#L22",
"meta": ["metagenome"]
},
"co-assembly": {
"type": "boolean",
"description": "Whether a co-assembly strategy was used for the initial metagenomic assembly generation. Options: 'true' or 'false'",
"meta": ["co_assembly"]
},
"broad_environment": {
"type": "string",
"description": "Broad ecological context of the sample, for example 'marine biome', 'desert biome'. It is recommended to use subclasses of EnvO 'biome' class (http://purl.obolibrary.org/obo/ENVO_00000428). Documentation: https://github.com/EnvironmentOntology/envo/wiki/Using-ENVO-with-MIxS",
"meta": ["broad_environment"]
},
"local_environment": {
"type": "string",
"description": "Local environmental context of the sample, for example 'tropical dry broadleaf forest biome', 'marine abyssal zone biome'. It is recommended to use EnvO terms which are of smaller spatial grain than your entry for \"broad-scale environmental context\". Documentation: https://github.com/EnvironmentOntology/envo/wiki/Using-ENVO-with-MIxS",
"meta": ["local_environment"]
},
"environmental_medium": {
"type": "string",
"description": "Material surrounding and in contact with the sample, for example 'mucus', 'lake water'. It is recommended to use subclasses of EnvO 'environmental material' class (http://purl.obolibrary.org/obo/ENVO_00010483). Documentation: https://github.com/EnvironmentOntology/envo/wiki/Using-ENVO-with-MIxS",
"meta": ["environmental_medium"]
},
"RNA_presence": {
"anyOf": [
{
"type": "boolean"
},
{
"type": "string",
"maxLength": 0
}
],
"default": null,
"description": "Presence or absence of the 23S, 16S, and 5S rRNA genes and at least 18 tRNAs. This is used for MISAG/MIMAG assembly quality classification. Options: 'true' or 'false'",
"meta": ["RNA_presence"]
},
"NCBI_lineage": {
"type": "string",
"default": null,
"description": "Full NCBI lineage - format: x;y;z. For example, the lineage for E. coli can be: 'Bacteria;Pseudomonadati;Pseudomonadota;Gammaproteobacteria;Enterobacterales;Enterobacteriaceae;Escherichia' or '2;1224;1236;91347;543;561;562'. For more info check https://www.ncbi.nlm.nih.gov/datasets/docs/v2/data-processing/taxonomy-processing/taxonomy/",
"meta": ["NCBI_lineage"]
}
},
"required": [
"id",
"fasta",
"accession",
"assembly_software",
"co-assembly",
"binning_software",
"binning_parameters",
"metagenome",
"broad_environment",
"local_environment",
"environmental_medium"
],
"anyOf": [
{
"properties": {
"fastq_1": {
"type": "string",
"minLength": 1
}
},
"required": ["fastq_1"]
},
{
"properties": {
"genome_coverage": {
"type": "number",
"exclusiveMinimum": 0
}
},
"required": ["genome_coverage"]
}
],
"errorMessage": {
"anyOf": "Either reads or coverage must be provided in the sample sheet for each genome"
}
}
}