comparison mitokmer.xml @ 5:8f4abeb27625 draft

planemo upload commit 3118e6e386354d664c85c47c67b75a92629d358c
author galaxytrakr
date Tue, 15 Sep 2026 18:31:11 +0000
parents ecf96ad08611
children e4d4fc748584
comparison
equal deleted inserted replaced
4:ecf96ad08611 5:8f4abeb27625
1 <tool id="mitokmer" name="mitoKmer" version="2.0+galaxy0.2" python_template_version="3.5" profile="21.05"> 1 <tool id="mitokmer" name="mitoKmer" version="2.0+galaxy0.2" profile="21.05">
2 <description>Identify metagenomic mitochondrial reads by k-mer database matching</description> 2 <description>Identify metagenomic mitochondrial reads by k-mer database matching</description>
3 <requirements> 3 <requirements>
4 <container type="docker">quay.io/galaxytrakr/mitokmer:e57a559</container> 4 <container type="docker">quay.io/galaxytrakr/mitokmer:e57a559</container>
5 </requirements> 5 </requirements>
6 6
7 <command detect_errors="exit_code"><![CDATA[ 7 <command detect_errors="exit_code"><![CDATA[
8 ## ── All paths are hardcoded in kmer_read_m7.py: ────────────────────── 8 #import re
9 ## database: ./mitochondria7/<multiple .txt files> 9 #def fix_ext($ext):
10 ## jobs file: ./jobs7m/jobs7m.txt 10 #set ext = $ext.replace('fastqsanger', 'fastq')
11 ## output: ./jobs7m/jobs7m.csv 11 #set ext = $ext.replace('fastqillumina', 'fastq')
12 ## binary: ./kmerread7 (called as subprocess by the Python script) 12 #return $ext
13 #end def
14
13 mkdir -p ./mitochondria7 ./jobs7m && 15 mkdir -p ./mitochondria7 ./jobs7m &&
14 16
15 ## ── Link database files from the data table directory ───────────────── 17 ## ── Link database files into working directory ────────────────────────
16 ## Symlink each file individually from the registered database directory.
17 ## Note: glob must be outside quotes so the shell expands it correctly.
18 ln -sf '${probe_db.fields.path}'/* ./mitochondria7/ && 18 ln -sf '${probe_db.fields.path}'/* ./mitochondria7/ &&
19 19
20 ## ── Symlink kmerread7 into the working directory ────────────────────── 20 ## ── Symlink kmerread7 binary into working directory ───────────────────
21 ## kmer_read_m7.py calls ./kmerread7 (relative path), so it must exist
22 ## in the Galaxy job working directory
23 ln -sf /usr/local/bin/kmerread7 ./kmerread7 && 21 ln -sf /usr/local/bin/kmerread7 ./kmerread7 &&
24 22
25 ## ── Write the jobs file ─────────────────────────────────────────────── 23 ## ── Stage input reads and build jobs file ────────────────────────────
26 ## Format: sample_name <tab> num_files 24 #if $reads.reads_type == "single"
27 ## /abs/path/to/read1 25 mkdir -p ./reads &&
28 ## /abs/path/to/read2 ... 26 #set sample_name = $reads.input.name.replace(' ', '_')
29 ## Use len() to safely get collection size as an integer 27 #set read_count = 0
30 #set sample_name = $reads.name.replace(' ', '_') 28 #for $read in $reads.input
31 #set read_count = len($reads) 29 #set ext = $fix_ext($read.ext)
32 printf '%s\t%d\n' '${sample_name}' ${read_count} > ./jobs7m/jobs7m.txt && 30 #set fname = re.sub('[^\w\-_.]', '_', $read.element_identifier) + '.' + $ext
33 #for read in $reads 31 ln -sf '$read' './reads/${fname}' &&
34 printf '%s\n' '${read.file_name}' >> ./jobs7m/jobs7m.txt && 32 #set read_count = $read_count + 1
35 #end for 33 #end for
36 34 printf '%s\t%d\n' '${sample_name}' ${read_count} > ./jobs7m/jobs7m.txt &&
37 ## ── Run the Python orchestrator (no arguments) ──────────────────────── 35 #for $read in $reads.input
38 ## kmer_read_m7.py reads jobs7m/jobs7m.txt, calls ./kmerread7, 36 #set ext = $fix_ext($read.ext)
39 ## and writes results to jobs7m/jobs7m.csv 37 #set fname = re.sub('[^\w\-_.]', '_', $read.element_identifier) + '.' + $ext
38 printf '%s\n' './reads/${fname}' >> ./jobs7m/jobs7m.txt &&
39 #end for
40 #else
41 ## paired collection: forward + reverse
42 mkdir -p ./reads &&
43 #set sample_name = $reads.input.name.replace(' ', '_')
44 #set fwd_ext = $fix_ext($reads.input.forward.ext)
45 #set rev_ext = $fix_ext($reads.input.reverse.ext)
46 #set base = re.sub('[^\w\-_.]', '_', $reads.input.element_identifier)
47 #set fwd_fname = $base + '_1.' + $fwd_ext
48 #set rev_fname = $base + '_2.' + $rev_ext
49 ln -sf '$reads.input.forward' './reads/${fwd_fname}' &&
50 ln -sf '$reads.input.reverse' './reads/${rev_fname}' &&
51 printf '%s\t2\n' '${sample_name}' > ./jobs7m/jobs7m.txt &&
52 printf '%s\n' './reads/${fwd_fname}' >> ./jobs7m/jobs7m.txt &&
53 printf '%s\n' './reads/${rev_fname}' >> ./jobs7m/jobs7m.txt &&
54 #end if
55
56 ## ── Run the Python orchestrator ───────────────────────────────────────
40 python3 /opt/mitokmer2/kmer_read_m7.py && 57 python3 /opt/mitokmer2/kmer_read_m7.py &&
41 58
42 ## ── Copy output CSV to Galaxy output path ───────────────────────────── 59 ## ── Copy output CSV to Galaxy output path ─────────────────────────────
43 cp ./jobs7m/jobs7m.csv '${results_csv}' 60 cp ./jobs7m/jobs7m.csv '${results_csv}'
44 ]]></command> 61 ]]></command>
45 62
46 <inputs> 63 <inputs>
47 <!-- Probe database directory selected from Galaxy data table --> 64 <param name="probe_db" type="select" label="Mitochondrial k-mer probe database"
48 <param name="probe_db"
49 type="select"
50 label="Mitochondrial k-mer probe database"
51 help="Select a pre-installed mitochondrial k-mer probe database. 65 help="Select a pre-installed mitochondrial k-mer probe database.
52 Databases are managed by your Galaxy administrator via the 66 Databases are managed by your Galaxy administrator via the
53 mitokmer_probe_db data table."> 67 mitokmer_probe_db data table.">
54 <options from_data_table="mitokmer_probe_db"> 68 <options from_data_table="mitokmer_probe_db">
55 <filter type="sort_by" column="1" /> 69 <filter type="sort_by" column="1" />
56 <validator type="no_options" 70 <validator type="no_options" message="No mitochondrial k-mer probe databases are
57 message="No mitochondrial k-mer probe databases are 71 currently installed. Please contact your Galaxy administrator." />
58 currently installed. Please contact your
59 Galaxy administrator." />
60 </options> 72 </options>
61 </param> 73 </param>
62 74
63 <param name="reads" 75 <conditional name="reads">
64 type="data_collection" 76 <param name="reads_type" type="select" label="Input read type">
65 collection_type="list" 77 <option value="single">Single-end or unpaired reads / FASTA (list collection)</option>
66 format="fastq,fastq.gz,fasta,fasta.gz" 78 <option value="paired">Paired-end reads (paired collection)</option>
67 label="Input reads (FASTQ or FASTA, gzipped or plain)" 79 </param>
68 help="Provide one or more read files for a single sample as a Galaxy 80 <when value="single">
69 list collection. Paired-end files (R1 + R2) should both be 81 <param name="input" type="data_collection" collection_type="list"
70 included in the same collection. Light quality trimming of 82 format="fastqsanger,fastqsanger.gz,fastqillumina,fastqillumina.gz,fasta,fasta.gz"
71 read ends is performed internally; pre-trimming is optional." /> 83 label="Input reads (FASTQ or FASTA, gzipped or plain)"
84 help="Provide one or more single-end or unpaired read files as a Galaxy
85 list collection. For paired-end data, use the 'Paired-end reads'
86 option instead." />
87 </when>
88 <when value="paired">
89 <param name="input" type="data_collection" collection_type="paired"
90 format="fastqsanger,fastqsanger.gz,fastqillumina,fastqillumina.gz"
91 label="Paired-end reads (paired collection)"
92 help="Provide a Galaxy paired collection containing forward (R1) and
93 reverse (R2) reads. Light quality trimming of read ends is
94 performed internally; pre-trimming is optional." />
95 </when>
96 </conditional>
72 </inputs> 97 </inputs>
73 98
74 <outputs> 99 <outputs>
75 <data name="results_csv" 100 <data name="results_csv" format="csv" label="mitoKmer results for ${on_string}" />
76 format="csv"
77 label="mitoKmer results for ${on_string}" />
78 </outputs> 101 </outputs>
79 102
80 <tests> 103 <tests>
104 <!-- Test 1: paired-end FASTQ via paired collection -->
81 <test> 105 <test>
82 <param name="probe_db" value="mitoch_probes_sample" /> 106 <param name="probe_db" value="mitoch_probes_sample" />
83 <param name="reads"> 107 <conditional name="reads">
84 <collection type="list"> 108 <param name="reads_type" value="paired" />
85 <element name="Plodia_R1" value="test/Plodia_R1.fastq.gz" /> 109 <param name="input">
86 <element name="Plodia_R2" value="test/Plodia_R2.fastq.gz" /> 110 <collection type="paired">
87 </collection> 111 <element name="forward" value="test/Plodia_R1.fastq.gz" ftype="fastqsanger.gz" />
88 </param> 112 <element name="reverse" value="test/Plodia_R2.fastq.gz" ftype="fastqsanger.gz" />
113 </collection>
114 </param>
115 </conditional>
116 <output name="results_csv">
117 <assert_contents>
118 <has_text text="Plodia" />
119 </assert_contents>
120 </output>
121 </test>
122 <!-- Test 2: single-end FASTQ via list collection -->
123 <test>
124 <param name="probe_db" value="mitoch_probes_sample" />
125 <conditional name="reads">
126 <param name="reads_type" value="single" />
127 <param name="input">
128 <collection type="list">
129 <element name="Plodia_R1" value="test/Plodia_R1.fastq.gz" ftype="fastqsanger.gz" />
130 </collection>
131 </param>
132 </conditional>
133 <output name="results_csv">
134 <assert_contents>
135 <has_text text="Plodia" />
136 </assert_contents>
137 </output>
138 </test>
139 <!-- Test 3: single FASTA via list collection -->
140 <test>
141 <param name="probe_db" value="mitoch_probes_sample" />
142 <conditional name="reads">
143 <param name="reads_type" value="single" />
144 <param name="input">
145 <collection type="list">
146 <element name="Plodia_assembly" value="test/Plodia_assembly.fasta" ftype="fasta" />
147 </collection>
148 </param>
149 </conditional>
89 <output name="results_csv"> 150 <output name="results_csv">
90 <assert_contents> 151 <assert_contents>
91 <has_text text="Plodia" /> 152 <has_text text="Plodia" />
92 </assert_contents> 153 </assert_contents>
93 </output> 154 </output>
100 161
101 Overview 162 Overview
102 -------- 163 --------
103 mitoKmer identifies the taxonomic origin of short-read shotgun sequencing data 164 mitoKmer identifies the taxonomic origin of short-read shotgun sequencing data
104 by matching reads against a database of species-specific mitochondrial k-mer 165 by matching reads against a database of species-specific mitochondrial k-mer
105 probes. It reports the relative abundance of taxa at multiple taxonomic ranks 166 probes. It reports the relative abundance of taxa at multiple taxonomic ranks
106 (e.g. order, species) along with the number of matching reads and supporting 167 (e.g. order, species) along with the number of matching reads and supporting
107 unique k-mers. 168 unique k-mers.
108 169
109 Inputs 170 Inputs
110 ------ 171 ------
111 **Mitochondrial k-mer probe database** 172 **Mitochondrial k-mer probe database**
112 Select a pre-installed probe database from the dropdown. Databases are 173 Select a pre-installed probe database from the dropdown. Databases are
113 managed by your Galaxy administrator and registered in the 174 managed by your Galaxy administrator and registered in the
114 ``mitokmer_probe_db`` data table. The database consists of a directory 175 ``mitokmer_probe_db`` data table. The database consists of a directory
115 of supporting ``.txt`` files. 176 of ``.txt`` files.
116 177
117 **Input reads** 178 **Input read type**
118 One or more FASTQ or FASTA files (gzipped or plain) for a single sample, 179 Choose the input mode that matches your data:
119 supplied as a Galaxy list collection. Both paired-end files (R1 and R2) 180
120 should be included in the same collection. 181 *Single-end or unpaired reads / FASTA (list collection)*
121 182 Provide a Galaxy **list** collection containing one or more FASTQ
122 * Pre-trimming is optional — the tool performs simple window-based quality 183 (``fastqsanger``, ``fastqsanger.gz``) or FASTA (``fasta``,
123 trimming of read ends internally. 184 ``fasta.gz``) files. Use this for single-end sequencing data or
185 assembled FASTA sequences.
186
187 *Paired-end reads (paired collection)*
188 Provide a Galaxy **paired** collection where the forward (R1) and
189 reverse (R2) reads are paired together. Both files must be FASTQ
190 format (``fastqsanger`` or ``fastqsanger.gz``).
191
192 In all cases, light window-based quality trimming of read ends is
193 performed internally; pre-trimming is optional.
124 194
125 Output 195 Output
126 ------ 196 ------
127 **Results CSV** 197 **Results CSV**
128 A comma-separated file with columns: taxid, reads, abundance, uniq. 198 A comma-separated file with one row per taxon detected. Columns:
199
200 ========= ============================================================
201 Column Description
202 ========= ============================================================
203 Rank Taxonomic rank (e.g. Order, Species)
204 Taxon Taxon name
205 Reads Number of reads matching this taxon
206 Rel_Abund Relative abundance (%)
207 Unique_kmers Number of unique k-mers supporting the match
208 ========= ============================================================
209
129 Only taxa with non-zero relative abundance are reported. 210 Only taxa with non-zero relative abundance are reported.
130 211
131 Example output for the "Plodia" test sample:: 212 Example output for the "Plodia" test sample::
132 213
133 Rank Taxon Reads Rel_Abund(%) Unique_kmers 214 Rank Taxon Reads Rel_Abund(%) Unique_kmers