Mercurial > repos > galaxytrakr > mitokmer
comparison mitokmer.xml @ 5:8f4abeb27625 draft
planemo upload commit 3118e6e386354d664c85c47c67b75a92629d358c
| author | galaxytrakr |
|---|---|
| date | Tue, 15 Sep 2026 18:31:11 +0000 |
| parents | ecf96ad08611 |
| children | e4d4fc748584 |
comparison
equal
deleted
inserted
replaced
| 4:ecf96ad08611 | 5:8f4abeb27625 |
|---|---|
| 1 <tool id="mitokmer" name="mitoKmer" version="2.0+galaxy0.2" python_template_version="3.5" profile="21.05"> | 1 <tool id="mitokmer" name="mitoKmer" version="2.0+galaxy0.2" profile="21.05"> |
| 2 <description>Identify metagenomic mitochondrial reads by k-mer database matching</description> | 2 <description>Identify metagenomic mitochondrial reads by k-mer database matching</description> |
| 3 <requirements> | 3 <requirements> |
| 4 <container type="docker">quay.io/galaxytrakr/mitokmer:e57a559</container> | 4 <container type="docker">quay.io/galaxytrakr/mitokmer:e57a559</container> |
| 5 </requirements> | 5 </requirements> |
| 6 | 6 |
| 7 <command detect_errors="exit_code"><![CDATA[ | 7 <command detect_errors="exit_code"><![CDATA[ |
| 8 ## ── All paths are hardcoded in kmer_read_m7.py: ────────────────────── | 8 #import re |
| 9 ## database: ./mitochondria7/<multiple .txt files> | 9 #def fix_ext($ext): |
| 10 ## jobs file: ./jobs7m/jobs7m.txt | 10 #set ext = $ext.replace('fastqsanger', 'fastq') |
| 11 ## output: ./jobs7m/jobs7m.csv | 11 #set ext = $ext.replace('fastqillumina', 'fastq') |
| 12 ## binary: ./kmerread7 (called as subprocess by the Python script) | 12 #return $ext |
| 13 #end def | |
| 14 | |
| 13 mkdir -p ./mitochondria7 ./jobs7m && | 15 mkdir -p ./mitochondria7 ./jobs7m && |
| 14 | 16 |
| 15 ## ── Link database files from the data table directory ───────────────── | 17 ## ── Link database files into working directory ──────────────────────── |
| 16 ## Symlink each file individually from the registered database directory. | |
| 17 ## Note: glob must be outside quotes so the shell expands it correctly. | |
| 18 ln -sf '${probe_db.fields.path}'/* ./mitochondria7/ && | 18 ln -sf '${probe_db.fields.path}'/* ./mitochondria7/ && |
| 19 | 19 |
| 20 ## ── Symlink kmerread7 into the working directory ────────────────────── | 20 ## ── Symlink kmerread7 binary into working directory ─────────────────── |
| 21 ## kmer_read_m7.py calls ./kmerread7 (relative path), so it must exist | |
| 22 ## in the Galaxy job working directory | |
| 23 ln -sf /usr/local/bin/kmerread7 ./kmerread7 && | 21 ln -sf /usr/local/bin/kmerread7 ./kmerread7 && |
| 24 | 22 |
| 25 ## ── Write the jobs file ─────────────────────────────────────────────── | 23 ## ── Stage input reads and build jobs file ──────────────────────────── |
| 26 ## Format: sample_name <tab> num_files | 24 #if $reads.reads_type == "single" |
| 27 ## /abs/path/to/read1 | 25 mkdir -p ./reads && |
| 28 ## /abs/path/to/read2 ... | 26 #set sample_name = $reads.input.name.replace(' ', '_') |
| 29 ## Use len() to safely get collection size as an integer | 27 #set read_count = 0 |
| 30 #set sample_name = $reads.name.replace(' ', '_') | 28 #for $read in $reads.input |
| 31 #set read_count = len($reads) | 29 #set ext = $fix_ext($read.ext) |
| 32 printf '%s\t%d\n' '${sample_name}' ${read_count} > ./jobs7m/jobs7m.txt && | 30 #set fname = re.sub('[^\w\-_.]', '_', $read.element_identifier) + '.' + $ext |
| 33 #for read in $reads | 31 ln -sf '$read' './reads/${fname}' && |
| 34 printf '%s\n' '${read.file_name}' >> ./jobs7m/jobs7m.txt && | 32 #set read_count = $read_count + 1 |
| 35 #end for | 33 #end for |
| 36 | 34 printf '%s\t%d\n' '${sample_name}' ${read_count} > ./jobs7m/jobs7m.txt && |
| 37 ## ── Run the Python orchestrator (no arguments) ──────────────────────── | 35 #for $read in $reads.input |
| 38 ## kmer_read_m7.py reads jobs7m/jobs7m.txt, calls ./kmerread7, | 36 #set ext = $fix_ext($read.ext) |
| 39 ## and writes results to jobs7m/jobs7m.csv | 37 #set fname = re.sub('[^\w\-_.]', '_', $read.element_identifier) + '.' + $ext |
| 38 printf '%s\n' './reads/${fname}' >> ./jobs7m/jobs7m.txt && | |
| 39 #end for | |
| 40 #else | |
| 41 ## paired collection: forward + reverse | |
| 42 mkdir -p ./reads && | |
| 43 #set sample_name = $reads.input.name.replace(' ', '_') | |
| 44 #set fwd_ext = $fix_ext($reads.input.forward.ext) | |
| 45 #set rev_ext = $fix_ext($reads.input.reverse.ext) | |
| 46 #set base = re.sub('[^\w\-_.]', '_', $reads.input.element_identifier) | |
| 47 #set fwd_fname = $base + '_1.' + $fwd_ext | |
| 48 #set rev_fname = $base + '_2.' + $rev_ext | |
| 49 ln -sf '$reads.input.forward' './reads/${fwd_fname}' && | |
| 50 ln -sf '$reads.input.reverse' './reads/${rev_fname}' && | |
| 51 printf '%s\t2\n' '${sample_name}' > ./jobs7m/jobs7m.txt && | |
| 52 printf '%s\n' './reads/${fwd_fname}' >> ./jobs7m/jobs7m.txt && | |
| 53 printf '%s\n' './reads/${rev_fname}' >> ./jobs7m/jobs7m.txt && | |
| 54 #end if | |
| 55 | |
| 56 ## ── Run the Python orchestrator ─────────────────────────────────────── | |
| 40 python3 /opt/mitokmer2/kmer_read_m7.py && | 57 python3 /opt/mitokmer2/kmer_read_m7.py && |
| 41 | 58 |
| 42 ## ── Copy output CSV to Galaxy output path ───────────────────────────── | 59 ## ── Copy output CSV to Galaxy output path ───────────────────────────── |
| 43 cp ./jobs7m/jobs7m.csv '${results_csv}' | 60 cp ./jobs7m/jobs7m.csv '${results_csv}' |
| 44 ]]></command> | 61 ]]></command> |
| 45 | 62 |
| 46 <inputs> | 63 <inputs> |
| 47 <!-- Probe database directory selected from Galaxy data table --> | 64 <param name="probe_db" type="select" label="Mitochondrial k-mer probe database" |
| 48 <param name="probe_db" | |
| 49 type="select" | |
| 50 label="Mitochondrial k-mer probe database" | |
| 51 help="Select a pre-installed mitochondrial k-mer probe database. | 65 help="Select a pre-installed mitochondrial k-mer probe database. |
| 52 Databases are managed by your Galaxy administrator via the | 66 Databases are managed by your Galaxy administrator via the |
| 53 mitokmer_probe_db data table."> | 67 mitokmer_probe_db data table."> |
| 54 <options from_data_table="mitokmer_probe_db"> | 68 <options from_data_table="mitokmer_probe_db"> |
| 55 <filter type="sort_by" column="1" /> | 69 <filter type="sort_by" column="1" /> |
| 56 <validator type="no_options" | 70 <validator type="no_options" message="No mitochondrial k-mer probe databases are |
| 57 message="No mitochondrial k-mer probe databases are | 71 currently installed. Please contact your Galaxy administrator." /> |
| 58 currently installed. Please contact your | |
| 59 Galaxy administrator." /> | |
| 60 </options> | 72 </options> |
| 61 </param> | 73 </param> |
| 62 | 74 |
| 63 <param name="reads" | 75 <conditional name="reads"> |
| 64 type="data_collection" | 76 <param name="reads_type" type="select" label="Input read type"> |
| 65 collection_type="list" | 77 <option value="single">Single-end or unpaired reads / FASTA (list collection)</option> |
| 66 format="fastq,fastq.gz,fasta,fasta.gz" | 78 <option value="paired">Paired-end reads (paired collection)</option> |
| 67 label="Input reads (FASTQ or FASTA, gzipped or plain)" | 79 </param> |
| 68 help="Provide one or more read files for a single sample as a Galaxy | 80 <when value="single"> |
| 69 list collection. Paired-end files (R1 + R2) should both be | 81 <param name="input" type="data_collection" collection_type="list" |
| 70 included in the same collection. Light quality trimming of | 82 format="fastqsanger,fastqsanger.gz,fastqillumina,fastqillumina.gz,fasta,fasta.gz" |
| 71 read ends is performed internally; pre-trimming is optional." /> | 83 label="Input reads (FASTQ or FASTA, gzipped or plain)" |
| 84 help="Provide one or more single-end or unpaired read files as a Galaxy | |
| 85 list collection. For paired-end data, use the 'Paired-end reads' | |
| 86 option instead." /> | |
| 87 </when> | |
| 88 <when value="paired"> | |
| 89 <param name="input" type="data_collection" collection_type="paired" | |
| 90 format="fastqsanger,fastqsanger.gz,fastqillumina,fastqillumina.gz" | |
| 91 label="Paired-end reads (paired collection)" | |
| 92 help="Provide a Galaxy paired collection containing forward (R1) and | |
| 93 reverse (R2) reads. Light quality trimming of read ends is | |
| 94 performed internally; pre-trimming is optional." /> | |
| 95 </when> | |
| 96 </conditional> | |
| 72 </inputs> | 97 </inputs> |
| 73 | 98 |
| 74 <outputs> | 99 <outputs> |
| 75 <data name="results_csv" | 100 <data name="results_csv" format="csv" label="mitoKmer results for ${on_string}" /> |
| 76 format="csv" | |
| 77 label="mitoKmer results for ${on_string}" /> | |
| 78 </outputs> | 101 </outputs> |
| 79 | 102 |
| 80 <tests> | 103 <tests> |
| 104 <!-- Test 1: paired-end FASTQ via paired collection --> | |
| 81 <test> | 105 <test> |
| 82 <param name="probe_db" value="mitoch_probes_sample" /> | 106 <param name="probe_db" value="mitoch_probes_sample" /> |
| 83 <param name="reads"> | 107 <conditional name="reads"> |
| 84 <collection type="list"> | 108 <param name="reads_type" value="paired" /> |
| 85 <element name="Plodia_R1" value="test/Plodia_R1.fastq.gz" /> | 109 <param name="input"> |
| 86 <element name="Plodia_R2" value="test/Plodia_R2.fastq.gz" /> | 110 <collection type="paired"> |
| 87 </collection> | 111 <element name="forward" value="test/Plodia_R1.fastq.gz" ftype="fastqsanger.gz" /> |
| 88 </param> | 112 <element name="reverse" value="test/Plodia_R2.fastq.gz" ftype="fastqsanger.gz" /> |
| 113 </collection> | |
| 114 </param> | |
| 115 </conditional> | |
| 116 <output name="results_csv"> | |
| 117 <assert_contents> | |
| 118 <has_text text="Plodia" /> | |
| 119 </assert_contents> | |
| 120 </output> | |
| 121 </test> | |
| 122 <!-- Test 2: single-end FASTQ via list collection --> | |
| 123 <test> | |
| 124 <param name="probe_db" value="mitoch_probes_sample" /> | |
| 125 <conditional name="reads"> | |
| 126 <param name="reads_type" value="single" /> | |
| 127 <param name="input"> | |
| 128 <collection type="list"> | |
| 129 <element name="Plodia_R1" value="test/Plodia_R1.fastq.gz" ftype="fastqsanger.gz" /> | |
| 130 </collection> | |
| 131 </param> | |
| 132 </conditional> | |
| 133 <output name="results_csv"> | |
| 134 <assert_contents> | |
| 135 <has_text text="Plodia" /> | |
| 136 </assert_contents> | |
| 137 </output> | |
| 138 </test> | |
| 139 <!-- Test 3: single FASTA via list collection --> | |
| 140 <test> | |
| 141 <param name="probe_db" value="mitoch_probes_sample" /> | |
| 142 <conditional name="reads"> | |
| 143 <param name="reads_type" value="single" /> | |
| 144 <param name="input"> | |
| 145 <collection type="list"> | |
| 146 <element name="Plodia_assembly" value="test/Plodia_assembly.fasta" ftype="fasta" /> | |
| 147 </collection> | |
| 148 </param> | |
| 149 </conditional> | |
| 89 <output name="results_csv"> | 150 <output name="results_csv"> |
| 90 <assert_contents> | 151 <assert_contents> |
| 91 <has_text text="Plodia" /> | 152 <has_text text="Plodia" /> |
| 92 </assert_contents> | 153 </assert_contents> |
| 93 </output> | 154 </output> |
| 100 | 161 |
| 101 Overview | 162 Overview |
| 102 -------- | 163 -------- |
| 103 mitoKmer identifies the taxonomic origin of short-read shotgun sequencing data | 164 mitoKmer identifies the taxonomic origin of short-read shotgun sequencing data |
| 104 by matching reads against a database of species-specific mitochondrial k-mer | 165 by matching reads against a database of species-specific mitochondrial k-mer |
| 105 probes. It reports the relative abundance of taxa at multiple taxonomic ranks | 166 probes. It reports the relative abundance of taxa at multiple taxonomic ranks |
| 106 (e.g. order, species) along with the number of matching reads and supporting | 167 (e.g. order, species) along with the number of matching reads and supporting |
| 107 unique k-mers. | 168 unique k-mers. |
| 108 | 169 |
| 109 Inputs | 170 Inputs |
| 110 ------ | 171 ------ |
| 111 **Mitochondrial k-mer probe database** | 172 **Mitochondrial k-mer probe database** |
| 112 Select a pre-installed probe database from the dropdown. Databases are | 173 Select a pre-installed probe database from the dropdown. Databases are |
| 113 managed by your Galaxy administrator and registered in the | 174 managed by your Galaxy administrator and registered in the |
| 114 ``mitokmer_probe_db`` data table. The database consists of a directory | 175 ``mitokmer_probe_db`` data table. The database consists of a directory |
| 115 of supporting ``.txt`` files. | 176 of ``.txt`` files. |
| 116 | 177 |
| 117 **Input reads** | 178 **Input read type** |
| 118 One or more FASTQ or FASTA files (gzipped or plain) for a single sample, | 179 Choose the input mode that matches your data: |
| 119 supplied as a Galaxy list collection. Both paired-end files (R1 and R2) | 180 |
| 120 should be included in the same collection. | 181 *Single-end or unpaired reads / FASTA (list collection)* |
| 121 | 182 Provide a Galaxy **list** collection containing one or more FASTQ |
| 122 * Pre-trimming is optional — the tool performs simple window-based quality | 183 (``fastqsanger``, ``fastqsanger.gz``) or FASTA (``fasta``, |
| 123 trimming of read ends internally. | 184 ``fasta.gz``) files. Use this for single-end sequencing data or |
| 185 assembled FASTA sequences. | |
| 186 | |
| 187 *Paired-end reads (paired collection)* | |
| 188 Provide a Galaxy **paired** collection where the forward (R1) and | |
| 189 reverse (R2) reads are paired together. Both files must be FASTQ | |
| 190 format (``fastqsanger`` or ``fastqsanger.gz``). | |
| 191 | |
| 192 In all cases, light window-based quality trimming of read ends is | |
| 193 performed internally; pre-trimming is optional. | |
| 124 | 194 |
| 125 Output | 195 Output |
| 126 ------ | 196 ------ |
| 127 **Results CSV** | 197 **Results CSV** |
| 128 A comma-separated file with columns: taxid, reads, abundance, uniq. | 198 A comma-separated file with one row per taxon detected. Columns: |
| 199 | |
| 200 ========= ============================================================ | |
| 201 Column Description | |
| 202 ========= ============================================================ | |
| 203 Rank Taxonomic rank (e.g. Order, Species) | |
| 204 Taxon Taxon name | |
| 205 Reads Number of reads matching this taxon | |
| 206 Rel_Abund Relative abundance (%) | |
| 207 Unique_kmers Number of unique k-mers supporting the match | |
| 208 ========= ============================================================ | |
| 209 | |
| 129 Only taxa with non-zero relative abundance are reported. | 210 Only taxa with non-zero relative abundance are reported. |
| 130 | 211 |
| 131 Example output for the "Plodia" test sample:: | 212 Example output for the "Plodia" test sample:: |
| 132 | 213 |
| 133 Rank Taxon Reads Rel_Abund(%) Unique_kmers | 214 Rank Taxon Reads Rel_Abund(%) Unique_kmers |
