Mercurial > repos > galaxytrakr > mitokmer
diff mitokmer.xml @ 5:8f4abeb27625 draft
planemo upload commit 3118e6e386354d664c85c47c67b75a92629d358c
| author | galaxytrakr |
|---|---|
| date | Tue, 15 Sep 2026 18:31:11 +0000 |
| parents | ecf96ad08611 |
| children | e4d4fc748584 |
line wrap: on
line diff
--- a/mitokmer.xml Tue Sep 15 12:02:36 2026 +0000 +++ b/mitokmer.xml Tue Sep 15 18:31:11 2026 +0000 @@ -1,42 +1,59 @@ -<tool id="mitokmer" name="mitoKmer" version="2.0+galaxy0.2" python_template_version="3.5" profile="21.05"> +<tool id="mitokmer" name="mitoKmer" version="2.0+galaxy0.2" profile="21.05"> <description>Identify metagenomic mitochondrial reads by k-mer database matching</description> <requirements> <container type="docker">quay.io/galaxytrakr/mitokmer:e57a559</container> </requirements> <command detect_errors="exit_code"><![CDATA[ - ## ── All paths are hardcoded in kmer_read_m7.py: ────────────────────── - ## database: ./mitochondria7/<multiple .txt files> - ## jobs file: ./jobs7m/jobs7m.txt - ## output: ./jobs7m/jobs7m.csv - ## binary: ./kmerread7 (called as subprocess by the Python script) + #import re + #def fix_ext($ext): + #set ext = $ext.replace('fastqsanger', 'fastq') + #set ext = $ext.replace('fastqillumina', 'fastq') + #return $ext + #end def + mkdir -p ./mitochondria7 ./jobs7m && - ## ── Link database files from the data table directory ───────────────── - ## Symlink each file individually from the registered database directory. - ## Note: glob must be outside quotes so the shell expands it correctly. + ## ── Link database files into working directory ──────────────────────── ln -sf '${probe_db.fields.path}'/* ./mitochondria7/ && - ## ── Symlink kmerread7 into the working directory ────────────────────── - ## kmer_read_m7.py calls ./kmerread7 (relative path), so it must exist - ## in the Galaxy job working directory + ## ── Symlink kmerread7 binary into working directory ─────────────────── ln -sf /usr/local/bin/kmerread7 ./kmerread7 && - ## ── Write the jobs file ─────────────────────────────────────────────── - ## Format: sample_name <tab> num_files - ## /abs/path/to/read1 - ## /abs/path/to/read2 ... - ## Use len() to safely get collection size as an integer - #set sample_name = $reads.name.replace(' ', '_') - #set read_count = len($reads) - printf '%s\t%d\n' '${sample_name}' ${read_count} > ./jobs7m/jobs7m.txt && - #for read in $reads - printf '%s\n' '${read.file_name}' >> ./jobs7m/jobs7m.txt && - #end for + ## ── Stage input reads and build jobs file ──────────────────────────── + #if $reads.reads_type == "single" + mkdir -p ./reads && + #set sample_name = $reads.input.name.replace(' ', '_') + #set read_count = 0 + #for $read in $reads.input + #set ext = $fix_ext($read.ext) + #set fname = re.sub('[^\w\-_.]', '_', $read.element_identifier) + '.' + $ext + ln -sf '$read' './reads/${fname}' && + #set read_count = $read_count + 1 + #end for + printf '%s\t%d\n' '${sample_name}' ${read_count} > ./jobs7m/jobs7m.txt && + #for $read in $reads.input + #set ext = $fix_ext($read.ext) + #set fname = re.sub('[^\w\-_.]', '_', $read.element_identifier) + '.' + $ext + printf '%s\n' './reads/${fname}' >> ./jobs7m/jobs7m.txt && + #end for + #else + ## paired collection: forward + reverse + mkdir -p ./reads && + #set sample_name = $reads.input.name.replace(' ', '_') + #set fwd_ext = $fix_ext($reads.input.forward.ext) + #set rev_ext = $fix_ext($reads.input.reverse.ext) + #set base = re.sub('[^\w\-_.]', '_', $reads.input.element_identifier) + #set fwd_fname = $base + '_1.' + $fwd_ext + #set rev_fname = $base + '_2.' + $rev_ext + ln -sf '$reads.input.forward' './reads/${fwd_fname}' && + ln -sf '$reads.input.reverse' './reads/${rev_fname}' && + printf '%s\t2\n' '${sample_name}' > ./jobs7m/jobs7m.txt && + printf '%s\n' './reads/${fwd_fname}' >> ./jobs7m/jobs7m.txt && + printf '%s\n' './reads/${rev_fname}' >> ./jobs7m/jobs7m.txt && + #end if - ## ── Run the Python orchestrator (no arguments) ──────────────────────── - ## kmer_read_m7.py reads jobs7m/jobs7m.txt, calls ./kmerread7, - ## and writes results to jobs7m/jobs7m.csv + ## ── Run the Python orchestrator ─────────────────────────────────────── python3 /opt/mitokmer2/kmer_read_m7.py && ## ── Copy output CSV to Galaxy output path ───────────────────────────── @@ -44,48 +61,92 @@ ]]></command> <inputs> - <!-- Probe database directory selected from Galaxy data table --> - <param name="probe_db" - type="select" - label="Mitochondrial k-mer probe database" + <param name="probe_db" type="select" label="Mitochondrial k-mer probe database" help="Select a pre-installed mitochondrial k-mer probe database. Databases are managed by your Galaxy administrator via the mitokmer_probe_db data table."> <options from_data_table="mitokmer_probe_db"> <filter type="sort_by" column="1" /> - <validator type="no_options" - message="No mitochondrial k-mer probe databases are - currently installed. Please contact your - Galaxy administrator." /> + <validator type="no_options" message="No mitochondrial k-mer probe databases are + currently installed. Please contact your Galaxy administrator." /> </options> </param> - <param name="reads" - type="data_collection" - collection_type="list" - format="fastq,fastq.gz,fasta,fasta.gz" - label="Input reads (FASTQ or FASTA, gzipped or plain)" - help="Provide one or more read files for a single sample as a Galaxy - list collection. Paired-end files (R1 + R2) should both be - included in the same collection. Light quality trimming of - read ends is performed internally; pre-trimming is optional." /> + <conditional name="reads"> + <param name="reads_type" type="select" label="Input read type"> + <option value="single">Single-end or unpaired reads / FASTA (list collection)</option> + <option value="paired">Paired-end reads (paired collection)</option> + </param> + <when value="single"> + <param name="input" type="data_collection" collection_type="list" + format="fastqsanger,fastqsanger.gz,fastqillumina,fastqillumina.gz,fasta,fasta.gz" + label="Input reads (FASTQ or FASTA, gzipped or plain)" + help="Provide one or more single-end or unpaired read files as a Galaxy + list collection. For paired-end data, use the 'Paired-end reads' + option instead." /> + </when> + <when value="paired"> + <param name="input" type="data_collection" collection_type="paired" + format="fastqsanger,fastqsanger.gz,fastqillumina,fastqillumina.gz" + label="Paired-end reads (paired collection)" + help="Provide a Galaxy paired collection containing forward (R1) and + reverse (R2) reads. Light quality trimming of read ends is + performed internally; pre-trimming is optional." /> + </when> + </conditional> </inputs> <outputs> - <data name="results_csv" - format="csv" - label="mitoKmer results for ${on_string}" /> + <data name="results_csv" format="csv" label="mitoKmer results for ${on_string}" /> </outputs> <tests> + <!-- Test 1: paired-end FASTQ via paired collection --> + <test> + <param name="probe_db" value="mitoch_probes_sample" /> + <conditional name="reads"> + <param name="reads_type" value="paired" /> + <param name="input"> + <collection type="paired"> + <element name="forward" value="test/Plodia_R1.fastq.gz" ftype="fastqsanger.gz" /> + <element name="reverse" value="test/Plodia_R2.fastq.gz" ftype="fastqsanger.gz" /> + </collection> + </param> + </conditional> + <output name="results_csv"> + <assert_contents> + <has_text text="Plodia" /> + </assert_contents> + </output> + </test> + <!-- Test 2: single-end FASTQ via list collection --> <test> <param name="probe_db" value="mitoch_probes_sample" /> - <param name="reads"> - <collection type="list"> - <element name="Plodia_R1" value="test/Plodia_R1.fastq.gz" /> - <element name="Plodia_R2" value="test/Plodia_R2.fastq.gz" /> - </collection> - </param> + <conditional name="reads"> + <param name="reads_type" value="single" /> + <param name="input"> + <collection type="list"> + <element name="Plodia_R1" value="test/Plodia_R1.fastq.gz" ftype="fastqsanger.gz" /> + </collection> + </param> + </conditional> + <output name="results_csv"> + <assert_contents> + <has_text text="Plodia" /> + </assert_contents> + </output> + </test> + <!-- Test 3: single FASTA via list collection --> + <test> + <param name="probe_db" value="mitoch_probes_sample" /> + <conditional name="reads"> + <param name="reads_type" value="single" /> + <param name="input"> + <collection type="list"> + <element name="Plodia_assembly" value="test/Plodia_assembly.fasta" ftype="fasta" /> + </collection> + </param> + </conditional> <output name="results_csv"> <assert_contents> <has_text text="Plodia" /> @@ -102,30 +163,50 @@ -------- mitoKmer identifies the taxonomic origin of short-read shotgun sequencing data by matching reads against a database of species-specific mitochondrial k-mer -probes. It reports the relative abundance of taxa at multiple taxonomic ranks +probes. It reports the relative abundance of taxa at multiple taxonomic ranks (e.g. order, species) along with the number of matching reads and supporting unique k-mers. Inputs ------ **Mitochondrial k-mer probe database** - Select a pre-installed probe database from the dropdown. Databases are + Select a pre-installed probe database from the dropdown. Databases are managed by your Galaxy administrator and registered in the - ``mitokmer_probe_db`` data table. The database consists of a directory - of supporting ``.txt`` files. + ``mitokmer_probe_db`` data table. The database consists of a directory + of ``.txt`` files. + +**Input read type** + Choose the input mode that matches your data: -**Input reads** - One or more FASTQ or FASTA files (gzipped or plain) for a single sample, - supplied as a Galaxy list collection. Both paired-end files (R1 and R2) - should be included in the same collection. + *Single-end or unpaired reads / FASTA (list collection)* + Provide a Galaxy **list** collection containing one or more FASTQ + (``fastqsanger``, ``fastqsanger.gz``) or FASTA (``fasta``, + ``fasta.gz``) files. Use this for single-end sequencing data or + assembled FASTA sequences. - * Pre-trimming is optional — the tool performs simple window-based quality - trimming of read ends internally. + *Paired-end reads (paired collection)* + Provide a Galaxy **paired** collection where the forward (R1) and + reverse (R2) reads are paired together. Both files must be FASTQ + format (``fastqsanger`` or ``fastqsanger.gz``). + + In all cases, light window-based quality trimming of read ends is + performed internally; pre-trimming is optional. Output ------ **Results CSV** - A comma-separated file with columns: taxid, reads, abundance, uniq. + A comma-separated file with one row per taxon detected. Columns: + + ========= ============================================================ + Column Description + ========= ============================================================ + Rank Taxonomic rank (e.g. Order, Species) + Taxon Taxon name + Reads Number of reads matching this taxon + Rel_Abund Relative abundance (%) + Unique_kmers Number of unique k-mers supporting the match + ========= ============================================================ + Only taxa with non-zero relative abundance are reported. Example output for the "Plodia" test sample::
