# HG changeset patch # User galaxytrakr # Date 1789497071 0 # Node ID 8f4abeb27625530654f97606b37c385f25741b66 # Parent ecf96ad086119d320116e6b2dc44260ec68bd3b8 planemo upload commit 3118e6e386354d664c85c47c67b75a92629d358c diff -r ecf96ad08611 -r 8f4abeb27625 mitokmer.xml --- a/mitokmer.xml Tue Sep 15 12:02:36 2026 +0000 +++ b/mitokmer.xml Tue Sep 15 18:31:11 2026 +0000 @@ -1,42 +1,59 @@ - + Identify metagenomic mitochondrial reads by k-mer database matching quay.io/galaxytrakr/mitokmer:e57a559 - ## jobs file: ./jobs7m/jobs7m.txt - ## output: ./jobs7m/jobs7m.csv - ## binary: ./kmerread7 (called as subprocess by the Python script) + #import re + #def fix_ext($ext): + #set ext = $ext.replace('fastqsanger', 'fastq') + #set ext = $ext.replace('fastqillumina', 'fastq') + #return $ext + #end def + mkdir -p ./mitochondria7 ./jobs7m && - ## ── Link database files from the data table directory ───────────────── - ## Symlink each file individually from the registered database directory. - ## Note: glob must be outside quotes so the shell expands it correctly. + ## ── Link database files into working directory ──────────────────────── ln -sf '${probe_db.fields.path}'/* ./mitochondria7/ && - ## ── Symlink kmerread7 into the working directory ────────────────────── - ## kmer_read_m7.py calls ./kmerread7 (relative path), so it must exist - ## in the Galaxy job working directory + ## ── Symlink kmerread7 binary into working directory ─────────────────── ln -sf /usr/local/bin/kmerread7 ./kmerread7 && - ## ── Write the jobs file ─────────────────────────────────────────────── - ## Format: sample_name num_files - ## /abs/path/to/read1 - ## /abs/path/to/read2 ... - ## Use len() to safely get collection size as an integer - #set sample_name = $reads.name.replace(' ', '_') - #set read_count = len($reads) - printf '%s\t%d\n' '${sample_name}' ${read_count} > ./jobs7m/jobs7m.txt && - #for read in $reads - printf '%s\n' '${read.file_name}' >> ./jobs7m/jobs7m.txt && - #end for + ## ── Stage input reads and build jobs file ──────────────────────────── + #if $reads.reads_type == "single" + mkdir -p ./reads && + #set sample_name = $reads.input.name.replace(' ', '_') + #set read_count = 0 + #for $read in $reads.input + #set ext = $fix_ext($read.ext) + #set fname = re.sub('[^\w\-_.]', '_', $read.element_identifier) + '.' + $ext + ln -sf '$read' './reads/${fname}' && + #set read_count = $read_count + 1 + #end for + printf '%s\t%d\n' '${sample_name}' ${read_count} > ./jobs7m/jobs7m.txt && + #for $read in $reads.input + #set ext = $fix_ext($read.ext) + #set fname = re.sub('[^\w\-_.]', '_', $read.element_identifier) + '.' + $ext + printf '%s\n' './reads/${fname}' >> ./jobs7m/jobs7m.txt && + #end for + #else + ## paired collection: forward + reverse + mkdir -p ./reads && + #set sample_name = $reads.input.name.replace(' ', '_') + #set fwd_ext = $fix_ext($reads.input.forward.ext) + #set rev_ext = $fix_ext($reads.input.reverse.ext) + #set base = re.sub('[^\w\-_.]', '_', $reads.input.element_identifier) + #set fwd_fname = $base + '_1.' + $fwd_ext + #set rev_fname = $base + '_2.' + $rev_ext + ln -sf '$reads.input.forward' './reads/${fwd_fname}' && + ln -sf '$reads.input.reverse' './reads/${rev_fname}' && + printf '%s\t2\n' '${sample_name}' > ./jobs7m/jobs7m.txt && + printf '%s\n' './reads/${fwd_fname}' >> ./jobs7m/jobs7m.txt && + printf '%s\n' './reads/${rev_fname}' >> ./jobs7m/jobs7m.txt && + #end if - ## ── Run the Python orchestrator (no arguments) ──────────────────────── - ## kmer_read_m7.py reads jobs7m/jobs7m.txt, calls ./kmerread7, - ## and writes results to jobs7m/jobs7m.csv + ## ── Run the Python orchestrator ─────────────────────────────────────── python3 /opt/mitokmer2/kmer_read_m7.py && ## ── Copy output CSV to Galaxy output path ───────────────────────────── @@ -44,48 +61,92 @@ ]]> - - - + - + + + + + + + + + + + + - + + + + + + + + + + + + + + + + + + + + - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + @@ -102,30 +163,50 @@ -------- mitoKmer identifies the taxonomic origin of short-read shotgun sequencing data by matching reads against a database of species-specific mitochondrial k-mer -probes. It reports the relative abundance of taxa at multiple taxonomic ranks +probes. It reports the relative abundance of taxa at multiple taxonomic ranks (e.g. order, species) along with the number of matching reads and supporting unique k-mers. Inputs ------ **Mitochondrial k-mer probe database** - Select a pre-installed probe database from the dropdown. Databases are + Select a pre-installed probe database from the dropdown. Databases are managed by your Galaxy administrator and registered in the - ``mitokmer_probe_db`` data table. The database consists of a directory - of supporting ``.txt`` files. + ``mitokmer_probe_db`` data table. The database consists of a directory + of ``.txt`` files. + +**Input read type** + Choose the input mode that matches your data: -**Input reads** - One or more FASTQ or FASTA files (gzipped or plain) for a single sample, - supplied as a Galaxy list collection. Both paired-end files (R1 and R2) - should be included in the same collection. + *Single-end or unpaired reads / FASTA (list collection)* + Provide a Galaxy **list** collection containing one or more FASTQ + (``fastqsanger``, ``fastqsanger.gz``) or FASTA (``fasta``, + ``fasta.gz``) files. Use this for single-end sequencing data or + assembled FASTA sequences. - * Pre-trimming is optional — the tool performs simple window-based quality - trimming of read ends internally. + *Paired-end reads (paired collection)* + Provide a Galaxy **paired** collection where the forward (R1) and + reverse (R2) reads are paired together. Both files must be FASTQ + format (``fastqsanger`` or ``fastqsanger.gz``). + + In all cases, light window-based quality trimming of read ends is + performed internally; pre-trimming is optional. Output ------ **Results CSV** - A comma-separated file with columns: taxid, reads, abundance, uniq. + A comma-separated file with one row per taxon detected. Columns: + + ========= ============================================================ + Column Description + ========= ============================================================ + Rank Taxonomic rank (e.g. Order, Species) + Taxon Taxon name + Reads Number of reads matching this taxon + Rel_Abund Relative abundance (%) + Unique_kmers Number of unique k-mers supporting the match + ========= ============================================================ + Only taxa with non-zero relative abundance are reported. Example output for the "Plodia" test sample::