Mercurial > repos > galaxytrakr > mitokmer
view mitokmer.xml @ 8:2b1f3db24c25 draft default tip
planemo upload commit 41caf97a4a9a7928318af8b24601f84a380fe2db
| author | galaxytrakr |
|---|---|
| date | Fri, 18 Sep 2026 12:26:54 +0000 |
| parents | 56b71adcaba7 |
| children |
line wrap: on
line source
<tool id="mitokmer" name="mitoKmer" version="2.0+galaxy0.24" profile="21.05"> <description>Identify metagenomic mitochondrial reads by k-mer database matching</description> <requirements> <container type="docker">quay.io/galaxytrakr/mitokmer:e57a559</container> </requirements> <command detect_errors="exit_code"><![CDATA[ #import re #def fix_ext($ext): #set ext = $ext.replace('fastqsanger', 'fastq') #set ext = $ext.replace('fastqillumina', 'fastq') #return $ext #end def mkdir -p ./mitochondria7 ./jobs7m && ## ── Link database files into working directory ──────────────────────── ln -sf '${probe_db.fields.path}'/* ./mitochondria7/ && ## ── Symlink kmerread7 binary into working directory ─────────────────── ln -sf /usr/local/bin/kmerread7 ./kmerread7 && ## ── Stage input reads and build jobs file ──────────────────────────── #if $reads.reads_type == "single_file" mkdir -p ./reads && #set sample_name = re.sub('[^\w\-_.]', '_', $reads.input.name) #set ext = $fix_ext($reads.input.ext) #set fname = re.sub('[^\w\-_.]', '_', $reads.input.element_identifier) + '.' + $ext ln -sf '$reads.input' './reads/${fname}' && printf '%s\t1\n' '${sample_name}' > ./jobs7m/jobs7m.txt && printf '%s\n' './reads/${fname}' >> ./jobs7m/jobs7m.txt && #else if $reads.reads_type == "single" mkdir -p ./reads && #set sample_name = re.sub('[^\w\-_.]', '_', $reads.input.name) #set read_count = 0 #for $read in $reads.input #set ext = $fix_ext($read.ext) #set fname = re.sub('[^\w\-_.]', '_', $read.element_identifier) + '.' + $ext ln -sf '$read' './reads/${fname}' && #set read_count = $read_count + 1 #end for printf '%s\t%d\n' '${sample_name}' ${read_count} > ./jobs7m/jobs7m.txt && #for $read in $reads.input #set ext = $fix_ext($read.ext) #set fname = re.sub('[^\w\-_.]', '_', $read.element_identifier) + '.' + $ext printf '%s\n' './reads/${fname}' >> ./jobs7m/jobs7m.txt && #end for #else ## paired collection: forward + reverse mkdir -p ./reads && #set sample_name = re.sub('[^\w\-_.]', '_', $reads.input.name) #set fwd_ext = $fix_ext($reads.input.forward.ext) #set rev_ext = $fix_ext($reads.input.reverse.ext) #set base = re.sub('[^\w\-_.]', '_', $reads.input.element_identifier) #set fwd_fname = $base + '_1.' + $fwd_ext #set rev_fname = $base + '_2.' + $rev_ext ln -sf '$reads.input.forward' './reads/${fwd_fname}' && ln -sf '$reads.input.reverse' './reads/${rev_fname}' && printf '%s\t2\n' '${sample_name}' > ./jobs7m/jobs7m.txt && printf '%s\n' './reads/${fwd_fname}' >> ./jobs7m/jobs7m.txt && printf '%s\n' './reads/${rev_fname}' >> ./jobs7m/jobs7m.txt && #end if ## ── Run the Python orchestrator ─────────────────────────────────────── python3 /opt/mitokmer2/kmer_read_m7.py && ## ── Copy output CSV to Galaxy output path ───────────────────────────── cp ./jobs7m/jobs7m.csv '${results_csv}' ]]></command> <inputs> <param name="probe_db" type="select" label="Mitochondrial k-mer probe database" help="Select a pre-installed mitochondrial k-mer probe database. Databases are managed by your Galaxy administrator via the mitokmer_probe_db data table."> <options from_data_table="mitokmer_probe_db"> <filter type="sort_by" column="1" /> <validator type="no_options" message="No mitochondrial k-mer probe databases are currently installed. Please contact your Galaxy administrator." /> </options> </param> <conditional name="reads"> <param name="reads_type" type="select" label="Input read type"> <option value="single_file">Single FASTA or FASTQ dataset</option> <option value="single">Single-end or unpaired reads / FASTA (list collection)</option> <option value="paired">Paired-end reads (paired collection)</option> </param> <when value="single_file"> <param name="input" type="data" format="fastqsanger,fastqsanger.gz,fastqillumina,fastqillumina.gz,fasta,fasta.gz" label="Input FASTA or FASTQ dataset" help="Provide a single FASTA or FASTQ file from your history." /> </when> <when value="single"> <param name="input" type="data_collection" collection_type="list" format="fastqsanger,fastqsanger.gz,fastqillumina,fastqillumina.gz,fasta,fasta.gz" label="Input reads (FASTQ or FASTA, gzipped or plain)" help="Provide one or more single-end or unpaired read files as a Galaxy list collection. For paired-end data, use the 'Paired-end reads' option instead." /> </when> <when value="paired"> <param name="input" type="data_collection" collection_type="paired" format="fastqsanger,fastqsanger.gz,fastqillumina,fastqillumina.gz" label="Paired-end reads (paired collection)" help="Provide a Galaxy paired collection containing forward (R1) and reverse (R2) reads. Light quality trimming of read ends is performed internally; pre-trimming is optional." /> </when> </conditional> </inputs> <outputs> <data name="results_csv" format="csv" label="mitoKmer results for ${on_string}" /> </outputs> <tests> <!-- Test 1: single FASTA dataset --> <test> <param name="probe_db" value="mitoch_probes_sample" /> <conditional name="reads"> <param name="reads_type" value="single_file" /> <param name="input" value="test/Plodia_assembly.fasta" ftype="fasta" /> </conditional> <output name="results_csv"> <assert_contents> <has_text text="Plodia" /> </assert_contents> </output> </test> <!-- Test 2: paired-end FASTQ via paired collection --> <test> <param name="probe_db" value="mitoch_probes_sample" /> <conditional name="reads"> <param name="reads_type" value="paired" /> <param name="input"> <collection type="paired"> <element name="forward" value="test/Plodia_R1.fastq.gz" ftype="fastqsanger.gz" /> <element name="reverse" value="test/Plodia_R2.fastq.gz" ftype="fastqsanger.gz" /> </collection> </param> </conditional> <output name="results_csv"> <assert_contents> <has_text text="Plodia" /> </assert_contents> </output> </test> <!-- Test 3: single-end FASTQ via list collection --> <test> <param name="probe_db" value="mitoch_probes_sample" /> <conditional name="reads"> <param name="reads_type" value="single" /> <param name="input"> <collection type="list"> <element name="Plodia_R1" value="test/Plodia_R1.fastq.gz" ftype="fastqsanger.gz" /> </collection> </param> </conditional> <output name="results_csv"> <assert_contents> <has_text text="Plodia" /> </assert_contents> </output> </test> <!-- Test 4: FASTA list collection --> <test> <param name="probe_db" value="mitoch_probes_sample" /> <conditional name="reads"> <param name="reads_type" value="single" /> <param name="input"> <collection type="list"> <element name="Plodia_assembly" value="test/Plodia_assembly.fasta" ftype="fasta" /> </collection> </param> </conditional> <output name="results_csv"> <assert_contents> <has_text text="Plodia" /> </assert_contents> </output> </test> </tests> <help><![CDATA[ **mitoKmer** — Metagenomic Mitochondrial Read Identification by K-mer Database =============================================================================== Overview -------- mitoKmer identifies the taxonomic origin of short-read shotgun sequencing data by matching reads against a database of species-specific mitochondrial k-mer probes. It reports the relative abundance of taxa at multiple taxonomic ranks (e.g. order, species) along with the number of matching reads and supporting unique k-mers. Inputs ------ **Mitochondrial k-mer probe database** Select a pre-installed probe database from the dropdown. Databases are managed by your Galaxy administrator and registered in the ``mitokmer_probe_db`` data table. The database consists of a directory of ``.txt`` files. **Input read type** Choose the input mode that matches your data: *Single FASTA or FASTQ dataset* Provide a single FASTA or FASTQ file directly from your history. This is the simplest option for a single assembled sequence or single-end read file. *Single-end or unpaired reads / FASTA (list collection)* Provide a Galaxy **list** collection containing one or more FASTQ (``fastqsanger``, ``fastqsanger.gz``) or FASTA (``fasta``, ``fasta.gz``) files. Use this for single-end sequencing data or multiple assembled FASTA sequences processed together. *Paired-end reads (paired collection)* Provide a Galaxy **paired** collection where the forward (R1) and reverse (R2) reads are paired together. Both files must be FASTQ format (``fastqsanger`` or ``fastqsanger.gz``). In all cases, light window-based quality trimming of read ends is performed internally; pre-trimming is optional. Output ------ **Results CSV** A comma-separated file with one row per taxon detected. Columns: ========= ============================================================ Column Description ========= ============================================================ Rank Taxonomic rank (e.g. Order, Species) Taxon Taxon name Reads Number of reads matching this taxon Rel_Abund Relative abundance (%) Unique_kmers Number of unique k-mers supporting the match ========= ============================================================ Only taxa with non-zero relative abundance are reported. Example output for the "Plodia" test sample:: Rank Taxon Reads Rel_Abund(%) Unique_kmers Order Lepidoptera 10 0.1 329 Species Plodia interpunctella 462 99.9 466 Citation -------- Please cite the mitoKmer GitHub repository if you use this tool in published work. ]]></help> <citations> <citation type="bibtex"> @misc{githubmitokmer2, author = {Mammel, Mark}, title = {mitoKmer2}, year = {2024}, publisher = {GitHub}, journal = {GitHub repository}, url = {https://github.com/mmammel8/mitokmer2}, } </citation> </citations> </tool>
