view mitokmer.xml @ 5:8f4abeb27625 draft

planemo upload commit 3118e6e386354d664c85c47c67b75a92629d358c
author galaxytrakr
date Tue, 15 Sep 2026 18:31:11 +0000
parents ecf96ad08611
children e4d4fc748584
line wrap: on
line source

<tool id="mitokmer" name="mitoKmer" version="2.0+galaxy0.2" profile="21.05">
    <description>Identify metagenomic mitochondrial reads by k-mer database matching</description>
    <requirements>
        <container type="docker">quay.io/galaxytrakr/mitokmer:e57a559</container>
    </requirements>

    <command detect_errors="exit_code"><![CDATA[
        #import re
        #def fix_ext($ext):
            #set ext = $ext.replace('fastqsanger', 'fastq')
            #set ext = $ext.replace('fastqillumina', 'fastq')
            #return $ext
        #end def

        mkdir -p ./mitochondria7 ./jobs7m &&

        ## ── Link database files into working directory ────────────────────────
        ln -sf '${probe_db.fields.path}'/* ./mitochondria7/ &&

        ## ── Symlink kmerread7 binary into working directory ───────────────────
        ln -sf /usr/local/bin/kmerread7 ./kmerread7 &&

        ## ── Stage input reads and build jobs file ────────────────────────────
        #if $reads.reads_type == "single"
            mkdir -p ./reads &&
            #set sample_name = $reads.input.name.replace(' ', '_')
            #set read_count = 0
            #for $read in $reads.input
                #set ext = $fix_ext($read.ext)
                #set fname = re.sub('[^\w\-_.]', '_', $read.element_identifier) + '.' + $ext
                ln -sf '$read' './reads/${fname}' &&
                #set read_count = $read_count + 1
            #end for
            printf '%s\t%d\n' '${sample_name}' ${read_count} > ./jobs7m/jobs7m.txt &&
            #for $read in $reads.input
                #set ext = $fix_ext($read.ext)
                #set fname = re.sub('[^\w\-_.]', '_', $read.element_identifier) + '.' + $ext
                printf '%s\n' './reads/${fname}' >> ./jobs7m/jobs7m.txt &&
            #end for
        #else
            ## paired collection: forward + reverse
            mkdir -p ./reads &&
            #set sample_name = $reads.input.name.replace(' ', '_')
            #set fwd_ext = $fix_ext($reads.input.forward.ext)
            #set rev_ext = $fix_ext($reads.input.reverse.ext)
            #set base = re.sub('[^\w\-_.]', '_', $reads.input.element_identifier)
            #set fwd_fname = $base + '_1.' + $fwd_ext
            #set rev_fname = $base + '_2.' + $rev_ext
            ln -sf '$reads.input.forward' './reads/${fwd_fname}' &&
            ln -sf '$reads.input.reverse' './reads/${rev_fname}' &&
            printf '%s\t2\n' '${sample_name}' > ./jobs7m/jobs7m.txt &&
            printf '%s\n' './reads/${fwd_fname}' >> ./jobs7m/jobs7m.txt &&
            printf '%s\n' './reads/${rev_fname}' >> ./jobs7m/jobs7m.txt &&
        #end if

        ## ── Run the Python orchestrator ───────────────────────────────────────
        python3 /opt/mitokmer2/kmer_read_m7.py &&

        ## ── Copy output CSV to Galaxy output path ─────────────────────────────
        cp ./jobs7m/jobs7m.csv '${results_csv}'
    ]]></command>

    <inputs>
        <param name="probe_db" type="select" label="Mitochondrial k-mer probe database"
               help="Select a pre-installed mitochondrial k-mer probe database.
                     Databases are managed by your Galaxy administrator via the
                     mitokmer_probe_db data table.">
            <options from_data_table="mitokmer_probe_db">
                <filter type="sort_by" column="1" />
                <validator type="no_options" message="No mitochondrial k-mer probe databases are
                    currently installed. Please contact your Galaxy administrator." />
            </options>
        </param>

        <conditional name="reads">
            <param name="reads_type" type="select" label="Input read type">
                <option value="single">Single-end or unpaired reads / FASTA (list collection)</option>
                <option value="paired">Paired-end reads (paired collection)</option>
            </param>
            <when value="single">
                <param name="input" type="data_collection" collection_type="list"
                       format="fastqsanger,fastqsanger.gz,fastqillumina,fastqillumina.gz,fasta,fasta.gz"
                       label="Input reads (FASTQ or FASTA, gzipped or plain)"
                       help="Provide one or more single-end or unpaired read files as a Galaxy
                             list collection. For paired-end data, use the 'Paired-end reads'
                             option instead." />
            </when>
            <when value="paired">
                <param name="input" type="data_collection" collection_type="paired"
                       format="fastqsanger,fastqsanger.gz,fastqillumina,fastqillumina.gz"
                       label="Paired-end reads (paired collection)"
                       help="Provide a Galaxy paired collection containing forward (R1) and
                             reverse (R2) reads. Light quality trimming of read ends is
                             performed internally; pre-trimming is optional." />
            </when>
        </conditional>
    </inputs>

    <outputs>
        <data name="results_csv" format="csv" label="mitoKmer results for ${on_string}" />
    </outputs>

    <tests>
        <!-- Test 1: paired-end FASTQ via paired collection -->
        <test>
            <param name="probe_db" value="mitoch_probes_sample" />
            <conditional name="reads">
                <param name="reads_type" value="paired" />
                <param name="input">
                    <collection type="paired">
                        <element name="forward" value="test/Plodia_R1.fastq.gz" ftype="fastqsanger.gz" />
                        <element name="reverse" value="test/Plodia_R2.fastq.gz" ftype="fastqsanger.gz" />
                    </collection>
                </param>
            </conditional>
            <output name="results_csv">
                <assert_contents>
                    <has_text text="Plodia" />
                </assert_contents>
            </output>
        </test>
        <!-- Test 2: single-end FASTQ via list collection -->
        <test>
            <param name="probe_db" value="mitoch_probes_sample" />
            <conditional name="reads">
                <param name="reads_type" value="single" />
                <param name="input">
                    <collection type="list">
                        <element name="Plodia_R1" value="test/Plodia_R1.fastq.gz" ftype="fastqsanger.gz" />
                    </collection>
                </param>
            </conditional>
            <output name="results_csv">
                <assert_contents>
                    <has_text text="Plodia" />
                </assert_contents>
            </output>
        </test>
        <!-- Test 3: single FASTA via list collection -->
        <test>
            <param name="probe_db" value="mitoch_probes_sample" />
            <conditional name="reads">
                <param name="reads_type" value="single" />
                <param name="input">
                    <collection type="list">
                        <element name="Plodia_assembly" value="test/Plodia_assembly.fasta" ftype="fasta" />
                    </collection>
                </param>
            </conditional>
            <output name="results_csv">
                <assert_contents>
                    <has_text text="Plodia" />
                </assert_contents>
            </output>
        </test>
    </tests>

    <help><![CDATA[
**mitoKmer** — Metagenomic Mitochondrial Read Identification by K-mer Database
===============================================================================

Overview
--------
mitoKmer identifies the taxonomic origin of short-read shotgun sequencing data
by matching reads against a database of species-specific mitochondrial k-mer
probes. It reports the relative abundance of taxa at multiple taxonomic ranks
(e.g. order, species) along with the number of matching reads and supporting
unique k-mers.

Inputs
------
**Mitochondrial k-mer probe database**
    Select a pre-installed probe database from the dropdown. Databases are
    managed by your Galaxy administrator and registered in the
    ``mitokmer_probe_db`` data table. The database consists of a directory
    of ``.txt`` files.

**Input read type**
    Choose the input mode that matches your data:

    *Single-end or unpaired reads / FASTA (list collection)*
        Provide a Galaxy **list** collection containing one or more FASTQ
        (``fastqsanger``, ``fastqsanger.gz``) or FASTA (``fasta``,
        ``fasta.gz``) files. Use this for single-end sequencing data or
        assembled FASTA sequences.

    *Paired-end reads (paired collection)*
        Provide a Galaxy **paired** collection where the forward (R1) and
        reverse (R2) reads are paired together. Both files must be FASTQ
        format (``fastqsanger`` or ``fastqsanger.gz``).

    In all cases, light window-based quality trimming of read ends is
    performed internally; pre-trimming is optional.

Output
------
**Results CSV**
    A comma-separated file with one row per taxon detected. Columns:

    =========  ============================================================
    Column     Description
    =========  ============================================================
    Rank       Taxonomic rank (e.g. Order, Species)
    Taxon      Taxon name
    Reads      Number of reads matching this taxon
    Rel_Abund  Relative abundance (%)
    Unique_kmers  Number of unique k-mers supporting the match
    =========  ============================================================

    Only taxa with non-zero relative abundance are reported.

    Example output for the "Plodia" test sample::

        Rank    Taxon                   Reads   Rel_Abund(%)   Unique_kmers
        Order   Lepidoptera             10      0.1            329
        Species Plodia interpunctella   462     99.9           466

Citation
--------
Please cite the mitoKmer GitHub repository if you use this tool in published work.
    ]]></help>

    <citations>
        <citation type="bibtex">
@misc{githubmitokmer2,
  author    = {Mammel, Mark},
  title     = {mitoKmer2},
  year      = {2024},
  publisher = {GitHub},
  journal   = {GitHub repository},
  url       = {https://github.com/mmammel8/mitokmer2},
}
        </citation>
    </citations>
</tool>