view mitokmer.xml @ 4:ecf96ad08611 draft

planemo upload commit 26a6ad9e9ae7fdb967832a65f5173ea347da7f6c
author galaxytrakr
date Tue, 15 Sep 2026 12:02:36 +0000
parents dd206296acbf
children 8f4abeb27625
line wrap: on
line source

<tool id="mitokmer" name="mitoKmer" version="2.0+galaxy0.2" python_template_version="3.5" profile="21.05">
    <description>Identify metagenomic mitochondrial reads by k-mer database matching</description>
    <requirements>
        <container type="docker">quay.io/galaxytrakr/mitokmer:e57a559</container>
    </requirements>

    <command detect_errors="exit_code"><![CDATA[
        ## ── All paths are hardcoded in kmer_read_m7.py: ──────────────────────
        ##   database:  ./mitochondria7/<multiple .txt files>
        ##   jobs file: ./jobs7m/jobs7m.txt
        ##   output:    ./jobs7m/jobs7m.csv
        ##   binary:    ./kmerread7  (called as subprocess by the Python script)
        mkdir -p ./mitochondria7 ./jobs7m &&

        ## ── Link database files from the data table directory ─────────────────
        ## Symlink each file individually from the registered database directory.
        ## Note: glob must be outside quotes so the shell expands it correctly.
        ln -sf '${probe_db.fields.path}'/* ./mitochondria7/ &&

        ## ── Symlink kmerread7 into the working directory ──────────────────────
        ## kmer_read_m7.py calls ./kmerread7 (relative path), so it must exist
        ## in the Galaxy job working directory
        ln -sf /usr/local/bin/kmerread7 ./kmerread7 &&

        ## ── Write the jobs file ───────────────────────────────────────────────
        ## Format: sample_name <tab> num_files
        ##         /abs/path/to/read1
        ##         /abs/path/to/read2  ...
        ## Use len() to safely get collection size as an integer
        #set sample_name = $reads.name.replace(' ', '_')
        #set read_count = len($reads)
        printf '%s\t%d\n' '${sample_name}' ${read_count} > ./jobs7m/jobs7m.txt &&
        #for read in $reads
            printf '%s\n' '${read.file_name}' >> ./jobs7m/jobs7m.txt &&
        #end for

        ## ── Run the Python orchestrator (no arguments) ────────────────────────
        ## kmer_read_m7.py reads jobs7m/jobs7m.txt, calls ./kmerread7,
        ## and writes results to jobs7m/jobs7m.csv
        python3 /opt/mitokmer2/kmer_read_m7.py &&

        ## ── Copy output CSV to Galaxy output path ─────────────────────────────
        cp ./jobs7m/jobs7m.csv '${results_csv}'
    ]]></command>

    <inputs>
        <!-- Probe database directory selected from Galaxy data table -->
        <param name="probe_db"
               type="select"
               label="Mitochondrial k-mer probe database"
               help="Select a pre-installed mitochondrial k-mer probe database.
                     Databases are managed by your Galaxy administrator via the
                     mitokmer_probe_db data table.">
            <options from_data_table="mitokmer_probe_db">
                <filter type="sort_by" column="1" />
                <validator type="no_options"
                           message="No mitochondrial k-mer probe databases are
                                    currently installed. Please contact your
                                    Galaxy administrator." />
            </options>
        </param>

        <param name="reads"
               type="data_collection"
               collection_type="list"
               format="fastq,fastq.gz,fasta,fasta.gz"
               label="Input reads (FASTQ or FASTA, gzipped or plain)"
               help="Provide one or more read files for a single sample as a Galaxy
                     list collection.  Paired-end files (R1 + R2) should both be
                     included in the same collection.  Light quality trimming of
                     read ends is performed internally; pre-trimming is optional." />
    </inputs>

    <outputs>
        <data name="results_csv"
              format="csv"
              label="mitoKmer results for ${on_string}" />
    </outputs>

    <tests>
        <test>
            <param name="probe_db" value="mitoch_probes_sample" />
            <param name="reads">
                <collection type="list">
                    <element name="Plodia_R1" value="test/Plodia_R1.fastq.gz" />
                    <element name="Plodia_R2" value="test/Plodia_R2.fastq.gz" />
                </collection>
            </param>
            <output name="results_csv">
                <assert_contents>
                    <has_text text="Plodia" />
                </assert_contents>
            </output>
        </test>
    </tests>

    <help><![CDATA[
**mitoKmer** — Metagenomic Mitochondrial Read Identification by K-mer Database
===============================================================================

Overview
--------
mitoKmer identifies the taxonomic origin of short-read shotgun sequencing data
by matching reads against a database of species-specific mitochondrial k-mer
probes.  It reports the relative abundance of taxa at multiple taxonomic ranks
(e.g. order, species) along with the number of matching reads and supporting
unique k-mers.

Inputs
------
**Mitochondrial k-mer probe database**
    Select a pre-installed probe database from the dropdown.  Databases are
    managed by your Galaxy administrator and registered in the
    ``mitokmer_probe_db`` data table.  The database consists of a directory
    of supporting ``.txt`` files.

**Input reads**
    One or more FASTQ or FASTA files (gzipped or plain) for a single sample,
    supplied as a Galaxy list collection.  Both paired-end files (R1 and R2)
    should be included in the same collection.

    * Pre-trimming is optional — the tool performs simple window-based quality
      trimming of read ends internally.

Output
------
**Results CSV**
    A comma-separated file with columns: taxid, reads, abundance, uniq.
    Only taxa with non-zero relative abundance are reported.

    Example output for the "Plodia" test sample::

        Rank    Taxon                   Reads   Rel_Abund(%)   Unique_kmers
        Order   Lepidoptera             10      0.1            329
        Species Plodia interpunctella   462     99.9           466

Citation
--------
Please cite the mitoKmer GitHub repository if you use this tool in published work.
    ]]></help>

    <citations>
        <citation type="bibtex">
@misc{githubmitokmer2,
  author    = {Mammel, Mark},
  title     = {mitoKmer2},
  year      = {2024},
  publisher = {GitHub},
  journal   = {GitHub repository},
  url       = {https://github.com/mmammel8/mitokmer2},
}
        </citation>
    </citations>
</tool>