diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml new file mode 100644 index 0000000..1a6dd37 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -0,0 +1,28 @@ +name: Bug report +description: Report a problem running MetaPathways or using its reports. +title: "[Bug]: " +body: + - type: input + id: version + attributes: + label: MetaPathways version + description: Shown at the bottom of the explorer or by metapathways version. + validations: + required: true + - type: textarea + id: environment + attributes: + label: Installation and execution environment + description: For example, Mamba installation on Linux; local execution or Slurm. + - type: textarea + id: reproduce + attributes: + label: What happened? + description: Include the command or explorer steps, expected result, and actual result. + validations: + required: true + - type: textarea + id: logs + attributes: + label: Relevant logs or screenshots + description: Paste relevant error messages or attach a small example. diff --git a/.github/ISSUE_TEMPLATE/feature_request.yml b/.github/ISSUE_TEMPLATE/feature_request.yml new file mode 100644 index 0000000..c8d345c --- /dev/null +++ b/.github/ISSUE_TEMPLATE/feature_request.yml @@ -0,0 +1,18 @@ +name: Feature request +description: Suggest an improvement to MetaPathways or its explorer. +title: "[Feature]: " +body: + - type: textarea + id: need + attributes: + label: What would you like to do? + description: Describe the task and what currently makes it difficult. + validations: + required: true + - type: textarea + id: proposal + attributes: + label: Suggested improvement + description: Describe the behavior or output you would find useful. + validations: + required: true diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md new file mode 100644 index 0000000..00cae57 --- /dev/null +++ b/.github/pull_request_template.md @@ -0,0 +1,14 @@ +## Problem and resulting behavior + +Describe the concrete change and any compatibility implications. + +## Validation + +- Tested commit: +- Local tests and installed-package checks: +- Single/two-sample test workflow and checkpoint reuse: +- Report/explorer and CSV export: +- Optional Pathway Tools and Slurm checks (or explicitly untested): +- Nonpublishing Release workflow/artifact links: + +See [the tester checklist](https://hallamlab-metapathways.readthedocs.io/en/latest/pr-testing.html). Feature PRs target `dev`; keep them in draft until single-server and Slurm benchmark review plus independent tester sign-off are complete. Production promotion uses a separate `dev` → `main` PR. Do not include licensed installers/images, credentials, or production outputs. diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 0a7ccea..048bf69 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -5,10 +5,26 @@ on: tags: ['v*'] workflow_dispatch: inputs: + publish_anaconda: + description: 'Publish validated Conda package to Anaconda.org' + type: boolean + default: false + publish_quay: + description: 'Publish validated container to Quay' + type: boolean + default: false + publish_github: + description: 'Publish GitHub release (may trigger linked Zenodo archiving)' + type: boolean + default: false release_tag: - description: 'Existing tag to publish/recover (blank = build only)' + description: 'Existing version tag to build or publish (required for publishing from a branch)' type: string default: '' + test_containers: + description: 'Build/test Docker and SIF without enabling publication' + type: boolean + default: false source_run_id: description: 'Optional successful build run to reuse; requires release_tag' type: string @@ -23,7 +39,21 @@ concurrency: cancel-in-progress: false jobs: + controls: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + - name: Validate publication selections + env: + PUBLISH_ANACONDA: ${{ inputs.publish_anaconda || false }} + PUBLISH_QUAY: ${{ inputs.publish_quay || false }} + PUBLISH_GITHUB: ${{ inputs.publish_github || false }} + RELEASE_TAG_INPUT: ${{ inputs.release_tag }} + run: python3 scripts/check_release_controls.py build: + needs: controls runs-on: ubuntu-22.04 timeout-minutes: 90 defaults: @@ -57,10 +87,13 @@ jobs: if: inputs.source_run_id == '' run: | conda install --yes --override-channels -c conda-forge conda-build conda-index pyyaml 'setuptools>=83,<85' wheel - python -m pip install build + python -m pip install build pandas pybedtools pyfastx tqdm scipy - name: Test release controls if: inputs.source_run_id == '' run: python -m unittest discover -s tests/release -v + - name: Test scheduling, containers, and read mapping + if: inputs.source_run_id == '' + run: python -m unittest discover -s tests -p 'test_*.py' -v - name: Build and test release if: inputs.source_run_id == '' env: @@ -102,8 +135,9 @@ jobs: if-no-files-found: ignore github-release: - needs: build - if: github.ref_type == 'tag' || inputs.release_tag != '' + environment: release + needs: [build, containers] + if: github.event_name == 'workflow_dispatch' && inputs.publish_github runs-on: ubuntu-24.04 permissions: contents: write @@ -129,8 +163,9 @@ jobs: run: python scripts/release.py github-release "$RELEASE_TAG" anaconda: - needs: [build, github-release] - if: (github.ref_type == 'tag' || inputs.release_tag != '') && vars.PUBLISH_CONDA == 'true' + needs: build + if: github.event_name == 'workflow_dispatch' && inputs.publish_anaconda + environment: release runs-on: ubuntu-24.04 steps: - uses: actions/checkout@v7 @@ -154,18 +189,19 @@ jobs: - name: Upload validated package shell: bash -el {0} env: + ANACONDA_OWNER: ${{ vars.ANACONDA_OWNER || 'hallamlab' }} BINSTAR_API_TOKEN: ${{ secrets.ANACONDA_API_TOKEN }} RELEASE_TAG: ${{ inputs.release_tag || github.ref_name }} run: | if [[ -z "$BINSTAR_API_TOKEN" ]]; then - echo "::error::Set the ANACONDA_API_TOKEN secret before enabling PUBLISH_CONDA." + echo "::error::Set the ANACONDA_API_TOKEN secret before selecting publish_anaconda." exit 1 fi python scripts/release.py upload-conda --ref "$RELEASE_TAG" containers: - needs: [build, github-release] - if: github.ref_type == 'tag' || inputs.release_tag != '' + needs: build + if: inputs.test_containers || inputs.publish_quay || inputs.publish_github runs-on: ubuntu-24.04 timeout-minutes: 90 permissions: @@ -190,7 +226,14 @@ jobs: - name: Build Docker and Apptainer; validate containers env: RELEASE_TAG: ${{ inputs.release_tag || github.ref_name }} - run: python scripts/release.py container-build --ref "$RELEASE_TAG" + RELEASE_REF_TYPE: ${{ github.ref_type }} + RELEASE_TAG_INPUT: ${{ inputs.release_tag }} + run: | + if [[ "$RELEASE_REF_TYPE" == tag || -n "$RELEASE_TAG_INPUT" ]]; then + python scripts/release.py container-build --ref "$RELEASE_TAG" + else + python scripts/release.py container-build + fi - name: Scan container for fixable high and critical vulnerabilities run: | mkdir -p dist/security @@ -216,6 +259,13 @@ jobs: name: container-security-report path: dist/security/ if-no-files-found: ignore + - name: Export tested Docker image for independent publication + run: | + set -o pipefail + image=$(python -c 'import json; print(json.load(open("dist/containers/container-validation.json"))["image"])') + docker save "$image" | gzip > dist/containers/docker-image.tar.gz + cd dist/containers + sha256sum docker-image.tar.gz >> container-SHA256SUMS - name: Retain container assets and diagnostics if: always() uses: actions/upload-artifact@v7.0.1 @@ -223,21 +273,73 @@ jobs: name: container-assets path: dist/containers/ if-no-files-found: warn + + github-containers: + needs: [containers, github-release] + if: github.event_name == 'workflow_dispatch' && inputs.publish_github + permissions: + contents: write + environment: release + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v7 + with: + fetch-depth: 0 + persist-credentials: false + - uses: actions/setup-python@v6 + with: + python-version: '3.11' + - uses: actions/download-artifact@v8.0.1 + with: + name: release-assets + path: dist/release + - uses: actions/download-artifact@v8.0.1 + with: + name: container-assets + path: dist/containers - name: Attach validated SIF and checksums to GitHub release env: GH_TOKEN: ${{ github.token }} GH_REPO: ${{ github.repository }} RELEASE_TAG: ${{ inputs.release_tag || github.ref_name }} run: python scripts/release.py container-attach --ref "$RELEASE_TAG" + + quay: + needs: containers + if: github.event_name == 'workflow_dispatch' && inputs.publish_quay + environment: release + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v7 + with: + fetch-depth: 0 + persist-credentials: false + - uses: actions/setup-python@v6 + with: + python-version: '3.11' + - uses: actions/download-artifact@v8.0.1 + with: + name: release-assets + path: dist/release + - uses: actions/download-artifact@v8.0.1 + with: + name: container-assets + path: dist/containers + - name: Verify and load the tested Docker image + env: + RELEASE_TAG: ${{ inputs.release_tag || github.ref_name }} + run: | + python scripts/release.py container-verify --ref "$RELEASE_TAG" + docker load -i dist/containers/docker-image.tar.gz - name: Publish Docker image to Quay - if: vars.PUBLISH_QUAY == 'true' env: + QUAY_REPOSITORY: ${{ vars.QUAY_REPOSITORY || 'quay.io/hallamlab/metapathways' }} QUAY_USERNAME: ${{ secrets.QUAY_USERNAME }} QUAY_PASSWORD: ${{ secrets.QUAY_PASSWORD }} RELEASE_TAG: ${{ inputs.release_tag || github.ref_name }} run: | if [[ -z "$QUAY_USERNAME" || -z "$QUAY_PASSWORD" ]]; then - echo "::error::Set QUAY_USERNAME and QUAY_PASSWORD secrets before enabling PUBLISH_QUAY." + echo "::error::Set QUAY_USERNAME and QUAY_PASSWORD secrets before selecting publish_quay." exit 1 fi export DOCKER_CONFIG="$RUNNER_TEMP/metapathways-docker-auth" @@ -246,7 +348,6 @@ jobs: printf '%s' "$QUAY_PASSWORD" | docker login quay.io --username "$QUAY_USERNAME" --password-stdin python scripts/release.py container-push --ref "$RELEASE_TAG" - name: Synchronize Quay repository overview - if: vars.PUBLISH_QUAY == 'true' env: QUAY_API_TOKEN: ${{ secrets.QUAY_API_TOKEN }} run: | diff --git a/.github/workflows/smoke.yml b/.github/workflows/smoke.yml index 6c3b110..4cc82db 100644 --- a/.github/workflows/smoke.yml +++ b/.github/workflows/smoke.yml @@ -2,9 +2,9 @@ name: Package and documentation smoke tests on: push: - branches: [dev] + branches: [master, main, dev, "feat/**"] pull_request: - branches: [dev] + branches: [master, main, dev] workflow_dispatch: permissions: @@ -21,21 +21,47 @@ jobs: - uses: actions/setup-python@v6 with: python-version: '3.11' - - name: Install build, documentation, and CLI import dependencies + - name: Install build and CLI import dependencies run: >- python -m pip install build 'setuptools>=83,<85' wheel - pandas pybedtools pyfastx tqdm scipy -r docs/requirements.txt + pandas pybedtools pyfastx tqdm scipy pyyaml cffconvert + - name: Validate software citation metadata + run: cffconvert --validate - name: Test release controls run: python -m unittest discover -s tests/release -v + - name: Test scheduling, containers, and read mapping + run: python -m unittest discover -s tests -p 'test_*.py' -v - name: Build source and wheel distributions run: python -m build --no-isolation - name: Check installed CLI outside the source directory run: | - python -m pip install --no-deps dist/*.whl + python -m pip install dist/*.whl + cp scripts/check_installed_assets.py "$RUNNER_TEMP/check_installed_assets.py" cd "$RUNNER_TEMP" + python check_installed_assets.py metapathways version - for command in run build_db mag_split ptools; do + magsplitter --help + python -c "import camelot_frs" + metapathways prepare_test -o test + for command in prepare_test run analysis_wf build_db mag_split build_pt ptools report; do metapathways "$command" -h done - - name: Build documentation with warnings treated as errors - run: sphinx-build -W --keep-going -b html docs/src docs/build + - name: Check GitHub documentation and CLI reference + run: python scripts/check_docs.py + + documentation: + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + - uses: actions/setup-python@v6 + with: + python-version: '3.11' + - name: Install documentation dependencies + run: python -m pip install -r docs/requirements.txt + - name: Build the complete guide + run: python -m sphinx -W --keep-going -b html docs docs/_build/html + - name: Check rendered links + run: python scripts/check_docs_html.py docs/_build/html --readme README.md diff --git a/.github/workflows/workflow-style.yml b/.github/workflows/workflow-style.yml new file mode 100644 index 0000000..ef9297c --- /dev/null +++ b/.github/workflows/workflow-style.yml @@ -0,0 +1,13 @@ +name: Workflow figure style +on: + pull_request: + push: +permissions: + contents: read +jobs: + mp-nodal-style: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - name: Check MP nodal workflow figures + run: python3 scripts/check_workflow_style.py diff --git a/.gitignore b/.gitignore index dd4943d..a6ac4de 100644 --- a/.gitignore +++ b/.gitignore @@ -85,3 +85,7 @@ global_errors_warnings.txt # Local manuscript figures and their export tools /docs/figures/ + +# Nextflow local metadata; invocation copies are retained with analysis outputs. +/.nextflow/ +/.nextflow.log* diff --git a/.readthedocs.yaml b/.readthedocs.yaml index c9e01cc..aae24d9 100644 --- a/.readthedocs.yaml +++ b/.readthedocs.yaml @@ -1,31 +1,11 @@ -# Read the Docs configuration file for Sphinx projects -# See https://docs.readthedocs.io/en/stable/config-file/v2.html for details - -# Required version: 2 - -# Set the OS, Python version and other tools you might need build: - os: ubuntu-22.04 + os: ubuntu-24.04 tools: - python: "3.10" - # You can also specify other tool versions: - # nodejs: "19" - # rust: "1.64" - # golang: "1.19" - -# Build documentation in the "docs/" directory with Sphinx + python: "3.11" sphinx: - configuration: docs/src/conf.py - -# Optionally build your docs in additional formats such as PDF and ePub -# formats: -# - pdf -# - epub - -# Optional but recommended, declare the Python requirements required -# to build your documentation -# See https://docs.readthedocs.io/en/stable/guides/reproducible-builds.html + configuration: docs/conf.py + fail_on_warning: true python: - install: - - requirements: docs/requirements.txt + install: + - requirements: docs/requirements.txt diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..b2a4722 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,16 @@ +# Repository instructions + +## Workflow figure aesthetic — standing user preference + +Read `docs/WORKFLOW_STYLE.md` before changing workflow diagrams, their generators, +exports or documentation embeds. Preserve the MetaPathways nodal publication +style: numbered square module spine, diamond compute nodes, circular input / +data / output nodes, Times serif labels, thin black connectors and the shared +blue / gray / green palette. Update scientific content within that layout. +Do not replace these figures with stacked text cards or generic box flowcharts. +A style change requires an explicit user request. + +Rebuild source-linked SVG / PDF / preview artifacts together, run +`python3 scripts/check_workflow_style.py`, and visually inspect the result. +This governs workflow figures; preserve the separately configured study palettes +for scientific plots. Apply this convention to new workflow figures too. diff --git a/CITATION.cff b/CITATION.cff new file mode 100644 index 0000000..87af383 --- /dev/null +++ b/CITATION.cff @@ -0,0 +1,44 @@ +cff-version: 1.2.0 +message: "Please cite this software and record the release version or Git commit used in your analysis." +type: software +title: MetaPathways +abstract: "A metagenomic analysis pipeline for functional and taxonomic annotation, abundance estimation, genome-bin splitting, pathway inference, and interactive exploration of linked result tables." +repository-code: "https://github.com/hallamlab/MetaPathways" +url: "https://github.com/hallamlab/MetaPathways" +license: MIT +# Authors and affiliations follow the MetaPathways v3.5 application note. +# Preserve manuscript order; do not substitute GitHub contributor profiles. +authors: + - family-names: McLaughlin + given-names: Ryan J. + affiliation: "Graduate Program in Bioinformatics, University of British Columbia, Vancouver, BC, Canada V6T 1Z4" + - family-names: Liu + given-names: Tony X. + affiliation: "Graduate Program in Bioinformatics, University of British Columbia, Vancouver, BC, Canada V6T 1Z4" + - family-names: Altman + given-names: Tomer + affiliation: "Altman Analytics LLC, Piedmont, CA 94611, USA" + - family-names: Nallan + given-names: Aditi N. + affiliation: "Graduate Program in Bioinformatics, University of British Columbia, Vancouver, BC, Canada V6T 1Z4" + - family-names: Hahn + given-names: Aria S. + affiliation: "Department of Microbiology & Immunology, University of British Columbia, Vancouver, BC, Canada V6T 1Z3; Ecosystem Services, Commercialization and Entrepreneurship (ECOSCOPE), University of British Columbia, Vancouver, BC. Canada V6T 1Z3" + - family-names: Anstett + given-names: Julia + affiliation: "Genome Sciences and Technology Training Program, University of British Columbia, Vancouver, BC, Canada V6T 1Z3" + - family-names: Morgan-Lang + given-names: Connor + affiliation: "Graduate Program in Bioinformatics, University of British Columbia, Vancouver, BC, Canada V6T 1Z4; Department of Microbiology & Immunology, University of British Columbia, Vancouver, BC, Canada V6T 1Z3" + - family-names: Konwar + given-names: Kishori M. + affiliation: "Computer Science and Artificial Intelligence Laboratory, Massachusetts Institute of Technology, Cambridge, MA 02139, USA; Meta, Cambridge, MA 02142, USA" + - family-names: Hallam + given-names: Steven J. + affiliation: "Graduate Program in Bioinformatics, University of British Columbia, Vancouver, BC, Canada V6T 1Z4; Department of Microbiology & Immunology, University of British Columbia, Vancouver, BC, Canada V6T 1Z3; Ecosystem Services, Commercialization and Entrepreneurship (ECOSCOPE), University of British Columbia, Vancouver, BC. Canada V6T 1Z3; Genome Sciences and Technology Training Program, University of British Columbia, Vancouver, BC, Canada V6T 1Z3; Bradshaw Research Institute for Minerals and Mining (BRIMM), University of British Columbia, Vancouver, BC, Canada V6T 1Z4; Life Sciences Institute, University of British Columbia, Vancouver, BC, Canada V6T 1Z3" +keywords: + - metagenomics + - functional annotation + - pathways + - Nextflow +version: "4.0.0" diff --git a/MANIFEST.in b/MANIFEST.in index 9a73a6e..ca9f217 100644 --- a/MANIFEST.in +++ b/MANIFEST.in @@ -13,5 +13,34 @@ recursive-include scripts *.py recursive-include conda_recipe *.py *.yaml include docker/conda_base.yml recursive-include tests/release *.py +include tests/test_read_mapping.py +include tests/test_pt_container.py +include tests/test_nextflow.py include docker/Dockerfile.release include docker/README.quay.md + +recursive-include metapathways/report_assets *.html *.js *.css +include tests/test_reporting.py +recursive-include docs *.md + +include tests/test_analysis_workflow.py +include tests/test_rna_abundance.py +include tests/test_metacyc_db.py +include tests/test_fast_isolation.py +include tests/test_pt_sequences.py + +include CITATION.cff + +recursive-include tests *.py +recursive-include docs/validation *.json + +include requirements-workflow.txt + +include .readthedocs.yaml +recursive-include docs/assets *.css +recursive-include docs/templates *.html +recursive-include docs/assets *.bib + +recursive-include docs/assets *.svg + +recursive-include docs/diagrams *.mmd *.json diff --git a/Makefile b/Makefile index aec4b77..5d84101 100644 --- a/Makefile +++ b/Makefile @@ -194,7 +194,7 @@ deploy-conda: release-upload-conda ### Docs: docs-local: - sphinx-build ./docs/src ./docs/build + $(PYTHON) scripts/check_docs.py ### Build & Install Extensions ## diff --git a/README.md b/README.md index 540ce8b..7f8b892 100644 --- a/README.md +++ b/README.md @@ -1,104 +1,68 @@ # MetaPathways -Functional and taxonomic annotation of environmental genomes, with community- and population-level pathway inference. - -## Quickstart - -**Linux x86-64 · Conda/Mamba · included K12 example.** Install and run in a writable environment with internet access: - -```bash -mamba create -n metapathways_env --override-channels \ - -c hallamlab -c conda-forge -c bioconda metapathways=3.5.1 -mamba activate metapathways_env -metapathways --version - -mkdir -p ~/metapathways-review -cd ~/metapathways-review -metapathways build_db --test -metapathways run --test -``` - -**Check the result:** outputs are in `test/k12_test/`. Inspect -`metapathways_steps_log.txt` for successful stages and `errors_warnings_log.txt` -for problems; do not rely on the exit code alone. Key outputs are -`results/annotation_table/k12_test.functional_and_taxonomic_table.txt`, -`genbank/k12_test.gbk`, and `results/rpkm/k12_test.contig_counts.tsv`. +## Abstract -The example includes small K12 FASTA/FASTQ and SwissProt/SILVA fixtures. Database -preparation downloads ExPASy enzyme records and NCBI taxonomy and writes indexes -inside the installed package. This is an installation test; it does not reproduce -the manuscript's CAMI2 benchmark or run Pathway Tools. +The development of high-throughput sequencing technologies over the past decade has generated a tidal wave of environmental sequence information from a variety of natural and human engineered ecosystems. The resulting flood of information into public databases and archived sequencing projects has exponentially expanded computational resource requirements rendering most local homology-based search methods inefficient. MetaPathways v1.0 is a modular annotation and analysis pipeline for constructing environmental Pathway/Genome Databases (ePGDBs) from environmental sequence information capable of using the Sun Grid engine for external resource partitioning. However, a command-line interface and facile task management introduced user activation barriers with concomitant decrease in fault tolerance. -[Docker / Apptainer instructions](docker/README.quay.md) · -[Release downloads](https://github.com/hallamlab/MetaPathways/releases) · -[Full usage](https://metapathways.readthedocs.io/en/latest/usage.html) · -[Benchmark provenance](docs/src/reproducibility.rst) +MetaPathways has since advanced as a modular tool, deepening our understanding of microbial metabolism at various biological levels. With this release, we have addressed previous challenges in modularity and database management. v3.5 enhances user accessibility through streamlined installation via package indexes or containers, refined modules, and interface upgrades. It boasts updated algorithm support for sequence feature prediction, annotation, metabolic inference, and coverage metrics. Tested on mock community data, Metapathways v3.5 demonstrates improved performance and usability. With automated installation and database management, this open-source tool makes advanced metagenomic analysis more accessible. Metapathways v3.5 represents a significant step forward in automated, comprehensive metagenomic analysis, facilitating a deeper exploration of microbial interactions and metabolic functions in environmental genomics. -[![Version 3.5.1](https://img.shields.io/badge/Version-3.5.1-blue.svg)](https://github.com/hallamlab/MetaPathways/releases) -[![Python 3.11](https://img.shields.io/badge/Python-3.11-blue.svg)](https://www.python.org/) -[![MIT license](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE) +**[Full user guide](https://hallamlab-metapathways.readthedocs.io/en/latest/index.html)** · [Workflow test](https://hallamlab-metapathways.readthedocs.io/en/latest/test.html) · [Issues and feature requests](https://github.com/hallamlab/MetaPathways/issues) -## Install this source revision +## Quick start -For development on Linux x86-64 with Python 3.11: +On Linux x86-64, install MetaPathways and its workflow dependencies with Mamba: ```bash -git clone https://github.com/hallamlab/MetaPathways.git -cd MetaPathways -mamba env create -f docker/conda_base.yml -mamba activate metapathways -mamba install -c conda-forge pip wheel 'setuptools>=83,<85' -python -m pip install --no-deps --no-build-isolation . +mamba create -n metapathways --override-channels --strict-channel-priority \ + -c hallamlab -c conda-forge -c bioconda metapathways=4.0.0 +conda activate metapathways ``` -Tagged releases validate packages and containers through the -[release workflow](docs/releasing.md). Record the release build or container digest -used for a review. Optional MAG splitting and Pathway Tools commands require -additional dependencies; see the [full usage documentation](https://metapathways.readthedocs.io/en/latest/usage.html). - -## Inputs, outputs, and reproducibility +Prefer a container or a source checkout? Follow the [Docker / Apptainer guide](https://hallamlab-metapathways.readthedocs.io/en/latest/containers.html) or [GitHub installation guide](https://hallamlab-metapathways.readthedocs.io/en/latest/installation.html). -`metapathways run` accepts nucleotide FASTA or amino-acid FASTA (`--input_format -fasta-amino`). Paired or interleaved FASTQ reads can be supplied for abundance -estimation. The current CLI does not accept GFF or GenBank as primary inputs. +## Try the bundled three-sample dataset -Results include annotation tables, annotated GFF/GenBank files, predicted -sequences, abundance tables when reads are supplied, run statistics, and Pathway -Tools input files. Reference databases are prepared separately with -`metapathways build_db`; see [reproducibility notes](docs/src/reproducibility.rst) -for dependency versions, output locations, and the manuscript benchmark provenance. -The small included example is an installation test, not a reproduction of the -manuscript's CAMI2 performance benchmark. - -## Abstract - -The development of high-throughput sequencing technologies over the past decade has generated a tidal wave of environmental sequence information from a variety of natural and human engineered ecosystems. The resulting flood of information into public databases and archived sequencing projects has exponentially expanded computational resource requirements rendering most local homology-based search methods inefficient. MetaPathways v1.0 is a modular annotation and analysis pipeline for constructing environmental Pathway/Genome Databases (ePGDBs) from environmental sequence information capable of using the Sun Grid engine for external resource partitioning. However, a command-line interface and facile task management introduced user activation barriers with concomitant decrease in fault tolerance. +```bash +metapathways prepare_test -o ~/mp-test +cd ~/mp-test +metapathways build_db --test -d MPDB +metapathways analysis_wf \ + --manifest cami-test/all.tsv -o all -d MPDB \ + --annotation_dbs swissprot_test \ + --rRNA_refdbs SILVA_SSU_test SILVA_LSU_test \ + --skip_ptools --threads 4 --memory '4 GB' --max_tasks 2 +metapathways report -o all --serve --no-browser --port 8765 +``` -MetaPathways has since advanced as a modular tool, deepening our understanding of microbial metabolism at various biological levels. With this release, we have addressed previous challenges in modularity and database management. v3.5 enhances user accessibility through streamlined installation via package indexes or containers, refined modules, and interface upgrades. It boasts updated algorithm support for sequence feature prediction, annotation, metabolic inference, and coverage metrics. Tested on mock community data, Metapathways v3.5 demonstrates improved performance and usability. With automated installation and database management, this open-source tool makes advanced metagenomic analysis more accessible. Metapathways v3.5 represents a significant step forward in automated, comprehensive metagenomic analysis, facilitating a deeper exploration of microbial interactions and metabolic functions in environmental genomics. +Open the report server's printed URL; for remote runs, follow the [SSH viewing guide](https://hallamlab-metapathways.readthedocs.io/en/latest/reports-tutorial.html#view-a-remote-report-through-ssh). The bundled 2.4 MiB CAMI II subset ([Meyer et al., 2022](#cami-references)) exercises annotation, read abundance, genome splitting, reports and exploration. Database preparation downloads supporting reference records. Pathway inference requires your own license and is skipped in this test. See the [test walkthrough](https://hallamlab-metapathways.readthedocs.io/en/latest/test.html) for validation and expected results. -## Team and repository +## Run your own data -**Current Team:** Ryan J. McLaughlin, Tony X. Liu, Tomer Altman, Aditi N. Nallan, Aria S. Hahn, Julia Anstett, Connor Morgan-Lang, Kishori M. Konwar, and Steven J. Hallam +```bash +metapathways build_db -d ~/MPDB --func swissprot -a fast +metapathways run -i /path/to/assembly.fasta -o results -d ~/MPDB --threads 8 +``` -**Previous Team Members:** Niels W. Hanson and Shang-Ju Wu +**For PGDBs, follow the [Pathway Tools installation guide](https://hallamlab-metapathways.readthedocs.io/en/latest/pathway-tools.html) first.** Then follow the [complete workflow guide](https://hallamlab-metapathways.readthedocs.io/en/latest/analysis.html) to include reads and genome maps. Test references are for testing only. -The canonical source is [hallamlab/MetaPathways](https://github.com/hallamlab/MetaPathways). Earlier code is preserved separately in [MetaPathways-legacy](https://github.com/hallamlab/MetaPathways-legacy). +## Workflow -## [Documentation](https://metapathways.readthedocs.io/en/latest/) +[![MetaPathways appnote-style workflow: sequence processing, annotation, optional pathways and abundance, reports and explorer.](docs/assets/workflow-main.svg?v=mp-tools-20261008)](https://hallamlab-metapathways.readthedocs.io/en/latest/workflow.html) -## Support +The **[full user guide](https://hallamlab-metapathways.readthedocs.io/)** covers inputs, databases, Pathway Tools, local and Slurm resources, all commands, reporting and troubleshooting. [Detailed workflow and tool citations](https://hallamlab-metapathways.readthedocs.io/en/latest/detailed-workflow.html). -[Technical questions, bug reports, and general inquires can be made here.](https://github.com/hallamlab/MetaPathways/issues) +## Team, support and citation -## Citation +Current team: Ryan J. McLaughlin, Tony X. Liu, Tomer Altman, Aditi N. Nallan, Aria S. Hahn, Julia Anstett, Connor Morgan-Lang, Kishori M. Konwar and Steven J. Hallam. Previous contributors include Niels W. Hanson and Shang-Ju Wu. -If you use MetaPathways in your research, please cite the following article: +Source: [hallamlab/MetaPathways](https://github.com/hallamlab/MetaPathways). Historical code: [MetaPathways-legacy](https://github.com/hallamlab/MetaPathways-legacy). Questions and bug reports: [GitHub issues](https://github.com/hallamlab/MetaPathways/issues). License: [MIT](LICENSE), with bundled third-party license notices retained. -> Ryan J. McLaughlin, Tony X. Liu, Tomer Altman, Aditi N. Nallan, Aria S. Hahn, Julia Anstett, Connor Morgan-Lang, Kishori M. Konwar, Steven J. Hallam. *MetaPathways v3.5: Modularity and Scalability Improvements for Pathway Inference from Environmental Genomes* bioRxiv (2024): 2024-06. [doi: https://doi.org/10.1101/2024.06.04.597460](https://doi.org/10.1101/2024.06.04.597460) +Please cite: +> McLaughlin RJ, Liu TX, Altman T, Nallan AN, Hahn AS, Anstett J, Morgan-Lang C, Konwar KM, Hallam SJ. *MetaPathways v3.5: Modularity and Scalability Improvements for Pathway Inference from Environmental Genomes*. bioRxiv (2024). [doi:10.1101/2024.06.04.597460](https://doi.org/10.1101/2024.06.04.597460). -## Maintainer releases +## CAMI references -Use [the release controller and CI workflow](docs/releasing.md) to set a version, -build and test packages, and publish downloadable GitHub releases and optional -Anaconda.org packages. The source version is declared in `metapathways/_version.py`. +- **CAMI:** Sczyrba, A., Hofmann, P., Belmann, P., et al. (2017). *Critical Assessment of Metagenome Interpretation—a benchmark of metagenomics software*. **Nature Methods 14**(11), 1063–1071. [DOI: 10.1038/nmeth.4458](https://doi.org/10.1038/nmeth.4458). [CAMI project website](https://cami-challenge.org/). +- **CAMI II:** Meyer, F., Fritz, A., Deng, Z.-L., et al. (2022). *Critical Assessment of Metagenome Interpretation: the second round of challenges*. **Nature Methods 19**(4), 429–440. [DOI: 10.1038/s41592-022-01431-4](https://doi.org/10.1038/s41592-022-01431-4). [CAMI project website](https://cami-challenge.org/). +- **Source dataset for the MP test subset:** CAMI II multi-sample human microbiome dataset. [Dataset DOI: 10.4126/FRL01-006425518](https://doi.org/10.4126/FRL01-006425518). The bundled inputs are selected and cropped subsets of this collection; their exact transformations and file hashes are recorded in the bundle provenance. diff --git a/changelog.md b/changelog.md index c9b9d6c..7955167 100644 --- a/changelog.md +++ b/changelog.md @@ -1,4 +1,15 @@ # Changelog + +## 4.0.0 + +- Nextflow controllers for local and Slurm workflows, multi-sample manifests, resource budgets, database builds and licensed Pathway Tools container preparation. +- Read-mapping/checkpoint fixes and sequence-backed PGDB staging, including valid zero-pathway exports. +- Compact results with node-local execution, archived diagnostics and resumable checkpoints. +- Licensed MetaCyc preparation, reaction compatibility screening and explicit taxonomic-scope controls. +- SwissProt, UniRef and eggNOG taxonomy scoped to each annotation database and target; independent within-database LCA. +- Searchable reports with a sample-level landing page, wide abundance tables, source provenance, CSV exports and GitHub feedback links. +- Bundled three-sample CAMI test inputs and documented local validation. + ### Updates **November 27, 2014**: [MetaPathways v2.5 released](https://github.com/hallamlab/metapathways2/releases/tag/v2.5) with upgrades to the pipeline: diff --git a/conda_recipe/compile_recipe.py b/conda_recipe/compile_recipe.py index 184746c..1f0c494 100644 --- a/conda_recipe/compile_recipe.py +++ b/conda_recipe/compile_recipe.py @@ -23,7 +23,15 @@ def main(): deps = yaml.safe_load((ROOT / "docker/conda_base.yml").read_text()) if not all(isinstance(d, str) for d in deps["dependencies"]): parser.error("Conda runtime dependencies must be explicit package strings, not pip/VCS entries.") + helper_sources = [] + for line in (ROOT / "requirements-workflow.txt").read_text().splitlines(): + if not line.strip() or line.startswith("#"): + continue + name, source = line.split(" @ ", 1) + url, checksum = source.split("#sha256=", 1) + helper_sources.append(f" - url: {url}\n sha256: {checksum}\n folder: helper-sources/{name}") replacements = { + "": "\n".join(helper_sources), "": NAME, "": VERSION, "": "\n".join(f" - {e}" for e in ENTRY_POINTS), diff --git a/conda_recipe/meta_template.yaml b/conda_recipe/meta_template.yaml index 660719a..3dec7bd 100644 --- a/conda_recipe/meta_template.yaml +++ b/conda_recipe/meta_template.yaml @@ -3,14 +3,17 @@ package: version: source: - url: - sha256: + - url: + sha256: + build: # The bundled FAST/metacount executables target Linux x86-64. skip: true # [not (linux and x86_64)] - number: 1 - script: "{{ PYTHON }} -m pip install . --no-deps --no-build-isolation -vv" + number: 0 + script: + - "{{ PYTHON }} -m pip install helper-sources/magsplitter helper-sources/camelot-frs --no-deps --no-build-isolation -vv" + - "{{ PYTHON }} -m pip install . --no-deps --no-build-isolation -vv" entry_points: @@ -18,9 +21,18 @@ test: imports: - metapathways commands: + - magsplitter --help + - python -c "import camelot_frs" + - metapathways prepare_test --help - metapathways version - metapathways run --help - metapathways build_db --help + - metapathways analysis_wf --help + - metapathways build_pt --help + - metapathways mag_split --help + - metapathways ptools --help + - metapathways report --help + - nextflow -version requirements: host: @@ -28,6 +40,7 @@ requirements: - pip - setuptools >=83,<85 - wheel + - setuptools-scm >=6.2 run: - libgcc - libstdcxx @@ -36,5 +49,7 @@ requirements: about: home: https://github.com/hallamlab/MetaPathways/ summary: metagenomics pipeline - license: MIT - license_file: LICENSE + license: MIT AND MPL-2.0 + license_file: + - LICENSE + - helper-sources/camelot-frs/LICENSE diff --git a/dev/abund_calc.py b/dev/abund_calc.py index 9629248..c756de8 100644 --- a/dev/abund_calc.py +++ b/dev/abund_calc.py @@ -42,60 +42,58 @@ def calculate_tpm(counts, gene_lengths): return tpm_values +def abundance_table(counts_file, gtf_file, gene_lengths_file=None): + """One abundance row per featureCounts gene ID, using its union feature length.""" + counts_df = pd.read_csv(counts_file, sep="\t", index_col=0, comment="#") + if counts_df.index.has_duplicates: + raise ValueError("featureCounts contains duplicate gene IDs") + if 'Length' in counts_df: + lengths = pd.to_numeric(counts_df['Length'], errors='raise') + elif gene_lengths_file: + legacy = pd.read_csv(gene_lengths_file, sep="\t", index_col=0, header=None) + if legacy.index.has_duplicates or set(legacy.index) != set(counts_df.index): + raise ValueError("Legacy gene lengths must contain each counted gene ID exactly once") + lengths = pd.to_numeric(legacy.iloc[:, 0].reindex(counts_df.index), errors='raise') + else: + raise ValueError("Missing featureCounts Length column") + counts = pd.to_numeric(counts_df.iloc[:, -1], errors='raise') + if not np.isfinite(lengths).all() or (lengths <= 0).any(): + raise ValueError("Gene lengths must be finite and positive") + if not np.isfinite(counts).all() or (counts < 0).any(): + raise ValueError("Counts must be finite and nonnegative") + results = pd.DataFrame({ + 'Gene_ID': counts_df.index, 'Count': counts.to_numpy(), + 'Length': lengths.to_numpy(), + 'RPKM': calculate_rpkm(counts, lengths) if counts.sum() else np.zeros(len(counts)), + 'TPM': calculate_tpm(counts, lengths) if counts.sum() else np.zeros(len(counts)), + }) + columns = ['seqname', 'source', 'feature', 'start', 'end', 'score', 'strand', 'frame', 'attributes'] + gtf = pd.read_csv(gtf_file, sep="\t", comment="#", header=None, names=columns, + dtype=str, keep_default_na=False) + gtf['Gene_ID'] = gtf['attributes'].str.extract(r'(?:^|;)\s*gene_id "([^"]+)"') + if gtf['Gene_ID'].isna().any(): + raise ValueError("GTF row is missing gene_id") + missing = set(counts_df.index) - set(gtf['Gene_ID']) + uncounted = set(gtf['Gene_ID']) - set(counts_df.index) + if missing or uncounted: + raise ValueError(f'Gene IDs differ between counts and GTF: missing from GTF={sorted(missing)[:5]}, missing from counts={sorted(uncounted)[:5]}') + if gtf['Gene_ID'].duplicated().any(): + duplicates = gtf.loc[gtf['Gene_ID'].duplicated(), 'Gene_ID'].unique()[:5] + raise ValueError(f"GTF contains duplicate gene IDs: {list(duplicates)}. Regenerate annotations with corrected RNA IDs; do not pool distinct loci.") + return results.merge(gtf, on='Gene_ID', how='left', validate='one_to_one') + + + def main(): parser = argparse.ArgumentParser(description="Calculate RPKM and TPM values for gene expression data.") parser.add_argument("--output", type=str, help="Output file name", required=True) parser.add_argument("--counts-file", type=str, help="Path to the file containing read counts in TSV format", required=True) - parser.add_argument("--gene-lengths-file", type=str, help="Path to the file containing gene lengths in TSV format", required=True) + parser.add_argument("--gene-lengths-file", type=str, help="Legacy length table (optional); featureCounts Length is preferred", required=False) parser.add_argument("--gtf-file", type=str, help="Path to the GTF file", required=True) args = parser.parse_args() - # Read counts and gene lengths from input files using pandas - counts_df = pd.read_csv(args.counts_file, sep="\t", index_col=0, comment="#") - gene_lengths_df = pd.read_csv(args.gene_lengths_file, sep="\t", index_col=0, header=None) - gene_lengths_df.columns = ['counts'] - # Make sure the input DataFrames have the same index (gene IDs) - if not counts_df.index.equals(gene_lengths_df.index): - raise ValueError("Gene IDs in the input files do not match.") - - # Get the read counts from the last column - counts = counts_df.iloc[:, -1].tolist() - # Get the gene lengths as lists - gene_lengths = gene_lengths_df.iloc[:, 0].tolist() - - # Calculate RPKM and TPM values - rpkm_values = calculate_rpkm(counts, gene_lengths) - tpm_values = calculate_tpm(counts, gene_lengths) - - # Create a pandas DataFrame with the results - data = { - "Gene_ID": counts_df.index, - "Count": counts, - "RPKM": rpkm_values, - "TPM": tpm_values - } - results_df = pd.DataFrame(data) - - # Read the GTF file using pandas with specified column names - gtf_columns = ["seqname", "source", "feature", "start", "end", "score", "strand", "frame", "attributes", "gene_id"] - gtf_df = pd.read_csv(args.gtf_file, sep="\t", comment="#", header=None, names=gtf_columns) - - # Extract gene ID information from the GTF file - attributes = gtf_df["attributes"].str.split(';', expand=True) - gtf_df["gene_id"] = attributes[attributes[0].str.contains('gene_id')][0].str.extract(r'gene_id "([^"]+)"') - - # Rename the columns of the GTF DataFrame - gtf_df.rename(columns={0: "seqname", 1: "source", 2: "feature", 3: "start", 4: "end", 5: "score", - 6: "strand", 7: "frame", 8: "attributes"}, inplace=True) - - - # Merge the results table with the GTF information based on gene ID - merged_df = pd.merge(results_df, gtf_df, left_on="Gene_ID", right_on="gene_id", how="left") - merged_df.drop(columns=["gene_id"], inplace=True) - - # Clean up the Gene_ID now that all the merging is done - #merged_df['Gene_ID'] = merged_df.apply(clean_gene_id, axis=1) + merged_df = abundance_table(args.counts_file, args.gtf_file, args.gene_lengths_file) # Save the merged results to a tab-separated file using pandas merged_df.to_csv(args.output, sep="\t", index=False) diff --git a/dev/gff2gtf.py b/dev/gff2gtf.py index 316312d..55b8b34 100644 --- a/dev/gff2gtf.py +++ b/dev/gff2gtf.py @@ -5,20 +5,13 @@ def extract_gene_id(gff_row): - seqname = gff_row['seqname'] - feature = gff_row['feature'] - start = gff_row['start'] - end = gff_row['end'] - attribute_str = gff_row['attribute'] - attributes = attribute_str.split(';') - #if feature == 'CDS': - for attribute in attributes: + for attribute in str(gff_row['attribute']).split(';'): + if '=' not in attribute: + continue key, value = attribute.strip().split('=', 1) - if key == 'ID': - gene_id = "gene_id \"" + value + "\"" - #else: - # gene_id = "gene_id \"" + seqname + "\"" - return gene_id + if key == 'ID' and value: + return 'gene_id "' + value + '";' + raise ValueError(f"GFF feature is missing ID: {gff_row['seqname']}:{gff_row['start']}") def gff_to_gtf(input_file, output_file, feature_types): @@ -40,6 +33,14 @@ def gff_to_gtf(input_file, output_file, feature_types): # Extract gene_id from the attribute column and create a new column 'gene_id' gff_df['gene_id'] = gff_df.apply(extract_gene_id, axis=1) + if gff_df['gene_id'].duplicated().any(): + raise ValueError('Duplicate gene IDs in GFF; regenerate annotations with corrected RNA IDs before counting.') + for column in ('start', 'end'): + values = pd.to_numeric(gff_df[column], errors='raise') + if values.isna().any() or (values < 1).any() or (values % 1 != 0).any(): + raise ValueError(f'Invalid GFF {column} coordinates') + gff_df[column] = values.astype('int64') + # Write the GTF output gff_df.to_csv(output_file, sep='\t', index=False, columns=['seqname', 'source', 'feature', 'start', 'end', 'score', 'strand', 'frame', 'gene_id'], header=False, quoting=3) diff --git a/dev/pgdb_build_single.py b/dev/pgdb_build_single.py index ec6a1f8..a82bb48 100755 --- a/dev/pgdb_build_single.py +++ b/dev/pgdb_build_single.py @@ -12,7 +12,7 @@ --sif=FILE Path to the Ptools SIF file. --tmp_dir=DIR TMP working dir for Ptools to save intermediates. --tag=STR Tag for metagenome PGDB. - --taxprune Use taxonomic pruning when building PGDBs [True or False; default: False] + --taxprune Use taxonomic pruning when building PGDBs [enabled by default; --no_taxprune disables] """ @@ -78,6 +78,9 @@ def extract_pwy(pt_outputs): pt_id = os.path.basename(glob.glob(pt_outputs + '/*.tar.bz2')[0]).split('cyc', 1)[0] flatpath = os.path.join(pt_outputs, '1.0/data') pwy_outfile = os.path.join(pt_outputs, pt_id + '_pwy.tsv') + from metapathways.pt_exports import write_verified_empty_pathways + if write_verified_empty_pathways(flatpath, pwy_outfile): + return verfile = os.path.join(flatpath.rsplit('/', 2)[0], 'default-version') new_verfile = os.path.join(flatpath, 'version.dat') shutil.copyfile(verfile, new_verfile) @@ -298,10 +301,8 @@ def map_orfs2pwys(mp_outdir, pt_outdir): parser.add_argument("--input_dir", type=str, help="input directory.", required=True) parser.add_argument("--ptout_dir", type=str, help="output directory.", required=True) parser.add_argument("--tag", type=str, help="Tag for metagenome PGDB.", required=True) -parser.add_argument("--taxprune", action='store_true', - help="Use taxonomic pruning when building PGDBs [True or False; default: False]", - required=False - ) +from metapathways.pt_taxonomy import add_pruning_options +add_pruning_options(parser) parser.add_argument("--container", action='store_true', dest="container", default=False, required=False, help="Use when using containerized env", ) diff --git a/dev/pgdb_build_wf.py b/dev/pgdb_build_wf.py index 9b7cea9..3f70584 100755 --- a/dev/pgdb_build_wf.py +++ b/dev/pgdb_build_wf.py @@ -12,7 +12,7 @@ --sif=FILE Path to the Ptools SIF file. --tmp_dir=DIR TMP working dir for Ptools to save intermediates. --tag=STR Tag for metagenome PGDB. - --taxprune Use taxonomic pruning when building PGDBs [True or False; default: False] + --taxprune Use taxonomic pruning when building PGDBs [enabled by default; --no_taxprune disables] """ @@ -32,30 +32,39 @@ import html2text import time import traceback - - -def create_pgdb(pt_inputs, pt_outputs, tprune, tag, container): - - rename_pgdb(pt_inputs, tag) - tag_id = tag - # Create output dir if doesn't exist - Path(pt_outputs).mkdir(parents=True, exist_ok=True) - if container: - print("NOTE: Using containerized version of Pathway Tools...") - sh_tax = 'run-pathway-tools-and-copy-pgdb-taxprune.sh' - sh_notax = 'run-pathway-tools-and-copy-pgdb.sh' - else: - sh_tax = 'run-pathway-tools-and-copy-pgdb-taxprune_local.sh' - sh_notax = 'run-pathway-tools-and-copy-pgdb_local.sh' - if tprune == True: - pt_cmd = [sh_tax, pt_inputs, pt_outputs, tag_id] - elif tprune == False: - pt_cmd = [sh_notax, pt_inputs, pt_outputs, tag_id] - subprocess.run(pt_cmd) - # Uncompress PGDB to create PWYs table - pgdb_arc = glob.glob(pt_outputs + '/*.tar.bz2')[0] - tar_cmd = ['tar', '-xf', pgdb_arc, '-C', pt_outputs] - tar_out = subprocess.run(tar_cmd) +import tempfile + + +def create_pgdb(pt_inputs, pt_outputs, tprune, tag, container, image=None, sample_output=None, transport_inference=True, taxon_id=None): + from metapathways.pt_container import run_pgdb + Path(pt_outputs).mkdir(parents=True, exist_ok=True) + # Renaming must not modify the annotation products or another task's inputs. + with tempfile.TemporaryDirectory(prefix='mp-pgdb-input-') as work: + inputs = str(Path(work) / 'input') + shutil.copytree(pt_inputs, inputs) + rename_pgdb(inputs, tag) + if taxon_id is not None: + from metapathways.pt_sequences import set_organism_taxon + set_organism_taxon(inputs, taxon_id) + if image: + run_pgdb(image, inputs, pt_outputs, tag, tprune, sample_output=sample_output, transport_inference=transport_inference) + else: + if not transport_inference: + raise ValueError('--no_transport_inference requires a Pathway Tools SIF') + if sample_output is not None: + from metapathways.pt_sequences import attach_sequences + attach_sequences(inputs, sample_output) + from metapathways.pt_reactions import filter_reactions + filter_reactions(inputs, sample_output=sample_output) + shutil.copy2(Path(inputs)/'ptools-reaction-filter.json', Path(pt_outputs)/'ptools-reaction-filter.json') + suffix = '' if container else '_local' + pruning = '_taxprune' if tprune else '' + script = f'run-pathway-tools-and-copy-pgdb{pruning}{suffix}.sh' + subprocess.run([script, inputs, pt_outputs, tag], check=True) + archive = Path(pt_outputs) / f'{tag}cyc.tar.bz2' + if not archive.is_file(): + raise RuntimeError(f'Pathway Tools did not produce {archive}') + subprocess.run(['tar', '-xf', str(archive), '-C', pt_outputs], check=True) def rename_pgdb(pt_inputs, tag): @@ -74,11 +83,15 @@ def rename_pgdb(pt_inputs, tag): os.rename(o_params + '.tmp', o_params) -def extract_pwy(pt_outputs): +def extract_pwy(pt_outputs, pt_id=None): ## version.dat file is not in expected directory, create it - pt_id = os.path.basename(glob.glob(pt_outputs + '/*.tar.bz2')[0]).split('cyc', 1)[0] + if pt_id is None: + pt_id = os.path.basename(glob.glob(pt_outputs + '/*.tar.bz2')[0]).rsplit('cyc', 1)[0] flatpath = os.path.join(pt_outputs, '1.0/data') pwy_outfile = os.path.join(pt_outputs, pt_id + '_pwy.tsv') + from metapathways.pt_exports import write_verified_empty_pathways + if write_verified_empty_pathways(flatpath, pwy_outfile): + return verfile = os.path.join(flatpath.rsplit('/', 2)[0], 'default-version') new_verfile = os.path.join(flatpath, 'version.dat') shutil.copyfile(verfile, new_verfile) @@ -257,8 +270,9 @@ def get_present_rxns(pwy_frame): return pwy_rxns -def map_orfs2pwys(mp_outdir, pt_outdir): - pt_id = os.path.basename(glob.glob(pt_outdir + '/*.tar.bz2')[0]).split('cyc', 1)[0] +def map_orfs2pwys(mp_outdir, pt_outdir, pt_id=None): + if pt_id is None: + pt_id = os.path.basename(glob.glob(pt_outdir + '/*.tar.bz2')[0]).rsplit('cyc', 1)[0] orf_mapfile = glob.glob(os.path.join(mp_outdir, 'results/annotation_table/*.EC_RXN_map.tsv'))[0] pwy_outfile = os.path.join(pt_outdir, pt_id + '_pwy.tsv') pwy2orf_outfile = os.path.join(pt_outdir, pt_id + '_pwy2orf.tsv') @@ -297,13 +311,16 @@ def map_orfs2pwys(mp_outdir, pt_outdir): parser = argparse.ArgumentParser(description="Run Ptools on MP output and collect outputs.") parser.add_argument("--mp_out", type=str, help="MP3 output directory.", required=True) parser.add_argument("--tag", type=str, help="Tag for metagenome PGDB.", required=True) -parser.add_argument("--taxprune", action='store_true', - help="Use taxonomic pruning when building PGDBs [True or False; default: False]", - required=False - ) +parser.add_argument("--taxon_id", type=int) +parser.add_argument("--no_transport_inference", action="store_true") +from metapathways.pt_taxonomy import add_pruning_options +add_pruning_options(parser) parser.add_argument("--container", action='store_true', dest="container", default=False, required=False, help="Use when using containerized env", ) +parser.add_argument('--image', help='Pathway Tools Apptainer image') +parser.add_argument('--entity', help=argparse.SUPPRESS) +parser.add_argument('--compact_results', action='store_true', help=argparse.SUPPRESS) args = parser.parse_args() mp_dir = args.mp_out @@ -312,42 +329,34 @@ def map_orfs2pwys(mp_outdir, pt_outdir): container = args.container -# Build Community-level PGDB -pt_in = os.path.join(mp_dir, 'ptools') -pt_out = os.path.join(mp_dir, 'results/pgdb/community') -print("Building Community-level PGDB.") -create_pgdb(pt_in, pt_out, taxprune, tag, container) -print("Completed Community-level PGDB.") -# Parse PGDB flatfiles to create PWYs TSV table -print("Extracting Community-level PGDB.") -extract_pwy(pt_out) -print("Extracting Complete.") -# Map inferred pwys to ORFs and ECs/RXNs used -print("Mapping ORFs to Inferred Pathways.") -map_orfs2pwys(mp_dir, pt_out) -print("Mapping Complete.") - -# Build MAG-level PGDBs if they exist -ms_dir = os.path.join(mp_dir, 'magsplitter/results') -if os.path.exists(ms_dir): - mag_list = glob.glob(ms_dir + '/*') - for pt_mag in mag_list: - if "non_binned" not in pt_mag: +def run_entity(entity): + if entity == 'community': + inputs = os.path.join(mp_dir, 'ptools') + output = os.path.join(mp_dir, 'results/pgdb/community') + entity_tag = tag + else: + inputs = os.path.join(mp_dir, 'magsplitter/results', entity) + output = os.path.join(mp_dir, 'results/pgdb/MAGs', entity) + entity_tag = entity + def build(destination): + create_pgdb(inputs, destination, taxprune, entity_tag, container, args.image, mp_dir, transport_inference=not args.no_transport_inference, taxon_id=args.taxon_id if args.taxon_id is not None else 131567) + extract_pwy(destination, entity_tag) + map_orfs2pwys(mp_dir, destination, entity_tag) + if args.compact_results: + from metapathways.compact_storage import compact_pgdb + compact_pgdb(output, entity_tag, build) + else: + build(output) + + +if args.entity: + run_entity(args.entity) +else: + run_entity('community') + for mag in sorted(Path(mp_dir, 'magsplitter/results').glob('*')): + if mag.is_dir() and 'non_binned' not in mag.name: try: - mag_id = os.path.basename(pt_mag) - mag_tag = mag_id - pt_out = os.path.join(mp_dir, 'results/pgdb/MAGs/' + mag_id) - print(f"Building PGDB for {mag_id}.") - create_pgdb(pt_mag, pt_out, taxprune, mag_tag, container) - print("Completed PGDB.") - # Parse PGDB flatfiles to create PWYs TSV table - print(f"Extracting PGDB for {mag_id}.") - extract_pwy(pt_out) - print("Extracting Complete.") - # Map inferred pwys to ORFs and ECs/RXNs used - print(f"Mapping {mag_id} ORFs to Inferred Pathways.") - map_orfs2pwys(mp_dir, pt_out) - print("Mapping Complete.") + run_entity(mag.name) except Exception as e: - print(f"PGDB build for {mag_id} failed due to: {e}") - traceback.print_exc() \ No newline at end of file + print(f'PGDB build for {mag.name} failed due to: {e}') + traceback.print_exc() diff --git a/dev/ptRNAscan.py b/dev/ptRNAscan.py index 4b5659d..1d16e77 100755 --- a/dev/ptRNAscan.py +++ b/dev/ptRNAscan.py @@ -24,7 +24,8 @@ def run_tRNAscan(thread_id, args): break print(f"Thread {thread_id}: Processing {input_file}") - cmd = ["tRNAscan-SE", input_file] + # Parallelism is provided by the outer workers; avoid nested CPU pools. + cmd = ["tRNAscan-SE", "--thread", "1", input_file] for flag, file_path in output_files.items(): cmd.extend([flag, file_path]) @@ -257,6 +258,10 @@ def main(input_file, num_threads, tmp_dir, min_length, args): args = parser.parse_args() + if os.environ.get('METAPATHWAYS_COMPACT_SCRATCH'): + import tempfile + args.tmp_dir = tempfile.mkdtemp(prefix='trna-', dir=os.environ['METAPATHWAYS_COMPACT_SCRATCH']) + if not os.path.exists(args.tmp_dir): os.makedirs(args.tmp_dir) diff --git a/docker/Dockerfile.release b/docker/Dockerfile.release index da1bb68..f2fdee3 100644 --- a/docker/Dockerfile.release +++ b/docker/Dockerfile.release @@ -5,7 +5,7 @@ USER $MAMBA_USER ARG VERSION ARG REVISION LABEL org.opencontainers.image.title="MetaPathways" \ - org.opencontainers.image.description="Core metagenomic annotation pipeline" \ + org.opencontainers.image.description="Metagenomic annotation and analysis workflow" \ org.opencontainers.image.source="https://github.com/hallamlab/MetaPathways" \ org.opencontainers.image.url="https://github.com/hallamlab/MetaPathways" \ org.opencontainers.image.licenses="MIT" \ @@ -15,6 +15,6 @@ COPY --chown=$MAMBA_USER:$MAMBA_USER package/ /tmp/release-package/ RUN micromamba install --yes --name base --file /tmp/release-package/explicit.txt && \ micromamba clean --all --yes && rm -rf /tmp/release-package # Apptainer exec bypasses Docker's entrypoint; expose the environment there too. -ENV PATH="/opt/conda/bin:${PATH}" PYTHONNOUSERSITE=1 +ENV PATH="/opt/conda/bin:${PATH}" PYTHONNOUSERSITE=1 NXF_HOME=/work/.nextflow WORKDIR /work CMD ["metapathways", "--help"] diff --git a/docker/README.quay.md b/docker/README.quay.md index 287a8d1..247e30c 100644 --- a/docker/README.quay.md +++ b/docker/README.quay.md @@ -1,35 +1,5 @@ -# MetaPathways +# MetaPathways containers -MetaPathways is a pipeline for processing and annotating assembled metagenomic sequences. +Run metagenomic annotation, read abundance, genome splitting and reports with the versioned `quay.io/hallamlab/metapathways:4.0.0` image. MAGSplitter, Camelot and the three-sample test dataset are included. Licensed Pathway Tools and production references are supplied separately. -- **Source and documentation:** https://github.com/hallamlab/MetaPathways -- **Releases and Apptainer downloads:** https://github.com/hallamlab/MetaPathways/releases -- **Issues:** https://github.com/hallamlab/MetaPathways/issues -- **License:** MIT -- **Platform:** Linux x86-64 (amd64), Python 3.11 - -Release containers contain the validated MetaPathways Conda package and its pinned dependency versions. They include the core annotation pipeline; optional MAGSplitter, camelot-frs, and licensed Pathway Tools require separate installation. - -## Docker - -Use a version tag for reproducible work (replace VERSION with a published release, such as 3.5.0): - -```bash -docker pull quay.io/hallamlab/metapathways:VERSION -docker run --rm quay.io/hallamlab/metapathways:VERSION metapathways version -docker run --rm -v "$PWD:/work" -w /work --user "$(id -u):$(id -g)" quay.io/hallamlab/metapathways:VERSION metapathways run --help -``` - -Mount your input files, configuration, reference databases, and output directory when running a pipeline. The `latest` tag tracks stable releases; release candidates do not update it. - -## Apptainer / Singularity - -Download the `.sif` file and container checksums from the corresponding GitHub release, or convert the Docker image: - -```bash -apptainer pull metapathways.sif docker://quay.io/hallamlab/metapathways:VERSION -apptainer exec metapathways.sif metapathways version -apptainer exec --bind "$PWD:/work" --pwd /work metapathways.sif metapathways run --help -``` - -The image contains software, not production reference databases. Follow the source repository's documentation to configure your data and databases. +Follow the [Docker and Apptainer guide](https://hallamlab-metapathways.readthedocs.io/en/latest/containers.html) for installation, the three-sample test, and persistent input/output storage. See the [GitHub quick start](https://github.com/hallamlab/MetaPathways#quick-start) and [report an issue](https://github.com/hallamlab/MetaPathways/issues). diff --git a/docker/conda_base.yml b/docker/conda_base.yml index e79b5ff..71f23af 100644 --- a/docker/conda_base.yml +++ b/docker/conda_base.yml @@ -4,11 +4,13 @@ channels: - bioconda dependencies: - python=3.11 - - snakemake-minimal=9.27.0 - - pulp>=2.7,<3.4 + - pip + - nextflow>=25.10,<27 + - apptainer>=1.3,<2 - setuptools>=83,<85 - urllib3>=2.8.0,<3 - curl + - wget - ncurses - pysam - cython diff --git a/docker/conda_dev.yml b/docker/conda_dev.yml index 7b56410..f10d118 100644 --- a/docker/conda_dev.yml +++ b/docker/conda_dev.yml @@ -6,8 +6,6 @@ dependencies: - boa - anaconda-client - conda-verify - - sphinx - - sphinx_rtd_theme - pip - pip: - twine diff --git a/docs/WORKFLOW_STYLE.md b/docs/WORKFLOW_STYLE.md new file mode 100644 index 0000000..c34eb77 --- /dev/null +++ b/docs/WORKFLOW_STYLE.md @@ -0,0 +1,87 @@ +--- +orphan: true +--- + +# Shared workflow figure style: MP nodal v1 + +The MetaPathways publication workflow is the visual reference for workflow +figures across MetaPathways, ASPIRE, SCARAB, BASINS, DECOI and related repositories. +This is the user's standing preference. Content updates must preserve this +visual grammar; a different layout requires an explicit user request. + +## Layout and node grammar + +- White canvas; a restrained gray title / execution header. +- Inputs and the node legend sit immediately below the header. +- Numbered square module nodes form a connected vertical spine on the left. + Module labels sit to its left. A number groups related operations; it does + not assert that independent analyses execute serially. +- Each module reads left to right: circular inputs, diamond compute steps, + circular data / outputs. Put short process labels above compute diamonds + and concise software / major-library names directly below each diamond. + Omit module-wide explanatory notes; keep those details in the accompanying + documentation. Preserve generous whitespace. +- Show actual forks and joins with clean orthogonal connectors. Label optional + branches explicitly. Keep cohort-specific paths scientifically accurate. +- Use thin black connectors with small arrowheads and heavier node outlines. + Avoid lines through text, overlapping labels and truncated exports. +- Do not replace the node graph with stacked paragraph cards, a text table, + a dashboard, gradients, decorative icons or a generic box flowchart. + +## Typography and palette + +| Element | Style | +|---|---| +| Labels | Times New Roman, Times, serif; black `#111111` | +| Canvas | White `#FFFFFF` | +| Header / legend | Gray `#CCCCCC`, outline `#666666` | +| Compute / intermediate data | `#F5F5F5`, outline `#666666` | +| Module square | `#F5F5F5`, outline `#111111` | +| Input circle | Blue `#DAE8FC`, outline `#6C8EBF` | +| Output circle | Green `#D5E8D4`, outline `#82B366` | +| Connectors | Black, approximately 1.5 SVG units | +| Node outline | Approximately 3 SVG units | + +At publication scale, use approximately 20–22-unit node labels and 26–29-unit +module / title text. Adjust the canvas and whitespace to fit the content; +do not shrink the entire figure to accommodate longer paragraphs. Full and +brief figures use the same grammar. Biological plot palettes are separate +study settings and are not replaced by these workflow colors. + +## Documentation progression + +1. Put the brief nodal overview on the documentation landing page and link + directly to the detailed workflow page. +2. Lead that detailed page with a separate, expanded nodal figure in the same + visual style. Provide a full-size SVG link and a PDF where exported. +3. Place the supporting architecture, dependency and data-flow diagrams below + the large figure, followed by process explanations and citations. +4. Keep a link back to the brief overview beside the detailed figure’s downloads. + +The brief and detailed figures must be distinct levels of detail. Supporting +Mermaid diagrams complement the expanded nodal figure rather than substitute +for it. Apply this progression to new documentation guides as well. + +## Sources, exports and review + +Maintain an editable SVG or a deterministic generator as the source of truth. +Update the generator when one exists, then rebuild SVGs, PDFs and documentation +previews together. Never copy a different diagram over a generated preview. +Preserve source-to-preview identity and shared preview scale where configured. + +Detailed software dependency diagrams may retain Mermaid where they already +serve as technical supplements, using the same serif typography and restrained +palette. They do not replace the primary full or brief MP nodal workflow. + +Before completing a workflow edit: + +1. Run `python3 scripts/check_workflow_style.py`. +2. Render and inspect both full and brief figures at readable size. Check the + documentation page and exported figure for clipping, overlap and spacing. +3. Build the documentation with warnings treated as errors where supported. +4. Confirm that module content and optional branches match the pipeline. + +The automated check rejects loss of the nodal grammar, serif font or palette. +It complements visual review; it cannot certify layout quality or scientific +correctness. The canonical visual reference is MetaPathways +`docs/assets/workflow-main.svg` (the appnote module-and-node layout). diff --git a/docs/analysis.md b/docs/analysis.md new file mode 100644 index 0000000..f6a651e --- /dev/null +++ b/docs/analysis.md @@ -0,0 +1,83 @@ +# Complete multi-sample analysis + +**To build PGDBs, first complete the [Pathway Tools installation guide](pathway-tools.md)** and build/register your licensed SIF with `metapathways build_pt`. Once that setup is complete, run the workflow below. If you do not want pathway inference, add `--skip_ptools`; no Pathway Tools installation is needed. + +`analysis_wf` runs annotation, optional read mapping, optional MAG splitting, community/MAG pathway inference, and a combined report. It accepts one or many metagenomes. All computational tasks share **one Nextflow DAG and one resource budget**; samples do not wait for other samples to finish. Community PGDB construction, read mapping, and MAG splitting can proceed together once their annotation inputs are ready. MAG PGDB jobs follow splitting. + +## Analysis wf input layout + +Use this exact layout for automatic discovery (directory and sample names are case-sensitive): + +```text +inputs/ + assemblies/ + SampleA.fasta + SampleB.fasta + reads/ + SampleA_R1.fastq.gz + SampleA_R2.fastq.gz + SampleB_interleaved.fastq.gz + mag_maps/ + SampleA.tsv + SampleB.tsv +``` + +Assembly suffixes are `.fa`, `.fna`, or `.fasta`, optionally `.gz`. Read suffixes are `.fq` or `.fastq`, optionally `.gz`. Read names must end with `_R1` and `_R2` for paired files, `_interleaved` for interleaved pairs, or `_single` for single-end reads. Sample IDs use letters, digits and underscores, beginning with a letter. Names `logs`, `reports`, `inputs`, `assemblies`, `reads`, and `mag_maps` are reserved. + +Every assembly needs exactly one read layout and one map by default. Each map is a **headerless, two-column TSV**: original assembly contig ID, then MAG ID. A contig may appear only once, and its ID must match the assembly FASTA header's first whitespace-delimited token. MAG IDs use letters, digits, underscores and periods, beginning with a letter. Periods become underscores in MAG output directory names; IDs that collide after that conversion are rejected. `community` and IDs containing `non_binned` are reserved. + +```bash +metapathways analysis_wf \ + -i /path/to/inputs \ + -o /path/to/analysis \ + -d /path/to/MPDB \ + --threads 8 --max_cpus 32 +``` + +The registered SIF from `metapathways build_pt` is used automatically; `--image /path/to/ptools.sif` overrides it. Container isolation allows multiple single-CPU Pathway Tools jobs. All workflow tasks inherit `--memory` (16 GB by default); `--ptools_memory` optionally overrides only PGDB jobs; memory availability can limit concurrency before CPUs do. Slurm uses the same [resource flags](resources.md) and requires all inputs, outputs, software and the SIF to be accessible on compute nodes. + +For existing flat directories, point `-i` at the assemblies directory and supply `--reads_dir /path/to/reads` and `--mag_maps_dir /path/to/maps`. If only one of those flags is supplied, the other branch defaults to `reads/` or `mag_maps/` under `-i`. Use `--no_reads` or `--no_mags` to explicitly omit those branches for every sample during discovery. Use `--skip_ptools` to omit PGDB construction while retaining annotation, read mapping, MAG splitting and reporting. The manifest below supports a different combination for each sample. + +Discovery does not recurse, merge sequencing lanes, or guess unmarked FASTQ layouts. Unexpected files/subdirectories, duplicate assembly names, orphan reads/maps, missing mates, reused input files, duplicate contig IDs and map IDs absent from their assembly stop the command **before any jobs are submitted**, with a link to this section. Hidden directory entries are ignored. Explicitly disabled branches are not scanned. Assembly headers and maps are checked fully; FASTQ files are checked for readability and nonzero size, not full sequencing integrity. + +Append `--dryrun` to validate inputs and write the combined task plan without running biological tools. This still creates output directories, logs, staged symlinks and `inputs.resolved.tsv`. Remove `--dryrun` from the same command to execute. Input paths may contain spaces because MP stages symlinks; the output and MPDB paths must currently use letters, digits, underscores, hyphens, periods and slashes for compatibility with legacy tool commands. + +## Custom analysis manifest + +Use a tab-separated file with this **exact header and column order**: + +```text +sample_id assembly read_layout reads_1 reads_2 mag_map +SampleA assemblies/assembly_a.fasta paired reads/a_1.fq.gz reads/a_2.fq.gz maps/a.tsv +SampleB assemblies/assembly_b.fasta interleaved reads/b.fq.gz maps/b.tsv +SampleC assemblies/assembly_c.fasta none "" "" "" +``` + +The example uses actual tabs, with `""` denoting empty fields in the final row. Paths are relative to the manifest's directory or absolute. `read_layout` is `paired`, `interleaved`, `single`, or `none`. Paired reads require both paths; single/interleaved reads require only `reads_1`; `none` requires both paths empty. A blank `mag_map` omits MAG splitting and MAG PGDBs for that sample. Community PGDB construction still runs unless `--skip_ptools` is supplied. Sample IDs come from the manifest, so original filenames do not need to match them. Assemblies must still have a supported FASTA suffix. + +```bash +metapathways analysis_wf \ + --manifest /path/to/samples.tsv \ + -o /path/to/analysis -d /path/to/MPDB \ + --threads 8 --max_cpus 32 +``` + +Do not combine `--manifest` with `-i`, discovery-directory flags, `--no_reads` or `--no_mags`. The single-sample `-1`, `-2`, `--interleaved`, `--samples` and `--test` options are not used by `analysis_wf`; each row declares its own reads and sample ID. Required annotation stages cannot be skipped; successful tasks are reused through receipts. + +## Outputs, restarting and exploration + +By default MP keeps all sample outputs. Add `--compact_results` to `analysis_wf` to remove large intermediates after each sample finishes while retaining rebuildable reports, final tables, logs and benchmark traces. PGDB archives are retained; builds and extraction use worker scratch, and diagnostics are bundled. See [compact results and restart behavior](benchmarking.md#compact-results-on-limited-storage). + +Each sample writes to `analysis/SAMPLE/`. The dataset root contains `inputs.resolved.tsv` with absolute original paths, combined `reports/`, and `logs/analysis_wf/RUN/` with the plan, tool logs, resource trace and task statuses. Staged symlinks and receipts stay under `.metapathways/analysis_wf/`; temporary Nextflow work/cache cleanup follows the ordinary MP resource options. + +Repeat the same command/output to resume. The first invocation requires a new output directory. A changed sample list, MAG ID set, or input path mapping requires a new output directory; this prevents silently mixing datasets. Input contents, commands and resource requests are checked through task receipts, and completed outputs without a matching receipt are not adopted by this workflow. `--force_redo` reruns all selected annotation, splitting and PGDB tasks. Do not run another MP command into the same output while an analysis is active. + +A community PGDB failure is a required-task failure. MAG PGDB failures are expected for some MAGs: they are logged as optional failures and do not stop the remaining MAGs. A MAG with no retained `0.pf` input after successful splitting is recorded as `SKIPPED`. The combined report retains per-sample/entity task statuses, including failures and skips. + +After completion, open `analysis/reports/MP_run_report.html`, or start the searchable explorer: + +```bash +metapathways report -o /path/to/analysis --serve --no-rebuild +``` + +The portal combines samples and keeps sample IDs on all related records so identically named MAGs and ORFs in different samples remain distinct. diff --git a/docs/annotation.md b/docs/annotation.md new file mode 100644 index 0000000..52a1fff --- /dev/null +++ b/docs/annotation.md @@ -0,0 +1,70 @@ +# Annotate your own data + +## One assembly + +```bash +metapathways run \ + -i sample.fasta \ + -o results \ + -d /data/MPDB +``` + +The sample output is `results/sample/`; the run-wide report is `results/reports/`. Input can be nucleotide FASTA, compressed FASTA, or amino-acid FASTA with `--input_format fasta-amino`. GFF and GenBank are output formats, not accepted primary inputs of this CLI. + +Choose references and inspect the plan: + +```bash +metapathways run -i sample.fasta -o results -d /data/MPDB \ + --annotation_dbs swissprot cazy --annotation_algorithm FAST --dryrun +``` + +A dry run checks dependencies and writes the execution plan and logs without executing annotation tasks. It can create sample directory scaffolding; it does not produce a completed biological report. + +## Read mapping + +Paired reads in separate files: + +```bash +metapathways run -i sample.fasta -o results -d /data/MPDB \ + -1 sample_R1.fastq.gz -2 sample_R2.fastq.gz +``` + +Interleaved paired reads: + +```bash +metapathways run -i sample.fasta -o results -d /data/MPDB \ + -1 sample.interleaved.fastq.gz --interleaved +``` + +Single-end reads use `-1` alone. With no reads, abundance calculation is skipped. `-1` and `-2` must identify different files; MP now rejects identical paired input paths, including aliases to the same file. The paired command passes the actual reverse file to CoverM. + +Historical output generated by the incorrect mate selection must not be treated as corrected merely because it appears in a report. Repeating the annotation command with the corrected reads can reuse completed annotation stages and recompute mapping. The portal copies existing abundance values; it does not certify their provenance or rerun mapping. + +## Interpreting feature and hit counts + +Protein search statistics distinguish alignment rows from distinct query ORFs; +multiple hits for one ORF count once in the latter. ORF identifiers are counted +in full, without assuming a particular naming format. Counts from different +reference databases can overlap. + +rRNA prediction and rRNA taxonomic annotation are separate outputs. MP uses +barrnap's GFF annotations to identify the subunit when extracted FASTA headers +contain only coordinates, and retains support for older headers containing the +subunit name. Alignment lengths include both endpoints and are independent of +alignment direction. A predicted rRNA feature need not have a qualifying +taxonomic hit. + +## Multiple assemblies + +Pass a directory of FASTA files to `-i`; each file becomes a sample with its own output directory. Use unique, simple filenames and `--samples` to select a subset. Read mapping arguments apply to one sample only: run separate commands for samples with different FASTQs. CPU/memory budgets are per invocation, so independent MP commands do not share a global resource limit. + +## Stage controls + +Stage flags accept `yes`, `skip` or `redo`. For example: + +```bash +metapathways run -i sample.fasta -o results -d /data/MPDB \ + --COMPUTE_TPM redo -1 sample_R1.fastq.gz -2 sample_R2.fastq.gz +``` + +`yes` reuses valid outputs, `redo` executes the stage again, and `skip` requires any needed downstream inputs to already exist. `--force_redo` forces all annotation stages. Changes to tracked inputs or outputs invalidate task receipts. See the [workflow guide](workflow.md) before skipping dependencies. diff --git a/docs/architecture.md b/docs/architecture.md new file mode 100644 index 0000000..4804545 --- /dev/null +++ b/docs/architecture.md @@ -0,0 +1,15 @@ +# Software architecture: local and HPC + +For the tool-by-tool diagrams and citations, see the [detailed workflow](detailed-workflow.md). + +The MP command validates inputs and creates a dependency graph. Nextflow schedules ready tasks using the selected executor; the biological tools perform the same work in either mode. The MP controller and Nextflow continue running until the workflow finishes. + +[![Architecture](assets/diagrams/architecture.svg)](assets/diagrams/architecture.svg) + +[Zoom diagram](assets/diagrams/architecture.svg) · [Mermaid source](diagrams/architecture.mmd) + +The two executor branches are alternatives selected by `--executor`; local execution is the default. Workers need access to the software environment, inputs, reference databases, image and output paths. On Slurm these must be accessible from the compute nodes, usually through shared storage. This diagram shows logical components; shared boxes do not mean jobs share a Pathway Tools instance. + +Thread-capable tools use the requested `--threads`; serial tools, including Pathway Tools, request one CPU. Nextflow can run independent tasks simultaneously. Local concurrency fits detected or specified CPU/memory budgets. Slurm requests resources per job, with `--max_tasks` and `--submit_rate` bounding submissions. See [resource settings](resources.md). + +Each containerized PGDB attempt has private Pathway Tools state, allowing separate entities to run concurrently. A community PGDB failure fails the required workflow task; MAG failures are retained as optional outcomes. Reports include task statuses and diagnostics. See [execution and restart behavior](execution.md) and [Pathway Tools](pathway-tools.md). diff --git a/docs/assets/cami.bib b/docs/assets/cami.bib new file mode 100644 index 0000000..6798a44 --- /dev/null +++ b/docs/assets/cami.bib @@ -0,0 +1,23 @@ +@article{Sczyrba2017CAMI, + author = {Sczyrba, Alexander and Hofmann, Peter and Belmann, Peter and Koslicki, David and Janssen, Stefan and Dröge, Johannes and Gregor, Ivan and Majda, Stephan and Fiedler, Jessika and Dahms, Eik and Bremges, Andreas and Fritz, Adrian and Garrido-Oter, Ruben and Jørgensen, Tue Sparholt and Shapiro, Nicole and Blood, Philip D and Gurevich, Alexey and Bai, Yang and Turaev, Dmitrij and DeMaere, Matthew Z and Chikhi, Rayan and Nagarajan, Niranjan and Quince, Christopher and Meyer, Fernando and Balvočiūtė, Monika and Hansen, Lars Hestbjerg and Sørensen, Søren J and Chia, Burton K H and Denis, Bertrand and Froula, Jeff L and Wang, Zhong and Egan, Robert and Don Kang, Dongwan and Cook, Jeffrey J and Deltel, Charles and Beckstette, Michael and Lemaitre, Claire and Peterlongo, Pierre and Rizk, Guillaume and Lavenier, Dominique and Wu, Yu-Wei and Singer, Steven W and Jain, Chirag and Strous, Marc and Klingenberg, Heiner and Meinicke, Peter and Barton, Michael D and Lingner, Thomas and Lin, Hsin-Hung and Liao, Yu-Chieh and Silva, Genivaldo Gueiros Z and Cuevas, Daniel A and Edwards, Robert A and Saha, Surya and Piro, Vitor C and Renard, Bernhard Y and Pop, Mihai and Klenk, Hans-Peter and Göker, Markus and Kyrpides, Nikos C and Woyke, Tanja and Vorholt, Julia A and Schulze-Lefert, Paul and Rubin, Edward M and Darling, Aaron E and Rattei, Thomas and McHardy, Alice C}, + title = {Critical Assessment of Metagenome Interpretation—a benchmark of metagenomics software}, + journal = {Nature Methods}, + year = {2017}, + volume = {14}, + number = {11}, + pages = {1063--1071}, + doi = {10.1038/nmeth.4458}, + url = {https://doi.org/10.1038/nmeth.4458} +} + +@article{Meyer2022CAMIII, + author = {Meyer, Fernando and Fritz, Adrian and Deng, Zhi-Luo and Koslicki, David and Lesker, Till Robin and Gurevich, Alexey and Robertson, Gary and Alser, Mohammed and Antipov, Dmitry and Beghini, Francesco and Bertrand, Denis and Brito, Jaqueline J. and Brown, C. Titus and Buchmann, Jan and Buluç, Aydin and Chen, Bo and Chikhi, Rayan and Clausen, Philip T. L. C. and Cristian, Alexandru and Dabrowski, Piotr Wojciech and Darling, Aaron E. and Egan, Rob and Eskin, Eleazar and Georganas, Evangelos and Goltsman, Eugene and Gray, Melissa A. and Hansen, Lars Hestbjerg and Hofmeyr, Steven and Huang, Pingqin and Irber, Luiz and Jia, Huijue and Jørgensen, Tue Sparholt and Kieser, Silas D. and Klemetsen, Terje and Kola, Axel and Kolmogorov, Mikhail and Korobeynikov, Anton and Kwan, Jason and LaPierre, Nathan and Lemaitre, Claire and Li, Chenhao and Limasset, Antoine and Malcher-Miranda, Fabio and Mangul, Serghei and Marcelino, Vanessa R. and Marchet, Camille and Marijon, Pierre and Meleshko, Dmitry and Mende, Daniel R. and Milanese, Alessio and Nagarajan, Niranjan and Nissen, Jakob and Nurk, Sergey and Oliker, Leonid and Paoli, Lucas and Peterlongo, Pierre and Piro, Vitor C. and Porter, Jacob S. and Rasmussen, Simon and Rees, Evan R. and Reinert, Knut and Renard, Bernhard and Robertsen, Espen Mikal and Rosen, Gail L. and Ruscheweyh, Hans-Joachim and Sarwal, Varuni and Segata, Nicola and Seiler, Enrico and Shi, Lizhen and Sun, Fengzhu and Sunagawa, Shinichi and Sørensen, Søren Johannes and Thomas, Ashleigh and Tong, Chengxuan and Trajkovski, Mirko and Tremblay, Julien and Uritskiy, Gherman and Vicedomini, Riccardo and Wang, Zhengyang and Wang, Ziye and Wang, Zhong and Warren, Andrew and Willassen, Nils Peder and Yelick, Katherine and You, Ronghui and Zeller, Georg and Zhao, Zhengqiao and Zhu, Shanfeng and Zhu, Jie and Garrido-Oter, Ruben and Gastmeier, Petra and Hacquard, Stephane and Häußler, Susanne and Khaledi, Ariane and Maechler, Friederike and Mesny, Fantin and Radutoiu, Simona and Schulze-Lefert, Paul and Smit, Nathiana and Strowig, Till and Bremges, Andreas and Sczyrba, Alexander and McHardy, Alice Carolyn}, + title = {Critical Assessment of Metagenome Interpretation: the second round of challenges}, + journal = {Nature Methods}, + year = {2022}, + volume = {19}, + number = {4}, + pages = {429--440}, + doi = {10.1038/s41592-022-01431-4}, + url = {https://doi.org/10.1038/s41592-022-01431-4} +} diff --git a/docs/assets/diagrams/architecture.svg b/docs/assets/diagrams/architecture.svg new file mode 100644 index 0000000..6fc5b7d --- /dev/null +++ b/docs/assets/diagrams/architecture.svg @@ -0,0 +1 @@ +ArchitectureUser command and manifestMP controllerValidate and planNextflowDependencies and task schedulingLocal executorServer CPU and memory budgetSlurm executorSubmission and queue limitsLocal worker processesCompute-node jobsAnnotation and mapping toolsOptional Pathway ToolsPrivate SIF instancesOutput files, logs and task receiptsMP report and explorer diff --git a/docs/assets/diagrams/data-flow.svg b/docs/assets/diagrams/data-flow.svg new file mode 100644 index 0000000..7870673 --- /dev/null +++ b/docs/assets/diagrams/data-flow.svg @@ -0,0 +1 @@ +Data FlowAssembly FASTAPreprocessed contigsPredicted CDS and RNA featuresFunctional and taxonomic annotationsMPDB reference sequences and mappingsOptional readsRead mapping and feature countingContig and feature abundanceCommunity Pathway Tools inputsGenome-specific inputsOptional contig-to-genome mapSequence-backed PGDB stagingOptional licensed Pathway Tools inferencePathway, reaction and gene tablesReport tables and relational indexExplorer searches, subsets and CSV exports diff --git a/docs/assets/diagrams/detailed-workflow-1.svg b/docs/assets/diagrams/detailed-workflow-1.svg new file mode 100644 index 0000000..86c96a1 --- /dev/null +++ b/docs/assets/diagrams/detailed-workflow-1.svg @@ -0,0 +1 @@ +Detailed Workflow 1optional matching referencepreparationSelected public referencesProteins, SILVA, taxonomy and EC databuild_db through NextflowDownload and prepare reference filesfastdb or makeblastdbProtein indexes and BLAST nucleotide indexesMP reference preparationIdentifiers, taxonomy and EC mappingsMPDBSearchable references and supporting tablesUser-supplied licensed installerbuild_pt through NextflowApptainer build and official SRI patchesValidate Pathway Tools startupand BLAST database creation/searchRegistered private SIFPathway Tools and bundled MetaCycMP MetaCyc preparationprotseq.fsa and companion flat filesFAST or BLAST protein indexEC, reaction and pathway mappings diff --git a/docs/assets/diagrams/detailed-workflow-2.svg b/docs/assets/diagrams/detailed-workflow-2.svg new file mode 100644 index 0000000..f3011df --- /dev/null +++ b/docs/assets/diagrams/detailed-workflow-2.svg @@ -0,0 +1 @@ +Detailed Workflow 2reference inputcontig sequencesreference inputcontig sequenceslookup dataAssembly FASTAOriginal contig identifiersPREPROCESS_INPUT - MPSequence filtering and identifier mapORF_PREDICTIONpProdigal wrapping ProdigalORF_TO_AMINO and FILTER_AMINOS - MPCDS coordinates, nucleotide and protein sequencesFUNC_SEARCH - one task per selected databaseFAST fastal or BLAST+ blastpMPDB protein indexesCOMPUTE_REFSCORES - MPSequence-based reference scores for hit filteringPARSE_FUNC_SEARCH - MPPer-database thresholds and parsed hitsSCAN_rRNA - barrnaprRNA coordinates and sequencesSCAN_rRNA - BLAST+ blastn and MPSearch selected rRNA references and summarizeMPDB rRNA indexesFor example SILVA SSU and LSUSCAN_tRNA - ptRNAscan wrapping tRNAscan-SEtRNA coordinates and identitiesANNOTATE_ORFS - MPCombine protein hits and RNA featurespybedtools / BEDTools overlap processingCREATE_ANNOT_REPORTS - MPFunctional assignments and per-database taxonomy / LCAReference taxon identifiersand NCBI taxonomy treeGENBANK_FILE - MPAnnotated sequence exportPATHOLOGIC_INPUT - MPPF features, feature coordinates, ORF map and EC / reaction mapContinue to abundance and optional PGDBs diff --git a/docs/assets/diagrams/detailed-workflow-3.svg b/docs/assets/diagrams/detailed-workflow-3.svg new file mode 100644 index 0000000..d185ec7 --- /dev/null +++ b/docs/assets/diagrams/detailed-workflow-3.svg @@ -0,0 +1 @@ +Detailed Workflow 3contig statisticsreport sourcesCompleted PATHOLOGIC_INPUTCommunity features and feature-to-contig relationshipsCOMPUTE_TPM - CoverMReads plus sample contigsMapping and contig coverageSAMtoolsName-sort the exact CoverM BAMfeatureCountsBAM plus MP-generated GTFCounts for annotated featuresAbundance outputsContig coverage and feature countsMP abund_calc.py normalizationCommunity PGDB staging - MPSample sequences, feature coordinatesand translation tablesMAGSplitterUse contig-to-genome mapPartition existing featuresPer-genome PGDB staging - MPCorresponding contig sequencesand feature coordinatesPathway Tools / PathoLogicLicensed SIF with MetaCycOne private instance per entityBuild and save PGDBPathway Tools export then Camelot / MPRead flat files and extract pathway-to-gene associationsPGDB archivePathway and pathway-to-ORF tablesSample terminal tasks finishOptional compact-results cleanupMP report controllerImport tables into SQLite with scoped identifiersAnnotation, taxonomy and membership tablesSample statistics and execution recordsMP_run_report.htmlRun details, outcomes and resource summariesEDA_portal.htmlSamples to features, annotations and pathwaysSearch, subset and export CSV diff --git a/docs/assets/diagrams/overview.svg b/docs/assets/diagrams/overview.svg new file mode 100644 index 0000000..e55be82 --- /dev/null +++ b/docs/assets/diagrams/overview.svg @@ -0,0 +1 @@ +OverviewAssembliesFunctional and taxonomic annotationReadsRead abundanceCommunity and genome pathwaysGenome assignmentsLicensed Pathway ToolsReports and explorerFiltered tables and CSV exports diff --git a/docs/assets/diagrams/readme.svg b/docs/assets/diagrams/readme.svg new file mode 100644 index 0000000..acf921d --- /dev/null +++ b/docs/assets/diagrams/readme.svg @@ -0,0 +1 @@ +ReadmeAssembliesFunctional and taxonomic annotationReadsRead abundanceCommunity and genome pathwaysGenome assignmentsLicensed Pathway ToolsReports and explorerFiltered tables and CSV exports diff --git a/docs/assets/diagrams/results-schema.svg b/docs/assets/diagrams/results-schema.svg new file mode 100644 index 0000000..0a7a580 --- /dev/null +++ b/docs/assets/diagrams/results-schema.svg @@ -0,0 +1 @@ +Results Schemacontainscontainshasdescribescontainsassignsbelongssuppliessuppliesinferssupportsparticipatesgroupssamplescontigsorfsannotationsannotation_termsentitiescontig_magsentity_orfspathwayspathway_orfsorf_groups diff --git a/docs/assets/diagrams/workflow-detailed.svg b/docs/assets/diagrams/workflow-detailed.svg new file mode 100644 index 0000000..726322d --- /dev/null +++ b/docs/assets/diagrams/workflow-detailed.svg @@ -0,0 +1,463 @@ +Workflow Detailed +MetaPathways complete software workflow +Numbered conceptual modules with inputs, computational steps, data products and optional branches. See the workflow guide for source-code references and execution order. + + + + +MetaPathways + Nextflow • Detailed nucleotide-assembly workflow +Numbered modules group operations • Dependencies govern execution • Optional branches use additional inputs + +Inputs +Assemblies, sample configuration and MPDB reference indexes +Optional: reads, contig-to-genome map and licensed Pathway Tools SIF + +Module + +Compute + +Data + +Input + +Output + +Public references + +1 + + +Selected public +reference databases + +Download and normalize +reference records +MP + +Build protein / rRNA +search indexes +FAST / BLAST+ + +Map identifiers, +taxonomy and ECs +MP + +MPDB indexes +and lookup tables + + + + + +Licensed image +(optional) + +2 + + +Licensed Pathway +Tools installer + +Build private +SIF image +Apptainer + +Validate startup +and sequence search +Pathway Tools + +Prepare matching +MetaCyc references +MP / Pathway Tools + +Registered SIF +and MetaCyc indexes + + + + + +Plan and schedule + +3 + + +Sample manifest, +configuration and MPDB + +Validate inputs; +resolve sample IDs +MP + +Schedule dependency +graph and resources +Nextflow / Slurm + +Check receipts; +reuse or run tasks +MP + +Per-sample tasks +and execution records + + + + + +Sequence preparation + +4 + + +Assembly +FASTA sequences + +Validate format; +normalize sequences +MP + +Filter contigs +by configured criteria +MP + +Map identifiers; +calculate sequence stats +MP + +QCed contigs +and identifier map + + + + + +Coding features + +5 + + +QCed nucleotide +contigs + +Predict coding +regions and coordinates +pProdigal + +Derive CDS and +amino-acid sequences +MP + +Filter predicted +protein sequences +MP + +CDS coordinates +and filtered proteins + + + + + +Functional search + +6 + + +Filtered proteins +and MPDB indexes + +Search selected +protein references +FAST / BLAST+ + +Compute sequence +reference scores +MP + +Apply thresholds; +parse hits per database +MP + +Database-specific +functional evidence + + + + + +RNA features + +7 + + +QCed contigs and +rRNA reference indexes + +Predict rRNA +coordinates +barrnap + +Search rRNA +reference sequences +BLAST+ + +Predict tRNA +coordinates and types +ptRNAscan-SE + +rRNA / tRNA features +and reference hits + + + + + +Feature integration + +8 + + +Protein evidence +and RNA features + +Combine functional +and RNA annotations +MP + +Resolve overlapping +sequence features +pybedtools / BEDTools + +Integrated features +and annotated GFF + + + + +Annotation tables + +9 + + +Integrated features +and reference taxonomy + +Map functional +identifiers and names +MP + +Assign per-database +hit taxonomy / LCA +MP + +Write annotation +and reference tables +MP / pandas + +Functional / taxonomic +annotation tables + + + + + +Annotated exports + +10 + + +Contigs, features +and annotation tables + +Export annotated +sequence records +MP + +Prepare PF features +and coordinate maps +MP + +Build EC / reaction +and ORF mappings +MP + +GenBank and +Pathway Tools inputs + + + + + +Read abundance +(optional) + +11 + + +Reads, contigs +and annotated features + +Map reads; +measure contig coverage +CoverM + +Name-sort the +exact mapping BAM +SAMtools + +Count features; +normalize abundances +featureCounts / MP + +Contig coverage, counts +and RPKM / TPM + + + + + +Community / MAG inputs +(optional) + +12 + + + +Community features +and contig sequences + +Stage sequence-backed +community inputs +MP + +Community +entity input files + + + + + +Existing annotations +and contig-to-MAG map + +Partition existing +features by genome +MAGSplitter + +Stage genome sequences +and coordinate maps +MP + + + + + +Prepared community +and genome entities + +Build PGDBs (optional) + +13 + + +Prepared entities +and licensed SIF + +Start private +instance per entity +Pathway Tools + +Infer metabolic +pathways +PathoLogic + +Save / export +pathway databases +Pathway Tools + +Community / genome +PGDB archives + + + + + +Pathway extraction +(optional) + +14 + + +Exported PGDB +flat files + +Extract pathways +and gene associations +Camelot / MP + +Preserve feature IDs; +write linked tables +MP + +Pathway and +pathway-to-ORF tables + + + + +Results integration + +15 + + +Available annotation, +abundance and +pathway tables + +Join scoped sample, +feature and genome IDs +MP / SQLite + +Record task outcomes +and resource summaries +MP / Nextflow + +Build run report +and results index +MP + +Run report and +SQLite results index + + + + + +Explore and export + +16 + + +Results index +and linked tables + +Browse samples, features +and pathways +MP + +Search, filter +and select records +MP / JavaScript + +Export selected +result tables +MP / SQLite + +EDA portal +and selected CSVs + + + + + + diff --git a/docs/assets/diagrams/workflow.svg b/docs/assets/diagrams/workflow.svg new file mode 100644 index 0000000..3a1ff55 --- /dev/null +++ b/docs/assets/diagrams/workflow.svg @@ -0,0 +1,353 @@ +Workflow + MetaPathways: annotation, abundance, pathway inference and exploration + Six conceptual modules extend the appnote workflow: preprocessing; sequence feature prediction; functional and taxonomic annotation; optional community and genome pathway inference; optional read abundance; and integrated reports and explorer. Nextflow schedules local or Slurm tasks. Assemblies and references are required; reads, genome maps and licensed pathway inference are optional. Numbered modules are not an execution schedule. + + + + + + + + + + MetaPathways + Nextflow • Local server or Slurm cluster + + + Input validation, dependencies, concurrent samples / tasks, resource requests, checkpoints and logs + + + + Inputs + + + Required: assembled metagenomes and reference databases + + + Optional: reads and contig-to-genome membership maps + + + For PGDBs: your licensed Pathway Tools installation / SIF + + + + Module + + + Compute + + + Data + + + Input + + + Output + + + + + + + + Preprocessing + + + + 1 + + + + + Assembly + + + + + Format + validation + + + + + Contig + filtering + + + + + Identifier + mapping + + + + + Sequence + statistics + + + + + QC'ed contigs + + + Sequence feature + prediction + + + + 2 + + + + + QC'ed contigs + + + + + CDS / ORFs + + + pProdigal + + + + + rRNAs + + + barrnap + + + + + tRNAs + + + ptRNAscan-SE + + + + + Features and + coordinates + + + Functional & + taxonomic + annotation + + + + 3 + + + + + Predicted features + and references + + + + + Similarity + search + + + FAST / BLAST + + + + + Score + filtering + + + + + Annotation + mapping + + + + + Taxonomy + and LCA + + MP + + + + Annotation tables, + GFF / GenBank + + + Community & + genome pathway + inference + (optional) + + + 4 + + + + + Annotations + and sequences + + + + Optional + genome map + + + Community + + + + MAGSplitter + + + Genomes + + + + Pathway + inference + + + PathoLogic + + + + + Export + PGDBs + + + + + Collect pathway– + gene links + + + Camelot / MP + + + + + PGDB archives, + pathway tables + + + Read abundance + (optional) + + + 5 + + + + + Reads, contigs + and feature GFF + + + + + Read + mapping + + + CoverM + + + + + BAM + sorting + + + SAMtools + + + + + Feature + counting + + + featureCounts + + + + + Abundance + normalization + + + MP + + + + + Contig coverage, + feature counts, + RPKM / TPM + + + Reports & + explorer + + + + 6 + + + + + Available tables + and execution logs + + + + + Import and + link results + + + MP / SQLite + + + + + Sample-first + exploration + + MP / JavaScript + + + + Search, subset + and export + + MP / JavaScript + + + + Run report and + selected CSVs + + + + + + + + + + + + MPMPMPMPMPMPPathway Tools diff --git a/docs/assets/docs.css b/docs/assets/docs.css new file mode 100644 index 0000000..1c860d7 --- /dev/null +++ b/docs/assets/docs.css @@ -0,0 +1,11 @@ +.wy-nav-content { max-width: 1150px; } +.mermaid { overflow-x: auto; margin: 1.5rem 0; } +.mermaid svg { max-width: 100%; height: auto; } +.rst-content table.docutils td { vertical-align: top; } + +/* Shared 2240-unit canvases preserve scale between workflow previews. */ +.rst-content img[src*="/diagrams/"] { display: block; width: 100%; max-width: 100%; height: auto; margin: 1.5rem auto; } + +/* Keep the primary workflow readable with modest responsive side margins. */ +.wy-nav-content:has(.mp-primary-workflow) { max-width: none; padding-inline: clamp(20px, 3vw, 56px); } +.mp-primary-workflow img { display: block; width: 100%; max-width: 100%; height: auto; margin: 1.5rem auto; } diff --git a/docs/assets/workflow-appnote-original.svg b/docs/assets/workflow-appnote-original.svg new file mode 100644 index 0000000..d185336 --- /dev/null +++ b/docs/assets/workflow-appnote-original.svg @@ -0,0 +1 @@ +MetaPathwaysPreprocessing234Sequenc FeaturePredictionFunctional &TaxonomicAnnotationCommunity &Population-levelPathway Inference1QC'ed ContigsDBsASMFormatValidationQC'ed ContigsQC'ed ContigsORFsrRNAsMapped FeaturesContig FilteringID mappingSeq Stats(pProdigal)(BARRNAP)tRNAs(ptRNAscan-SE)SimilaritySearchScore Filtering(FAST/BLAST)AnnotationMappingCDS/RNAReportsReports,AnnotatedGFF/GBKFunctionalAnnotationsMAGsPopulation-levelAnnotationsPathwayInference(MAGsplitter)ExportPGDBsCollectPathwaysCommnity- &Population-levelPathway Reports(PathoLogic)InputsAssembled Metagenome (ASM)Sequence Reads (FQ)Reference Databases (DBs)Optional:Metagenome-assembled Genomes (MAGs)DataProductComputeStepInputOutputAnalysisModule \ No newline at end of file diff --git a/docs/assets/workflow-detailed.pdf b/docs/assets/workflow-detailed.pdf new file mode 100644 index 0000000..374e2ec Binary files /dev/null and b/docs/assets/workflow-detailed.pdf differ diff --git a/docs/assets/workflow-detailed.svg b/docs/assets/workflow-detailed.svg new file mode 100644 index 0000000..f47caf5 --- /dev/null +++ b/docs/assets/workflow-detailed.svg @@ -0,0 +1,463 @@ + +MetaPathways complete software workflow +Numbered conceptual modules with inputs, computational steps, data products and optional branches. See the workflow guide for source-code references and execution order. + + + + +MetaPathways + Nextflow • Detailed nucleotide-assembly workflow +Numbered modules group operations • Dependencies govern execution • Optional branches use additional inputs + +Inputs +Assemblies, sample configuration and MPDB reference indexes +Optional: reads, contig-to-genome map and licensed Pathway Tools SIF + +Module + +Compute + +Data + +Input + +Output + +Public references + +1 + + +Selected public +reference databases + +Download and normalize +reference records +MP + +Build protein / rRNA +search indexes +FAST / BLAST+ + +Map identifiers, +taxonomy and ECs +MP + +MPDB indexes +and lookup tables + + + + + +Licensed image +(optional) + +2 + + +Licensed Pathway +Tools installer + +Build private +SIF image +Apptainer + +Validate startup +and sequence search +Pathway Tools + +Prepare matching +MetaCyc references +MP / Pathway Tools + +Registered SIF +and MetaCyc indexes + + + + + +Plan and schedule + +3 + + +Sample manifest, +configuration and MPDB + +Validate inputs; +resolve sample IDs +MP + +Schedule dependency +graph and resources +Nextflow / Slurm + +Check receipts; +reuse or run tasks +MP + +Per-sample tasks +and execution records + + + + + +Sequence preparation + +4 + + +Assembly +FASTA sequences + +Validate format; +normalize sequences +MP + +Filter contigs +by configured criteria +MP + +Map identifiers; +calculate sequence stats +MP + +QCed contigs +and identifier map + + + + + +Coding features + +5 + + +QCed nucleotide +contigs + +Predict coding +regions and coordinates +pProdigal + +Derive CDS and +amino-acid sequences +MP + +Filter predicted +protein sequences +MP + +CDS coordinates +and filtered proteins + + + + + +Functional search + +6 + + +Filtered proteins +and MPDB indexes + +Search selected +protein references +FAST / BLAST+ + +Compute sequence +reference scores +MP + +Apply thresholds; +parse hits per database +MP + +Database-specific +functional evidence + + + + + +RNA features + +7 + + +QCed contigs and +rRNA reference indexes + +Predict rRNA +coordinates +barrnap + +Search rRNA +reference sequences +BLAST+ + +Predict tRNA +coordinates and types +ptRNAscan-SE + +rRNA / tRNA features +and reference hits + + + + + +Feature integration + +8 + + +Protein evidence +and RNA features + +Combine functional +and RNA annotations +MP + +Resolve overlapping +sequence features +pybedtools / BEDTools + +Integrated features +and annotated GFF + + + + +Annotation tables + +9 + + +Integrated features +and reference taxonomy + +Map functional +identifiers and names +MP + +Assign per-database +hit taxonomy / LCA +MP + +Write annotation +and reference tables +MP / pandas + +Functional / taxonomic +annotation tables + + + + + +Annotated exports + +10 + + +Contigs, features +and annotation tables + +Export annotated +sequence records +MP + +Prepare PF features +and coordinate maps +MP + +Build EC / reaction +and ORF mappings +MP + +GenBank and +Pathway Tools inputs + + + + + +Read abundance +(optional) + +11 + + +Reads, contigs +and annotated features + +Map reads; +measure contig coverage +CoverM + +Name-sort the +exact mapping BAM +SAMtools + +Count features; +normalize abundances +featureCounts / MP + +Contig coverage, counts +and RPKM / TPM + + + + + +Community / MAG inputs +(optional) + +12 + + + +Community features +and contig sequences + +Stage sequence-backed +community inputs +MP + +Community +entity input files + + + + + +Existing annotations +and contig-to-MAG map + +Partition existing +features by genome +MAGSplitter + +Stage genome sequences +and coordinate maps +MP + + + + + +Prepared community +and genome entities + +Build PGDBs (optional) + +13 + + +Prepared entities +and licensed SIF + +Start private +instance per entity +Pathway Tools + +Infer metabolic +pathways +PathoLogic + +Save / export +pathway databases +Pathway Tools + +Community / genome +PGDB archives + + + + + +Pathway extraction +(optional) + +14 + + +Exported PGDB +flat files + +Extract pathways +and gene associations +Camelot / MP + +Preserve feature IDs; +write linked tables +MP + +Pathway and +pathway-to-ORF tables + + + + +Results integration + +15 + + +Available annotation, +abundance and +pathway tables + +Join scoped sample, +feature and genome IDs +MP / SQLite + +Record task outcomes +and resource summaries +MP / Nextflow + +Build run report +and results index +MP + +Run report and +SQLite results index + + + + + +Explore and export + +16 + + +Results index +and linked tables + +Browse samples, features +and pathways +MP + +Search, filter +and select records +MP / JavaScript + +Export selected +result tables +MP / SQLite + +EDA portal +and selected CSVs + + + + + + diff --git a/docs/assets/workflow-figure.txt b/docs/assets/workflow-figure.txt new file mode 100644 index 0000000..4d96933 --- /dev/null +++ b/docs/assets/workflow-figure.txt @@ -0,0 +1,16 @@ +MetaPathways workflow figure provenance + +Original: MP_3.5_AppNote_v1.docx, word/media/image2.svg (Figure 1). +Original SHA256: 1022b063bdcd08376b1fa0e5b556c32aaad3b2cea633317ceea6ee848d35f593 + +workflow-appnote-original.svg preserves the embedded SVG unchanged. +workflow-main.svg is an editable vector adaptation for the Nextflow workflow. +It retains the serif typography, numbered module rows, +blue inputs, green outputs, diamond compute symbols and thin black arrows. +Updates: optional reads and contig-to-genome maps; per-database taxonomy; +Nextflow/local/Slurm orchestration; read abundance; combined report and explorer. +The vertical banner and footer notes are omitted; optional modules are marked in their titles, and the symbol legend is compact. +Process labels sit above diamonds; concise tool/library names sit below every +compute diamond. Licensing requirements remain in the inputs panel and guide. +The numbered biological modules are conceptual, not the scheduler DAG. +The source DOCX has not been modified. diff --git a/docs/assets/workflow-main.svg b/docs/assets/workflow-main.svg new file mode 100644 index 0000000..f40a7e0 --- /dev/null +++ b/docs/assets/workflow-main.svg @@ -0,0 +1,353 @@ + + MetaPathways: annotation, abundance, pathway inference and exploration + Six conceptual modules extend the appnote workflow: preprocessing; sequence feature prediction; functional and taxonomic annotation; optional community and genome pathway inference; optional read abundance; and integrated reports and explorer. Nextflow schedules local or Slurm tasks. Assemblies and references are required; reads, genome maps and licensed pathway inference are optional. Numbered modules are not an execution schedule. + + + + + + + + + + MetaPathways + Nextflow • Local server or Slurm cluster + + + Input validation, dependencies, concurrent samples / tasks, resource requests, checkpoints and logs + + + + Inputs + + + Required: assembled metagenomes and reference databases + + + Optional: reads and contig-to-genome membership maps + + + For PGDBs: your licensed Pathway Tools installation / SIF + + + + Module + + + Compute + + + Data + + + Input + + + Output + + + + + + + + Preprocessing + + + + 1 + + + + + Assembly + + + + + Format + validation + + + + + Contig + filtering + + + + + Identifier + mapping + + + + + Sequence + statistics + + + + + QC'ed contigs + + + Sequence feature + prediction + + + + 2 + + + + + QC'ed contigs + + + + + CDS / ORFs + + + pProdigal + + + + + rRNAs + + + barrnap + + + + + tRNAs + + + ptRNAscan-SE + + + + + Features and + coordinates + + + Functional & + taxonomic + annotation + + + + 3 + + + + + Predicted features + and references + + + + + Similarity + search + + + FAST / BLAST + + + + + Score + filtering + + + + + Annotation + mapping + + + + + Taxonomy + and LCA + + MP + + + + Annotation tables, + GFF / GenBank + + + Community & + genome pathway + inference + (optional) + + + 4 + + + + + Annotations + and sequences + + + + Optional + genome map + + + Community + + + + MAGSplitter + + + Genomes + + + + Pathway + inference + + + PathoLogic + + + + + Export + PGDBs + + + + + Collect pathway– + gene links + + + Camelot / MP + + + + + PGDB archives, + pathway tables + + + Read abundance + (optional) + + + 5 + + + + + Reads, contigs + and feature GFF + + + + + Read + mapping + + + CoverM + + + + + BAM + sorting + + + SAMtools + + + + + Feature + counting + + + featureCounts + + + + + Abundance + normalization + + + MP + + + + + Contig coverage, + feature counts, + RPKM / TPM + + + Reports & + explorer + + + + 6 + + + + + Available tables + and execution logs + + + + + Import and + link results + + + MP / SQLite + + + + + Sample-first + exploration + + MP / JavaScript + + + + Search, subset + and export + + MP / JavaScript + + + + Run report and + selected CSVs + + + + + + + + + + + + MPMPMPMPMPMPPathway Tools \ No newline at end of file diff --git a/docs/assets/workflow-tools.bib b/docs/assets/workflow-tools.bib new file mode 100644 index 0000000..98c77ec --- /dev/null +++ b/docs/assets/workflow-tools.bib @@ -0,0 +1,392 @@ +% MetaPathways workflow references. Full author lists from publisher-deposited Crossref metadata. +% Cite only applicable components; record actual software and database versions separately. + +@article{mp_mp, + author = {McLaughlin, Ryan J. and Liu, Tony X. and Altman, Tomer and Nallan, Aditi N. and Hahn, Aria S. and Anstett, Julia and Morgan-Lang, Connor and Konwar, Kishori M. and Hallam, Steven J.}, + title = {{MetaPathways v3.5: Modularity and Scalability Improvements for Pathway Inference from Environmental Genomes}}, + year = {2024}, + journal = {bioRxiv}, + doi = {10.1101/2024.06.04.597460}, + url = {https://github.com/hallamlab/MetaPathways}, +} + +@article{mp_nextflow, + author = {Di Tommaso, Paolo and Chatzou, Maria and Floden, Evan W and Barja, Pablo Prieto and Palumbo, Emilio and Notredame, Cedric}, + title = {{Nextflow enables reproducible computational workflows}}, + year = {2017}, + journal = {Nature Biotechnology}, + volume = {35}, + number = {4}, + pages = {316--319}, + doi = {10.1038/nbt.3820}, + url = {https://www.nextflow.io/}, +} + +@article{mp_prodigal, + author = {Hyatt, Doug and Chen, Gwo-Liang and LoCascio, Philip F and Land, Miriam L and Larimer, Frank W and Hauser, Loren J}, + title = {{Prodigal: prokaryotic gene recognition and translation initiation site identification}}, + year = {2010}, + journal = {BMC Bioinformatics}, + volume = {11}, + number = {1}, + pages = {119}, + doi = {10.1186/1471-2105-11-119}, + url = {https://github.com/hyattpd/Prodigal}, +} + +@article{mp_last, + author = {Kiełbasa, Szymon M. and Wan, Raymond and Sato, Kengo and Horton, Paul and Frith, Martin C.}, + title = {{Adaptive seeds tame genomic sequence comparison}}, + year = {2011}, + journal = {Genome Research}, + volume = {21}, + number = {3}, + pages = {487--493}, + doi = {10.1101/gr.113985.110}, + url = {https://gitlab.com/mcfrith/last}, +} + +@article{mp_blast, + author = {Camacho, Christiam and Coulouris, George and Avagyan, Vahram and Ma, Ning and Papadopoulos, Jason and Bealer, Kevin and Madden, Thomas L}, + title = {{BLAST+: architecture and applications}}, + year = {2009}, + journal = {BMC Bioinformatics}, + volume = {10}, + number = {1}, + pages = {421}, + doi = {10.1186/1471-2105-10-421}, + url = {https://blast.ncbi.nlm.nih.gov/}, +} + +@article{mp_nhmmer, + author = {Wheeler, Travis J. and Eddy, Sean R.}, + title = {{nhmmer: DNA homology search with profile HMMs}}, + year = {2013}, + journal = {Bioinformatics}, + volume = {29}, + number = {19}, + pages = {2487--2489}, + doi = {10.1093/bioinformatics/btt403}, + url = {http://hmmer.org/}, +} + +@article{mp_infernal, + author = {Nawrocki, Eric P. and Eddy, Sean R.}, + title = {{Infernal 1.1: 100-fold faster RNA homology searches}}, + year = {2013}, + journal = {Bioinformatics}, + volume = {29}, + number = {22}, + pages = {2933--2935}, + doi = {10.1093/bioinformatics/btt509}, + url = {http://eddylab.org/infernal/}, +} + +@article{mp_trnascan, + author = {Chan, Patricia P and Lin, Brian Y and Mak, Allysia J and Lowe, Todd M}, + title = {{tRNAscan-SE 2.0: improved detection and functional classification of transfer RNA genes}}, + year = {2021}, + journal = {Nucleic Acids Research}, + volume = {49}, + number = {16}, + pages = {9077--9096}, + doi = {10.1093/nar/gkab688}, + url = {https://trna.ucsc.edu/tRNAscan-SE/}, +} + +@article{mp_pybedtools, + author = {Dale, Ryan K. and Pedersen, Brent S. and Quinlan, Aaron R.}, + title = {{Pybedtools: a flexible Python library for manipulating genomic datasets and annotations}}, + year = {2011}, + journal = {Bioinformatics}, + volume = {27}, + number = {24}, + pages = {3423--3424}, + doi = {10.1093/bioinformatics/btr539}, + url = {https://github.com/daler/pybedtools}, +} + +@article{mp_bedtools, + author = {Quinlan, Aaron R. and Hall, Ira M.}, + title = {{BEDTools: a flexible suite of utilities for comparing genomic features}}, + year = {2010}, + journal = {Bioinformatics}, + volume = {26}, + number = {6}, + pages = {841--842}, + doi = {10.1093/bioinformatics/btq033}, + url = {https://github.com/arq5x/bedtools2}, +} + +@article{mp_coverm, + author = {Aroney, Samuel T N and Newell, Rhys J P and Nissen, Jakob N and Camargo, Antonio Pedro and Tyson, Gene W and Woodcroft, Ben J}, + title = {{CoverM: read alignment statistics for metagenomics}}, + year = {2025}, + journal = {Bioinformatics}, + volume = {41}, + number = {4}, + pages = {btaf147}, + doi = {10.1093/bioinformatics/btaf147}, + url = {https://github.com/wwood/CoverM}, +} + +@article{mp_samtools, + author = {Danecek, Petr and Bonfield, James K and Liddle, Jennifer and Marshall, John and Ohan, Valeriu and Pollard, Martin O and Whitwham, Andrew and Keane, Thomas and McCarthy, Shane A and Davies, Robert M and Li, Heng}, + title = {{Twelve years of SAMtools and BCFtools}}, + year = {2021}, + journal = {GigaScience}, + volume = {10}, + number = {2}, + pages = {giab008}, + doi = {10.1093/gigascience/giab008}, + url = {https://www.htslib.org/}, +} + +@article{mp_featurecounts, + author = {Liao, Yang and Smyth, Gordon K. and Shi, Wei}, + title = {{featureCounts: an efficient general purpose program for assigning sequence reads to genomic features}}, + year = {2014}, + journal = {Bioinformatics}, + volume = {30}, + number = {7}, + pages = {923--930}, + doi = {10.1093/bioinformatics/btt656}, + url = {https://subread.sourceforge.net/}, +} + +@article{mp_ptools, + author = {Karp, Peter D. and Latendresse, Mario and Paley, Suzanne M. and Krummenacker, Markus and Ong, Quang D. and Billington, Richard and Kothari, Anamika and Weaver, Daniel and Lee, Thomas and Subhraveti, Pallavi and Spaulding, Aaron and Fulcher, Carol and Keseler, Ingrid M. and Caspi, Ron}, + title = {{Pathway Tools version 19.0 update: software for pathway/genome informatics and systems biology}}, + year = {2016}, + journal = {Briefings in Bioinformatics}, + volume = {17}, + number = {5}, + pages = {877--890}, + doi = {10.1093/bib/bbv079}, + url = {https://www.pathwaytools.org/}, +} + +@article{mp_minimap2, + author = {Li, Heng}, + title = {{Minimap2: pairwise alignment for nucleotide sequences}}, + year = {2018}, + journal = {Bioinformatics}, + volume = {34}, + number = {18}, + pages = {3094--3100}, + doi = {10.1093/bioinformatics/bty191}, + url = {https://github.com/lh3/minimap2}, +} + +@article{mp_bwa, + author = {Li, Heng and Durbin, Richard}, + title = {{Fast and accurate short read alignment with Burrows–Wheeler transform}}, + year = {2009}, + journal = {Bioinformatics}, + volume = {25}, + number = {14}, + pages = {1754--1760}, + doi = {10.1093/bioinformatics/btp324}, + url = {https://github.com/lh3/bwa}, +} + +@article{mp_strobealign, + author = {Sahlin, Kristoffer}, + title = {{Strobealign: flexible seed size enables ultra-fast and accurate read alignment}}, + year = {2022}, + journal = {Genome Biology}, + volume = {23}, + number = {1}, + pages = {260}, + doi = {10.1186/s13059-022-02831-7}, + url = {https://github.com/ksahlin/strobealign}, +} + +@article{mp_apptainer, + author = {Kurtzer, Gregory M. and Sochat, Vanessa and Bauer, Michael W.}, + title = {{Singularity: Scientific containers for mobility of compute}}, + year = {2017}, + journal = {PLOS ONE}, + volume = {12}, + number = {5}, + pages = {e0177459}, + doi = {10.1371/journal.pone.0177459}, + url = {https://github.com/apptainer/apptainer#citing-apptainer}, +} + +@inproceedings{mp_slurm, + author = {Yoo, Andy B. and Jette, Morris A. and Grondona, Mark}, + title = {{SLURM: Simple Linux Utility for Resource Management}}, + year = {2003}, + booktitle = {Job Scheduling Strategies for Parallel Processing}, + series = {Lecture Notes in Computer Science}, + volume = {2862}, + pages = {44--60}, + doi = {10.1007/10968987_3}, + url = {https://slurm.schedmd.com/}, +} + +@article{mp_bioconda, + author = {{The Bioconda Team} and Grüning, Björn and Dale, Ryan and Sjödin, Andreas and Chapman, Brad A. and Rowe, Jillian and Tomkins-Tinch, Christopher H. and Valieris, Renan and Köster, Johannes}, + title = {{Bioconda: sustainable and comprehensive software distribution for the life sciences}}, + year = {2018}, + journal = {Nature Methods}, + volume = {15}, + number = {7}, + pages = {475--476}, + doi = {10.1038/s41592-018-0046-7}, + url = {https://bioconda.github.io/}, +} + +@inproceedings{mp_pandas, + author = {McKinney, Wes}, + title = {{Data Structures for Statistical Computing in Python}}, + year = {2010}, + booktitle = {Proceedings of the Python in Science Conference}, + pages = {56--61}, + doi = {10.25080/majora-92bf1922-00a}, + url = {https://pandas.pydata.org/}, +} + +@article{mp_numpy, + author = {Harris, Charles R. and Millman, K. Jarrod and van der Walt, Stéfan J. and Gommers, Ralf and Virtanen, Pauli and Cournapeau, David and Wieser, Eric and Taylor, Julian and Berg, Sebastian and Smith, Nathaniel J. and Kern, Robert and Picus, Matti and Hoyer, Stephan and van Kerkwijk, Marten H. and Brett, Matthew and Haldane, Allan and del Río, Jaime Fernández and Wiebe, Mark and Peterson, Pearu and Gérard-Marchant, Pierre and Sheppard, Kevin and Reddy, Tyler and Weckesser, Warren and Abbasi, Hameer and Gohlke, Christoph and Oliphant, Travis E.}, + title = {{Array programming with NumPy}}, + year = {2020}, + journal = {Nature}, + volume = {585}, + number = {7825}, + pages = {357--362}, + doi = {10.1038/s41586-020-2649-2}, + url = {https://numpy.org/}, +} + +@article{mp_scipy, + author = {Virtanen, Pauli and Gommers, Ralf and Oliphant, Travis E. and Haberland, Matt and Reddy, Tyler and Cournapeau, David and Burovski, Evgeni and Peterson, Pearu and Weckesser, Warren and Bright, Jonathan and van der Walt, Stéfan J. and Brett, Matthew and Wilson, Joshua and Millman, K. Jarrod and Mayorov, Nikolay and Nelson, Andrew R. J. and Jones, Eric and Kern, Robert and Larson, Eric and Carey, C J and Polat, İlhan and Feng, Yu and Moore, Eric W. and VanderPlas, Jake and Laxalde, Denis and Perktold, Josef and Cimrman, Robert and Henriksen, Ian and Quintero, E. A. and Harris, Charles R. and Archibald, Anne M. and Ribeiro, Antônio H. and Pedregosa, Fabian and van Mulbregt, Paul and {SciPy 1.0 Contributors} and Vijaykumar, Aditya and Bardelli, Alessandro Pietro and Rothberg, Alex and Hilboll, Andreas and Kloeckner, Andreas and Scopatz, Anthony and Lee, Antony and Rokem, Ariel and Woods, C. Nathan and Fulton, Chad and Masson, Charles and Häggström, Christian and Fitzgerald, Clark and Nicholson, David A. and Hagen, David R. and Pasechnik, Dmitrii V. and Olivetti, Emanuele and Martin, Eric and Wieser, Eric and Silva, Fabrice and Lenders, Felix and Wilhelm, Florian and Young, G. and Price, Gavin A. and Ingold, Gert-Ludwig and Allen, Gregory E. and Lee, Gregory R. and Audren, Hervé and Probst, Irvin and Dietrich, Jörg P. and Silterra, Jacob and Webber, James T and Slavič, Janko and Nothman, Joel and Buchner, Johannes and Kulick, Johannes and Schönberger, Johannes L. and de Miranda Cardoso, José Vinícius and Reimer, Joscha and Harrington, Joseph and Rodríguez, Juan Luis Cano and Nunez-Iglesias, Juan and Kuczynski, Justin and Tritz, Kevin and Thoma, Martin and Newville, Matthew and Kümmerer, Matthias and Bolingbroke, Maximilian and Tartre, Michael and Pak, Mikhail and Smith, Nathaniel J. and Nowaczyk, Nikolai and Shebanov, Nikolay and Pavlyk, Oleksandr and Brodtkorb, Per A. and Lee, Perry and McGibbon, Robert T. and Feldbauer, Roman and Lewis, Sam and Tygier, Sam and Sievert, Scott and Vigna, Sebastiano and Peterson, Stefan and More, Surhud and Pudlik, Tadeusz and Oshima, Takuya and Pingel, Thomas J. and Robitaille, Thomas P. and Spura, Thomas and Jones, Thouis R. and Cera, Tim and Leslie, Tim and Zito, Tiziano and Krauss, Tom and Upadhyay, Utkarsh and Halchenko, Yaroslav O. and Vázquez-Baeza, Yoshiki}, + title = {{SciPy 1.0: fundamental algorithms for scientific computing in Python}}, + year = {2020}, + journal = {Nature Methods}, + volume = {17}, + number = {3}, + pages = {261--272}, + doi = {10.1038/s41592-019-0686-2}, + url = {https://scipy.org/}, +} + +@article{mp_pyfastx, + author = {Du, Lianming and Liu, Qin and Fan, Zhenxin and Tang, Jie and Zhang, Xiuyue and Price, Megan and Yue, Bisong and Zhao, Kelei}, + title = {{Pyfastx: a robust Python package for fast random access to sequences from plain and gzipped FASTA/Q files}}, + year = {2021}, + journal = {Briefings in Bioinformatics}, + volume = {22}, + number = {4}, + pages = {bbaa368}, + doi = {10.1093/bib/bbaa368}, + url = {https://github.com/lmdu/pyfastx}, +} + +@article{mp_cython, + author = {Behnel, Stefan and Bradshaw, Robert and Citro, Craig and Dalcin, Lisandro and Seljebotn, Dag Sverre and Smith, Kurt}, + title = {{Cython: The Best of Both Worlds}}, + year = {2011}, + journal = {Computing in Science \& Engineering}, + volume = {13}, + number = {2}, + pages = {31--39}, + doi = {10.1109/mcse.2010.118}, + url = {https://cython.org/}, +} + +@article{mp_uniprot, + author = {{The UniProt Consortium} and Bateman, Alex and Martin, Maria-Jesus and Orchard, Sandra and Magrane, Michele and Ahmad, Shadab and Alpi, Emanuele and Bowler-Barnett, Emily H and Britto, Ramona and Bye-A-Jee, Hema and Cukura, Austra and Denny, Paul and Dogan, Tunca and Ebenezer, ThankGod and Fan, Jun and Garmiri, Penelope and da Costa Gonzales, Leonardo Jose and Hatton-Ellis, Emma and Hussein, Abdulrahman and Ignatchenko, Alexandr and Insana, Giuseppe and Ishtiaq, Rizwan and Joshi, Vishal and Jyothi, Dushyanth and Kandasaamy, Swaathi and Lock, Antonia and Luciani, Aurelien and Lugaric, Marija and Luo, Jie and Lussi, Yvonne and MacDougall, Alistair and Madeira, Fabio and Mahmoudy, Mahdi and Mishra, Alok and Moulang, Katie and Nightingale, Andrew and Pundir, Sangya and Qi, Guoying and Raj, Shriya and Raposo, Pedro and Rice, Daniel L and Saidi, Rabie and Santos, Rafael and Speretta, Elena and Stephenson, James and Totoo, Prabhat and Turner, Edward and Tyagi, Nidhi and Vasudev, Preethi and Warner, Kate and Watkins, Xavier and Zaru, Rossana and Zellner, Hermann and Bridge, Alan J and Aimo, Lucila and Argoud-Puy, Ghislaine and Auchincloss, Andrea H and Axelsen, Kristian B and Bansal, Parit and Baratin, Delphine and Batista Neto, Teresa M and Blatter, Marie-Claude and Bolleman, Jerven T and Boutet, Emmanuel and Breuza, Lionel and Gil, Blanca Cabrera and Casals-Casas, Cristina and Echioukh, Kamal Chikh and Coudert, Elisabeth and Cuche, Beatrice and de Castro, Edouard and Estreicher, Anne and Famiglietti, Maria L and Feuermann, Marc and Gasteiger, Elisabeth and Gaudet, Pascale and Gehant, Sebastien and Gerritsen, Vivienne and Gos, Arnaud and Gruaz, Nadine and Hulo, Chantal and Hyka-Nouspikel, Nevila and Jungo, Florence and Kerhornou, Arnaud and Le Mercier, Philippe and Lieberherr, Damien and Masson, Patrick and Morgat, Anne and Muthukrishnan, Venkatesh and Paesano, Salvo and Pedruzzi, Ivo and Pilbout, Sandrine and Pourcel, Lucille and Poux, Sylvain and Pozzato, Monica and Pruess, Manuela and Redaschi, Nicole and Rivoire, Catherine and Sigrist, Christian J A and Sonesson, Karin and Sundaram, Shyamala and Wu, Cathy H and Arighi, Cecilia N and Arminski, Leslie and Chen, Chuming and Chen, Yongxing and Huang, Hongzhan and Laiho, Kati and McGarvey, Peter and Natale, Darren A and Ross, Karen and Vinayaka, C R and Wang, Qinghua and Wang, Yuqi and Zhang, Jian}, + title = {{UniProt: the Universal Protein Knowledgebase in 2023}}, + year = {2023}, + journal = {Nucleic Acids Research}, + volume = {51}, + number = {D1}, + pages = {D523--D531}, + doi = {10.1093/nar/gkac1052}, + url = {https://www.uniprot.org/}, +} + +@article{mp_uniref, + author = {Suzek, Baris E. and Huang, Hongzhan and McGarvey, Peter and Mazumder, Raja and Wu, Cathy H.}, + title = {{UniRef: comprehensive and non-redundant UniProt reference clusters}}, + year = {2007}, + journal = {Bioinformatics}, + volume = {23}, + number = {10}, + pages = {1282--1288}, + doi = {10.1093/bioinformatics/btm098}, + url = {https://www.uniprot.org/uniref}, +} + +@article{mp_cazy, + author = {Drula, Elodie and Garron, Marie-Line and Dogan, Suzan and Lombard, Vincent and Henrissat, Bernard and Terrapon, Nicolas}, + title = {{The carbohydrate-active enzyme database: functions and literature}}, + year = {2022}, + journal = {Nucleic Acids Research}, + volume = {50}, + number = {D1}, + pages = {D571--D577}, + doi = {10.1093/nar/gkab1045}, + url = {https://www.cazy.org/}, +} + +@article{mp_eggnog, + author = {Huerta-Cepas, Jaime and Szklarczyk, Damian and Heller, Davide and Hernández-Plaza, Ana and Forslund, Sofia K and Cook, Helen and Mende, Daniel R and Letunic, Ivica and Rattei, Thomas and Jensen, Lars J and von Mering, Christian and Bork, Peer}, + title = {{eggNOG 5.0: a hierarchical, functionally and phylogenetically annotated orthology resource based on 5090 organisms and 2502 viruses}}, + year = {2019}, + journal = {Nucleic Acids Research}, + volume = {47}, + number = {D1}, + pages = {D309--D314}, + doi = {10.1093/nar/gky1085}, + url = {http://eggnog.embl.de/}, +} + +@article{mp_silva, + author = {Quast, Christian and Pruesse, Elmar and Yilmaz, Pelin and Gerken, Jan and Schweer, Timmy and Yarza, Pablo and Peplies, Jörg and Glöckner, Frank Oliver}, + title = {{The SILVA ribosomal RNA gene database project: improved data processing and web-based tools}}, + year = {2013}, + journal = {Nucleic Acids Research}, + volume = {41}, + number = {D1}, + pages = {D590--D596}, + doi = {10.1093/nar/gks1219}, + url = {https://www.arb-silva.de/}, +} + +@article{mp_ncbi, + author = {Schoch, Conrad L and Ciufo, Stacy and Domrachev, Mikhail and Hotton, Carol L and Kannan, Sivakumar and Khovanskaya, Rogneda and Leipe, Detlef and Mcveigh, Richard and O’Neill, Kathleen and Robbertse, Barbara and Sharma, Shobha and Soussov, Vladimir and Sullivan, John P and Sun, Lu and Turner, Seán and Karsch-Mizrachi, Ilene}, + title = {{NCBI Taxonomy: a comprehensive update on curation, resources and tools}}, + year = {2020}, + journal = {Database}, + volume = {2020}, + pages = {baaa062}, + doi = {10.1093/database/baaa062}, + url = {https://www.ncbi.nlm.nih.gov/taxonomy}, +} + +@article{mp_enzyme, + author = {Bairoch, A.}, + title = {{The ENZYME database in 2000}}, + year = {2000}, + journal = {Nucleic Acids Research}, + volume = {28}, + number = {1}, + pages = {304--305}, + doi = {10.1093/nar/28.1.304}, + url = {https://enzyme.expasy.org/}, +} + +@article{mp_metacyc, + author = {Caspi, Ron and Billington, Richard and Keseler, Ingrid M and Kothari, Anamika and Krummenacker, Markus and Midford, Peter E and Ong, Wai Kit and Paley, Suzanne and Subhraveti, Pallavi and Karp, Peter D}, + title = {{The MetaCyc database of metabolic pathways and enzymes - a 2019 update}}, + year = {2020}, + journal = {Nucleic Acids Research}, + volume = {48}, + number = {D1}, + pages = {D445--D453}, + doi = {10.1093/nar/gkz862}, + url = {https://metacyc.org/}, +} diff --git a/docs/benchmarking.md b/docs/benchmarking.md new file mode 100644 index 0000000..8413887 --- /dev/null +++ b/docs/benchmarking.md @@ -0,0 +1,165 @@ +# Benchmarking and supplementary statistics + +[Home](index.md) · [Reproducibility records](reproducibility.md) · [Result schema](results-schema.md) + +## Define the experiment before starting + +Specify which assemblies, reads, genome assignments, reference releases, tool revisions and resource budgets are being evaluated. Use a complete input manifest and a new output directory for a fresh measurement. A reused task is a cache check, not a newly measured biological stage. + +Freeze the MPDB and SIF during the run. Do not change analysis code, dependencies or reference files while tasks are running. Editing documentation does not change the biological workflow, but keep a source record identifying the code actually used. A Git commit alone is insufficient when the working tree has uncommitted changes. + +## Example full workflow + +```bash +metapathways analysis_wf --manifest /project/benchmark/samples.tsv \ + -o /project/benchmark/fresh-results -d /project/MPDB \ + --annotation_dbs swissprot metacyc \ + --taxprune --taxonomic_scope all \ + --threads 8 --max_cpus 32 --max_memory '64 GB' +``` + +The command uses the registered image. Add `--image /path/to/frozen.sif` to pin it explicitly. Include a maximum CPU and memory budget in the methods. Omitting them uses detected local availability, which can vary by host and cgroup. This is a throughput benchmark with concurrency, not necessarily an isolated per-sample latency measurement. + +## What is recorded automatically + +| Record | Location / meaning | +| --- | --- | +| Resolved inputs and sample IDs | Output-root `inputs.resolved.tsv` | +| Planned community/genome entities | Output-root `inputs.entities.json` | +| User command and console | `logs/cli/` and `logs/analysis_wf/RUN_ID/console.log` | +| Planned commands, CPU/memory reservations and dependencies | `tasks.json` in the invocation directory | +| Task status and measured wrapper elapsed time | `summary.json` and `tasks/*.json` | +| Nextflow timing and sampled resource fields | `trace.tsv`, `report.html`, `timeline.html` | +| Tool-specific counts and warnings | Per-task logs and sample `run_statistics/`, annotation/RNA/abundance outputs | +| PGDB build/export status and internal diagnostics | Per-entity `results/pgdb/.../diagnostics/` | +| Indexed biological tables and source accounting | `reports/results.sqlite`, `schema.json`, `output_inventory.tsv` | + +Default Nextflow traces include status/exit, submit time, duration, realtime, CPU utilization, peak RSS/virtual memory and read/write character counters when available. Availability and units depend on the executor and trace format; retain the original headers and values. A missing measurement is unknown, not zero. `rchar`/`wchar` are process I/O counters, not a direct measurement of physical disk traffic. + +Peak RSS is sampled task memory, not its requested reservation and not the peak of the whole concurrent workflow. Do not sum per-task peaks to claim a simultaneous machine peak. CPU utilization can exceed 100% for threaded work; it is not automatically normalized by the requested CPU count. Approximate CPU-seconds derived from utilization and elapsed time must be labeled as derived, not directly measured. + +The standard trace does **not** provide a continuous whole-server CPU/RAM/disk time series, energy usage or complete network-traffic accounting. Those require a separate host monitor started with the benchmark. For the planned stage-runtime/resource figure, task-level trace measurements are appropriate; they do not support claims about an unmeasured whole-host peak. + +## Supplementary table structure + +Keep linked tables rather than forcing differently scoped quantities into one repeated wide table: + +| Table | Recommended fields | +| --- | --- | +| Sample/input inventory | Sample ID, body site/group, assembly/read/map paths and hashes, read layout, original contig count/bases, input genome-bin count | +| Sample result statistics | Retained contigs/bases, predicted/annotated features with feature types, reference-hit counts, read mapping/counting statistics | +| Entity/pathway statistics | Sample/entity, genome-assignment method, selected-input genes, outcome, base pathways, reactions and explicit unique pathway–gene associations | +| Task performance | Invocation, sample, stage, database/entity where applicable, status, requested CPUs/memory, elapsed time, trace CPU and memory/I/O fields | +| Run provenance | Host/OS/CPU/RAM, MP/environment/reference/SIF versions and hashes, complete command, start/end, concurrency budget and notes | + +Obtain each biological statistic from its authoritative source and define its unit. For example, an annotation-table row count is not necessarily the number of unique genes; a pathway-to-ORF export may contain multiple annotation rows per association. Count unique keys when claiming unique genes or links. The schema importer collapses some repeated associations deliberately. + +Separate input bin count, generated PF input count, attempted PGDB count, successful PGDB count, failed count and skipped count. Do not convert a missing/failed inference into zero pathways. Distinguish base pathway counts from superpathways or the total records in `pathways.dat`. + +## Runtime and resource figures + +For stage comparisons, group tasks by stage and, where relevant, reference database or entity. Show distributions across samples/bins rather than hiding all variation in one mean. Report whether a panel includes fresh successes only, failed attempts, or both. + +Parallel task times overlap. Summing their durations measures accumulated task elapsed time, not workflow makespan. Measure end-to-end wall time from controller start through final report completion. Nextflow's task trace does not include all validation/planning and post-workflow report-indexing time. MP's task elapsed time also includes wrapper overhead; it is not necessarily only the biological executable's compute time. + +Keep separate measurements for MPDB construction and Pathway Tools image construction if they are discussed. Sample workflow traces do not retroactively capture setup performed before the run. Copying/downloading inputs is likewise outside sample-stage measurements unless explicitly included in the experiment. + +## Save environment and machine information + +Run these on the analysis host before the benchmark, adapting the checkout path. Save them beside the manifest, not in a temporary Nextflow work directory: + +```bash +mkdir -p /project/benchmark/provenance +conda list --explicit > /project/benchmark/provenance/conda-explicit.txt +python -m pip freeze > /project/benchmark/provenance/pip-freeze.txt +metapathways version > /project/benchmark/provenance/mp-version.txt +nextflow -version > /project/benchmark/provenance/nextflow-version.txt +apptainer --version > /project/benchmark/provenance/apptainer-version.txt +lscpu > /project/benchmark/provenance/lscpu.txt +free -b > /project/benchmark/provenance/memory.txt +git -C /path/to/MetaPathways rev-parse HEAD > /project/benchmark/provenance/mp-commit.txt +git -C /path/to/MetaPathways status --short > /project/benchmark/provenance/mp-working-tree.txt +``` + +Also retain the source snapshot or checksums for uncommitted/untracked implementation files, the SIF and `.sif.json`, MetaCyc provenance, and reference release/checksum files. For very large raw datasets, compute checksums as a separate acquisition/validation step so that extra I/O does not distort measured analysis performance. + +## Finish, inspect, then summarize + +Read `summary.json` and per-entity `execution.json`, not just the top-level completion message. Inspect required task success, optional failures/skips, input coverage and report import notes. Preserve failed-attempt logs when retrying; a final successful attempt does not erase their computational cost. + +Default cleanup removes disposable work/cache state after archiving diagnostics. It retains final results, receipts, logs, traces and Nextflow report/timeline. Explicit `--work_dir`, `--conda_cache` or `--keep_work` retain additional scratch; retaining it is not necessary merely to make the supplementary resource table. Interrupted runs may also retain scratch for diagnosis. + +The final supplementary table and figure still require extraction, joining and quality checks after completion. The explorer provides biological subsets and an inventory; it does not automatically generate a publication-ready benchmark figure or certify manuscript statements. Audit every claim against the frozen final records. + +## Compare one server with a Slurm cluster + +Test the three small test samples on the cluster before submitting the full benchmark. Install the same feature-branch revision on shared storage, and activate its environment before starting MP. Nextflow submits with the logged-in user's Slurm identity; there are no MP password flags. Confirm that `sbatch`, `squeue`, and `scancel` are available and that compute nodes can use the same software, database and input paths. + +For a first three-sample cluster test, adapt the shared paths and allocation names: + +```bash +metapathways analysis_wf \ + --manifest /shared/project/cami-test/all.tsv \ + -o /shared/project/mp-test-slurm \ + -d /shared/project/MPDB \ + --annotation_dbs swissprot \ + --skip_ptools \ + --threads 4 --max_cpus 8 --memory '4 GB' --max_memory '16 GB' \ + --executor slurm --account my_project --partition compute \ + --max_tasks 2 --submit_rate 6 --time_limit 2h +``` + +The example uses the full public SwissProt/SILVA references in your MPDB. If using the bundled test references, select `swissprot_test` and `SILVA_SSU_test SILVA_LSU_test` as in the test walkthrough. The two-hour limit is a small-input example, not a full CAMI II ([Meyer et al., 2022](#cami-references)) task limit. Follow your site's policy for keeping the Nextflow controller running: some sites permit a persistent headnode session, others require a controller allocation. MP preparation and report generation run in that controller, while scheduled biological stages run on compute nodes. + +For the complete benchmark, copy the original assemblies, reads, CAMI genome maps, reference database and your licensed Pathway Tools SIF onto storage accessible to all selected nodes. Write an HPC-specific manifest pointing there; local server paths and symlinks are not portable. Use a new output directory. Supply the same analysis options, reference release, pruning scope, and image content as on the single server. Pass `--image /shared/project/pathway-tools.sif` explicitly if its user registration differs on the cluster. + +Record a run/scenario ID (`local` or `slurm`), Git commit, environment export, input/reference/SIF checksums, task resource requests, node hardware, scheduler settings, start/end times, and log directory. Compare elapsed run time separately from summed task CPU time; cluster queue delays are part of operational elapsed time but not biological compute time. Do not mix a resumed run with a fresh timing run. If total budgets differ, report that as a throughput/scaling scenario rather than attributing the entire speed difference to Nextflow or Slurm. Keep all per-task traces and both scenario manifests for the supplementary tables. + +For a larger concurrency test, the same per-job settings work on either executor: `--threads 8 --memory '64 GB' --max_tasks 100`. Omit aggregate maxima to avoid manual arithmetic. The local executor uses detected host capacity; Slurm permits up to 100 submitted jobs with no implicit aggregate caps. Add `--submit_rate 60` to allow up to one Slurm submission per second. Partition may be omitted to use the cluster default. + +## Compact results on limited storage + +Add `--compact_results` to `analysis_wf` on either local or Slurm execution. The default keeps all sample outputs. Compact mode is intended for the complete workflow; individual `run`, `mag_split`, and `ptools` commands do not perform this cleanup. + +Each sample gets a final cleanup task that waits for **all** its annotation, read-abundance, splitting and requested PGDB tasks. Expected optional MAG failures count as finished attempts; their diagnostics and status remain available. Required failures prevent that sample's cleanup. Other successfully completed samples can already be compacted while the remaining samples run. + +Retained files include the source tables and identifier maps used by the explorer, final supporting result tables, pathway TSVs, MAG input gene membership, ORF groups, run statistics, compressed PGDBs and bundled diagnostics. MP validates report relationships before deleting anything. Reports and CSV exports can be rebuilt normally with `metapathways report -o OUTPUT`. Complete PGDBs remain available as `*cyc.tar.bz2` archives. + +Compact tasks resolve temporary storage **on the compute node**. On Slurm, MP uses the job's `SLURM_TMPDIR`. Clusters that do not set it require `--scratch_dir /node/local/path`, using a directory available on each worker. MP deliberately does not assume an arbitrary Slurm `TMPDIR` is node-local. On a local server, MP uses the system temporary directory unless overridden. An override takes precedence and must point to an existing writable directory; do not point it at shared scratch if the goal is avoiding shared-filesystem inode pressure. No environment-variable setup is required on clusters that provide `SLURM_TMPDIR`. + +Each task logs its scratch location, free bytes and available inodes. MP requires at least 1 GiB and 1,000 available inodes at startup; this is a basic headroom check, **not a prediction or reservation** of what a large PGDB will need. Jobs on the same node share its disk capacity. Keep sufficient node-local space for concurrent jobs; RAM-backed temporary storage also consumes the job's memory allocation. + +PGDB construction, per-contig sequence/annotation staging, and archive extraction for pathway-table generation all occur in task scratch. MP publishes the PGDB archive and both pathway tables through temporary destination files and atomic renames. A success checkpoint is written only after the task completes. Successful internal diagnostics are bundled as `diagnostics.tar.gz`; failed attempts are bundled as `failed-attempt-*.tar.gz` for recovery. Normal task exit removes local working trees. A hard kill or node failure cannot guarantee recovery of unpublished local files; that task rebuilds on resume. Completed shared results remain reusable. + +FAST search temporary files, parallel tRNA chunks, and SAMtools sorting spill files also use task scratch. Annotation products and mapping BAMs remain on shared storage until sample cleanup so downstream tasks and incomplete-sample checkpoints still have their inputs. Removed sample files include BAMs, sequence intermediates, raw alignments, intermediate GenBank/GFF files, and Pathway Tools inputs except report-required membership records. Original assemblies/reads, MPDB, and SIF are untouched and must reside outside sample output directories. + +At normal controller exit (success or failure), compact mode bundles worker logs and invocation receipts in `logs/analysis_wf/RUN/task_logs.tar.gz` and Nextflow task working files in `nextflow_tasks.tar.gz`. Console logs, task plans, summary JSON, trace TSV, timeline and resource reports stay directly readable. Generated Nextflow code, work and run-local Conda cache are removed. A successful workflow also removes staging links and task-reuse receipts; sample completion markers remain. An interrupted controller retains work and loose logs because scheduler jobs might still be using them. MP does not prune active Nextflow task directories behind the scheduler. + +`--compact_results` cannot be combined with `--keep_work`, `--work_dir` or `--conda_cache`. `--scratch_dir` requires compact mode and controls worker temporary storage, not Nextflow's shared work directory. Installed Mamba environments and user-wide Nextflow installations/caches are not deleted. + +Repeat the compact workflow command to resume. Completed compact samples are skipped; incomplete samples use normal checkpoints. Interrupted cleanup resumes from its marker. Scheduling-only changes (`--max_tasks`, `--max_cpus`, `--max_memory`, executor, account, partition, QoS, reservation, walltime, submission rate, and scratch location) do not invalidate completed compact samples. Changed analysis settings, input/reference metadata, implementation, missing retained files, or `--force_redo` require a new output directory for compacted samples. This implementation change does not migrate older compact completion markers; use fresh output when upgrading an already-compacted benchmark. Do not delete completion markers to bypass checks. + +To inspect bundled task logs without extracting thousands of files: + +```bash +tar -tzf OUTPUT/logs/analysis_wf/RUN/task_logs.tar.gz +# Substitute one member name printed above: +tar -xOzf OUTPUT/logs/analysis_wf/RUN/task_logs.tar.gz tasks/TASK_HASH.log +``` + +For example, add this flag to the existing benchmark command: + +```bash +metapathways analysis_wf \ + --manifest /project/benchmark/samples.tsv \ + -o /project/benchmark/compact-results -d /project/MPDB \ + --image /project/containers/pathway-tools.sif \ + --annotation_dbs swissprot metacyc \ + --taxprune --taxonomic_scope all \ + --threads 8 --compact_results +``` + +The cleanup task appears separately in the execution/resource records; distinguish its time from biological stages when preparing benchmark tables. + +```{include} includes/cami-references.md +``` diff --git a/docs/cami-references.md b/docs/cami-references.md new file mode 100644 index 0000000..578a426 --- /dev/null +++ b/docs/cami-references.md @@ -0,0 +1,12 @@ +# Citing CAMI and CAMI II + +The MP test inputs come from the CAMI II human-microbiome collection ([Meyer et al., 2022](#cami-references)). The original CAMI publication describes the initiative and its first benchmark ([Sczyrba et al., 2017](#cami-references)); it is not a claim that MP's test subset comes from the first challenge. + +Use the study citations and dataset DOI when describing these inputs. The bundled subset and its transformations are documented in the [bundle README](test-bundle.md). + +```{include} includes/cami-references.md +``` + +## Reference-manager download + +{download}`Download BibTeX with complete author lists `. The records contain the publisher-deposited Crossref author lists for both papers. diff --git a/docs/cli-reference.md b/docs/cli-reference.md new file mode 100644 index 0000000..f6cc70a --- /dev/null +++ b/docs/cli-reference.md @@ -0,0 +1,559 @@ +# Complete CLI reference + +[User guide](index.md) · [Workflow guide](workflow.md) + +Generated by `python scripts/generate_cli_docs.py`. Run `metapathways COMMAND --help` for the help of your installed revision. + +## prepare_test + +```text +usage: metapathways [-h] -o OUTPUT_DIR + +Copy the bundled test inputs and reference seeds into a writable workspace. + +options: + -h, --help show this help message and exit + -o OUTPUT_DIR, --output_dir OUTPUT_DIR + workspace for cami-test/ inputs and MPDB/ reference seeds +``` + +## run + +```text +usage: Metapathways run [options] + +Minimum REQUIRED Command: +MetaPathways run -i INPUT_FILE -o OUTPUT_DIR -d REFDB_DIR + +options: + -h, --help show this help message and exit + --dryrun show the execution plan without running tasks + +Minimum Required Arguments: + -i INPUT_FILE, --input_file INPUT_FILE + path to the input fasta file/input dir [REQUIRED] + -o OUTPUT_DIR, --output_dir OUTPUT_DIR + path to the output directory [REQUIRED] + -d REFDB_DIR, --refdb_dir REFDB_DIR + path to the reference DB [REQUIRED] + +Quality Controls Arguments: + --input_format {fasta,fasta-amino} + Input format, FASTA support only [fasta] + --qc_min_length QC_MIN_LENGTH + Minimum length for quality control [180] + --qc_delete_replicates {yes,no} + Delete replicates in quality control [yes] + +ORF Prediction Arguments: + --orf_strand {pos,neg,both} + Strand for ORF prediction [both] + --orf_algorithm ORF_ALGORITHM + Algorithm for ORF prediction, Prodigal support only [prodigal] + --orf_min_length ORF_MIN_LENGTH + Minimum ORF length [60] + --orf_translation_table ORF_TRANSLATION_TABLE + Translation table for ORF prediction, see Prodigal for translation tables [11] + --orf_mode {single,meta} + Mode for ORF prediction [meta] + +Functional Annotation Arguments: + --annotation_algorithm {FAST,BLAST} + Algorithm for ORF annotation [FAST] + --annotation_dbs ANNOTATION_DBS [ANNOTATION_DBS ...] + Database(s) for annotation, space-separated list [swissprot] + --annotation_min_bsr ANNOTATION_MIN_BSR + Minimum BSR for annotation [0.4] + --annotation_max_evalue ANNOTATION_MAX_EVALUE + Maximum e-value for annotation [0.000001] + --annotation_min_score ANNOTATION_MIN_SCORE + Minimum score for annotation [20] + --annotation_min_length ANNOTATION_MIN_LENGTH + Minimum length for annotation [45] + --annotation_max_hits ANNOTATION_MAX_HITS + Maximum hits for annotation [5] + --annotation_run_mode {default,pervol} + Run mode for annotation, FAST only [pervol] + +rRNA Annotation Arguments: + --rRNA_refdbs RRNA_REFDBS [RRNA_REFDBS ...] + Reference databases for rRNA annotation, space-separated list + [SILVA_138.1_LSURef_NR99_tax_silva_trunc SILVA_138.1_SSURef_NR99_tax_silva_trunc] + --rRNA_max_evalue RRNA_MAX_EVALUE + Maximum e-value for rRNA annotation [0.000001] + --rRNA_min_identity RRNA_MIN_IDENTITY + Minimum identity for rRNA annotation [20] + --rRNA_min_bitscore RRNA_MIN_BITSCORE + Minimum bitscore for rRNA annotation [50] + +Read Mapping Arguments (single sample support only): + -1 FWD_FASTQ, --fastq FWD_FASTQ + location of the raw fastq file, either forward or interleaved + -2 REV_FASTQ, --rev_fastq REV_FASTQ + location of the raw reverse fastq file, if separate paired-end + --interleaved if paired-end is interleaved [False] + +Pipeline Step Arguments: + --PREPROCESS_INPUT {yes,skip,redo} + Step: PREPROCESS_INPUT [yes] + --ORF_PREDICTION {yes,skip,redo} + Step: ORF_PREDICTION [yes] + --FILTER_AMINOS {yes,skip,redo} + Step: FILTER_AMINOS [yes] + --SCAN_rRNA {yes,skip,redo} + Step: SCAN_rRNA [yes] + --SCAN_tRNA {yes,skip,redo} + Step: SCAN_tRNA [yes] + --FUNC_SEARCH {yes,skip,redo} + Step: FUNC_SEARCH [yes] + --PARSE_FUNC_SEARCH {yes,skip,redo} + Step: PARSE_FUNC_SEARCH [yes] + --ANNOTATE_ORFS {yes,skip,redo} + Step: ANNOTATE_ORFS [yes] + --GENBANK_FILE {yes,skip,redo} + Step: GENBANK_FILE [yes] + --CREATE_ANNOT_REPORTS {yes,skip,redo} + Step: CREATE_ANNOT_REPORTS [yes] + --PATHOLOGIC_INPUT {yes,skip,redo} + Step: PATHOLOGIC_INPUT [yes] + --COMPUTE_TPM {yes,skip,redo} + Step: COMPUTE_TPM [yes] + --force_redo Redo all steps [False] + +Execution Resources: + --max_cpus MAX_CPUS total CPU budget [local: available CPUs; Slurm: no aggregate cap] + --memory MEMORY memory reservation per task [16 GB] + --max_memory MAX_MEMORY + total memory budget [local: available memory; Slurm: no aggregate cap] + --max_tasks MAX_TASKS + maximum submitted tasks, including queued/running [local: CPU budget; Slurm: 4] + --executor {local,slurm} + execution backend [local]; Slurm uses your logged-in cluster identity + --account ACCOUNT Slurm allocation/account + --partition PARTITION + Slurm partition [cluster default] + --qos QOS Slurm quality of service + --reservation RESERVATION + Slurm reservation + --time_limit TIME_LIMIT + Slurm walltime per task [24h] + --submit_rate SUBMIT_RATE + maximum Slurm submissions per minute [6] + --work_dir WORK_DIR custom Nextflow work directory; retained after the run + --conda_cache CONDA_CACHE + custom Conda cache directory; retained after the run + --keep_work retain automatically allocated work/cache directories + +Miscellaneous Arguments: + -s SAMPLES [SAMPLES ...], --samples SAMPLES [SAMPLES ...] + process only specific samples, space-separated list + -t THREADS, --threads THREADS + threads per capable tool [8], capped by --max_cpus; serial stages use one CPU + -v, --verbose print more information on the stdout + --test use test values for all arguments +``` + +## analysis_wf + +```text +usage: Metapathways analysis_wf [options] + +Annotate multiple metagenomes, split MAGs, build PGDBs and generate a combined report. + +options: + -h, --help show this help message and exit + --dryrun show the execution plan without running tasks + --manifest MANIFEST TSV with sample_id, assembly, read_layout, reads_1, reads_2, mag_map + --reads_dir READS_DIR + Flat reads directory; -i then names the assemblies directory + --mag_maps_dir MAG_MAPS_DIR + Flat contig-to-MAG map directory; -i then names the assemblies directory + --no_reads Explicitly omit read mapping during automatic discovery + --no_mags Explicitly omit MAG splitting during automatic discovery + --skip_ptools Omit community and MAG PGDB construction + --compact_results Use task scratch, archive PGDBs/diagnostics, and remove completed sample intermediates + --scratch_dir SCRATCH_DIR + Worker-local scratch directory for compact mode [Slurm: SLURM_TMPDIR; local: system temporary directory] + --image IMAGE Pathway Tools SIF [registered by build_pt] + --taxon_id TAXON_ID Override PGDB NCBI taxon in private inputs; applies to every selected entity + --taxonomic_scope {all,bacteria,archaea,eukaryotes} + Named PGDB taxon override; all means cellular life; euks aliases eukaryotes. Taxonomic pruning is enabled by default. Default: all (cellular life). + --no_transport_inference + Disable TIP transport inference + --taxprune Enable taxonomic pruning [default] + --no_taxprune Disable taxonomic pruning and perform unpruned rescoring + --ptools_memory PTOOLS_MEMORY + Optional PGDB memory override [same as --memory] + +Minimum Required Arguments: + -i INPUT_FILE, --input_file INPUT_FILE + dataset root, or assemblies directory with --reads_dir/--mag_maps_dir; alternative: --manifest + -o OUTPUT_DIR, --output_dir OUTPUT_DIR + path to the output directory [REQUIRED] + -d REFDB_DIR, --refdb_dir REFDB_DIR + path to the reference DB [REQUIRED] + +Quality Controls Arguments: + --input_format {fasta,fasta-amino} + Input format, FASTA support only [fasta] + --qc_min_length QC_MIN_LENGTH + Minimum length for quality control [180] + --qc_delete_replicates {yes,no} + Delete replicates in quality control [yes] + +ORF Prediction Arguments: + --orf_strand {pos,neg,both} + Strand for ORF prediction [both] + --orf_algorithm ORF_ALGORITHM + Algorithm for ORF prediction, Prodigal support only [prodigal] + --orf_min_length ORF_MIN_LENGTH + Minimum ORF length [60] + --orf_translation_table ORF_TRANSLATION_TABLE + Translation table for ORF prediction, see Prodigal for translation tables [11] + --orf_mode {single,meta} + Mode for ORF prediction [meta] + +Functional Annotation Arguments: + --annotation_algorithm {FAST,BLAST} + Algorithm for ORF annotation [FAST] + --annotation_dbs ANNOTATION_DBS [ANNOTATION_DBS ...] + Database(s) for annotation, space-separated list [swissprot] + --annotation_min_bsr ANNOTATION_MIN_BSR + Minimum BSR for annotation [0.4] + --annotation_max_evalue ANNOTATION_MAX_EVALUE + Maximum e-value for annotation [0.000001] + --annotation_min_score ANNOTATION_MIN_SCORE + Minimum score for annotation [20] + --annotation_min_length ANNOTATION_MIN_LENGTH + Minimum length for annotation [45] + --annotation_max_hits ANNOTATION_MAX_HITS + Maximum hits for annotation [5] + --annotation_run_mode {default,pervol} + Run mode for annotation, FAST only [pervol] + +rRNA Annotation Arguments: + --rRNA_refdbs RRNA_REFDBS [RRNA_REFDBS ...] + Reference databases for rRNA annotation, space-separated list + [SILVA_138.1_LSURef_NR99_tax_silva_trunc SILVA_138.1_SSURef_NR99_tax_silva_trunc] + --rRNA_max_evalue RRNA_MAX_EVALUE + Maximum e-value for rRNA annotation [0.000001] + --rRNA_min_identity RRNA_MIN_IDENTITY + Minimum identity for rRNA annotation [20] + --rRNA_min_bitscore RRNA_MIN_BITSCORE + Minimum bitscore for rRNA annotation [50] + +Pipeline Step Arguments: + --PREPROCESS_INPUT {yes,skip,redo} + Step: PREPROCESS_INPUT [yes] + --ORF_PREDICTION {yes,skip,redo} + Step: ORF_PREDICTION [yes] + --FILTER_AMINOS {yes,skip,redo} + Step: FILTER_AMINOS [yes] + --SCAN_rRNA {yes,skip,redo} + Step: SCAN_rRNA [yes] + --SCAN_tRNA {yes,skip,redo} + Step: SCAN_tRNA [yes] + --FUNC_SEARCH {yes,skip,redo} + Step: FUNC_SEARCH [yes] + --PARSE_FUNC_SEARCH {yes,skip,redo} + Step: PARSE_FUNC_SEARCH [yes] + --ANNOTATE_ORFS {yes,skip,redo} + Step: ANNOTATE_ORFS [yes] + --GENBANK_FILE {yes,skip,redo} + Step: GENBANK_FILE [yes] + --CREATE_ANNOT_REPORTS {yes,skip,redo} + Step: CREATE_ANNOT_REPORTS [yes] + --PATHOLOGIC_INPUT {yes,skip,redo} + Step: PATHOLOGIC_INPUT [yes] + --COMPUTE_TPM {yes,skip,redo} + Step: COMPUTE_TPM [yes] + --force_redo Redo all steps [False] + +Execution Resources: + --max_cpus MAX_CPUS total CPU budget [local: available CPUs; Slurm: no aggregate cap] + --memory MEMORY memory reservation per task [16 GB] + --max_memory MAX_MEMORY + total memory budget [local: available memory; Slurm: no aggregate cap] + --max_tasks MAX_TASKS + maximum submitted tasks, including queued/running [local: CPU budget; Slurm: 4] + --executor {local,slurm} + execution backend [local]; Slurm uses your logged-in cluster identity + --account ACCOUNT Slurm allocation/account + --partition PARTITION + Slurm partition [cluster default] + --qos QOS Slurm quality of service + --reservation RESERVATION + Slurm reservation + --time_limit TIME_LIMIT + Slurm walltime per task [24h] + --submit_rate SUBMIT_RATE + maximum Slurm submissions per minute [6] + --work_dir WORK_DIR custom Nextflow work directory; retained after the run + --conda_cache CONDA_CACHE + custom Conda cache directory; retained after the run + --keep_work retain automatically allocated work/cache directories + +Miscellaneous Arguments: + -t THREADS, --threads THREADS + threads per capable tool [8], capped by --max_cpus; serial stages use one CPU + -v, --verbose print more information on the stdout +``` + +## build_db + +```text +usage: metapathways [-h] [-d PATH] [--func [CATEGORICAL ...]] [-a ALIGNER] [--skip_pt_screen] + [--screen_image SCREEN_IMAGE] [--metacyc_source METACYC_SOURCE] [-t INT] + [--dryrun] [--snakemake [SNAKEMAKE ...]] [--test] [--max_cpus MAX_CPUS] + [--memory MEMORY] [--max_memory MAX_MEMORY] [--max_tasks MAX_TASKS] + [--executor {local,slurm}] [--account ACCOUNT] [--partition PARTITION] + [--qos QOS] [--reservation RESERVATION] [--time_limit TIME_LIMIT] + [--submit_rate SUBMIT_RATE] [--work_dir WORK_DIR] [--conda_cache CONDA_CACHE] + [--keep_work] + +automated database install + +options: + -h, --help show this help message and exit + -t INT, --threads INT + total database-build CPU budget [available CPUs] + --dryrun show the database execution plan without running tasks + --snakemake [SNAKEMAKE ...] + legacy compatibility flags; use the resource flags for new runs + --test build test SwissProt/SILVA references; use -d for the prepared MPDB + directory + +database arguments: + -d PATH, --refdb_dir PATH + path to save the reference DB, [DEFAULT "./"] + --func [CATEGORICAL ...] + functional references, select any combination from ['swissprot', 'cazy', + 'eggnog', 'uniref50', 'uniref90', 'metacyc'], [DEFAULT ['swissprot']] + -a ALIGNER, --aligner ALIGNER + local aligner to index for, select one of ['fast', 'blast'], [DEFAULT + fast] + --skip_pt_screen Skip default PTools reaction compatibility screening for MetaCyc + --screen_image SCREEN_IMAGE + PTools SIF for screening a MetaCyc directory [registered SIF] + --metacyc_source METACYC_SOURCE + licensed MetaCyc data directory or Pathway Tools SIF [registered SIF when + --func includes metacyc] + +Execution Resources: + --max_cpus MAX_CPUS total CPU budget [local: available CPUs; Slurm: no aggregate cap] + --memory MEMORY memory reservation per task [16 GB] + --max_memory MAX_MEMORY + total memory budget [local: available memory; Slurm: no aggregate cap] + --max_tasks MAX_TASKS + maximum submitted tasks, including queued/running [local: CPU budget; + Slurm: 4] + --executor {local,slurm} + execution backend [local]; Slurm uses your logged-in cluster identity + --account ACCOUNT Slurm allocation/account + --partition PARTITION + Slurm partition [cluster default] + --qos QOS Slurm quality of service + --reservation RESERVATION + Slurm reservation + --time_limit TIME_LIMIT + Slurm walltime per task [24h] + --submit_rate SUBMIT_RATE + maximum Slurm submissions per minute [6] + --work_dir WORK_DIR custom Nextflow work directory; retained after the run + --conda_cache CONDA_CACHE + custom Conda cache directory; retained after the run + --keep_work retain automatically allocated work/cache directories +``` + +## mag_split + +```text +usage: Metapathways mag_split [options] + +Minimum REQUIRED Command: +Metapathways mag_split -o output_dir -m contig_mag_map + +options: + -h, --help show this help message and exit + -o OUTPUT_DIR, --output_dir OUTPUT_DIR + path where MP output was saved [REQUIRED] + -m MAG_MAP, --contig_mag_map MAG_MAP + TSV file that contains contig-to-MAG mapping [REQUIRED] + +Execution Resources: + --max_cpus MAX_CPUS total CPU budget [local: available CPUs; Slurm: no aggregate cap] + --memory MEMORY memory reservation per task [16 GB] + --max_memory MAX_MEMORY + total memory budget [local: available memory; Slurm: no aggregate cap] + --max_tasks MAX_TASKS + maximum submitted tasks, including queued/running [local: CPU budget; Slurm: 4] + --executor {local,slurm} + execution backend [local]; Slurm uses your logged-in cluster identity + --account ACCOUNT Slurm allocation/account + --partition PARTITION + Slurm partition [cluster default] + --qos QOS Slurm quality of service + --reservation RESERVATION + Slurm reservation + --time_limit TIME_LIMIT + Slurm walltime per task [24h] + --submit_rate SUBMIT_RATE + maximum Slurm submissions per minute [6] + --work_dir WORK_DIR custom Nextflow work directory; retained after the run + --conda_cache CONDA_CACHE + custom Conda cache directory; retained after the run + --keep_work retain automatically allocated work/cache directories +``` + +## build_pt + +```text +usage: metapathways build_pt [-h] -i INSTALLER [--ptools_version PTOOLS_VERSION] [-d REFDB_DIR] + [--skip_pt_screen] [-a {fast,blast}] [-o OUTPUT_DIR] [-t THREADS] + [--dryrun] [--max_cpus MAX_CPUS] [--memory MEMORY] + [--max_memory MAX_MEMORY] [--max_tasks MAX_TASKS] + [--executor {local,slurm}] [--account ACCOUNT] + [--partition PARTITION] [--qos QOS] [--reservation RESERVATION] + [--time_limit TIME_LIMIT] [--submit_rate SUBMIT_RATE] + [--work_dir WORK_DIR] [--conda_cache CONDA_CACHE] [--keep_work] + +Build a Pathway Tools SIF from a local Linux installer using Nextflow and Apptainer. + +options: + -h, --help show this help message and exit + -i INSTALLER, --installer INSTALLER + local Pathway Tools Linux x86-64 installer + --ptools_version PTOOLS_VERSION + release number if the installer was renamed; otherwise inferred from its + filename + -d REFDB_DIR, --refdb_dir REFDB_DIR + also export and prepare the licensed MetaCyc reference in this MPDB after + building the SIF + --skip_pt_screen Skip default reaction compatibility screening when building MetaCyc with + -d + -a {fast,blast}, --aligner {fast,blast} + MetaCyc reference index format with -d [fast] + -o OUTPUT_DIR, --output_dir OUTPUT_DIR + container directory [~/.local/share/metapathways/containers] + -t THREADS, --threads THREADS + CPUs for image compression [2]; Pathway Tools uses one CPU + --dryrun write and show the build plan without building or registering + +Execution Resources: + --max_cpus MAX_CPUS total CPU budget [local: available CPUs; Slurm: no aggregate cap] + --memory MEMORY memory reservation per task [4 GB] + --max_memory MAX_MEMORY + total memory budget [local: available memory; Slurm: no aggregate cap] + --max_tasks MAX_TASKS + maximum submitted tasks, including queued/running [local: CPU budget; + Slurm: 4] + --executor {local,slurm} + execution backend [local]; Slurm uses your logged-in cluster identity + --account ACCOUNT Slurm allocation/account + --partition PARTITION + Slurm partition [cluster default] + --qos QOS Slurm quality of service + --reservation RESERVATION + Slurm reservation + --time_limit TIME_LIMIT + Slurm walltime per task [24h] + --submit_rate SUBMIT_RATE + maximum Slurm submissions per minute [6] + --work_dir WORK_DIR custom Nextflow work directory; retained after the run + --conda_cache CONDA_CACHE + custom Conda cache directory; retained after the run + --keep_work retain automatically allocated work/cache directories +``` + +## screen_pt + +```text +usage: metapathways [-h] -d REFDB_DIR -o OUTPUT_DIR [--image IMAGE] [--publish] + [--reactions REACTIONS [REACTIONS ...]] [--batch_size BATCH_SIZE] + [--max_tasks MAX_TASKS] [--confirm_runs CONFIRM_RUNS] [--timeout TIMEOUT] + [--scratch_dir SCRATCH_DIR] + +Screen explicit MPDB reaction assignments against a licensed PTools SIF + +options: + -h, --help show this help message and exit + -d REFDB_DIR, --refdb_dir REFDB_DIR + -o OUTPUT_DIR, --output_dir OUTPUT_DIR + --image IMAGE PTools SIF [registered build_pt image] + --publish Save a completed full screen compatibility list in the MPDB + --reactions REACTIONS [REACTIONS ...] + Optional reaction IDs for a targeted screen + --batch_size BATCH_SIZE + --max_tasks MAX_TASKS + Concurrent isolated containers [1] + --confirm_runs CONFIRM_RUNS + --timeout TIMEOUT Seconds per build [1800]; timeout is inconclusive + --scratch_dir SCRATCH_DIR + Local temporary storage for private PGDB builds +``` + +## ptools + +```text +usage: Metapathways ptools [options] + +Minimum REQUIRED Command: +Metapathways ptools -o output_dir + +options: + -h, --help show this help message and exit + -o OUTPUT_DIR, --output_dir OUTPUT_DIR + path where MP output was saved [REQUIRED] + --tag TAG Custom name for ePGDB [optional] + --container Flag only used in containerized env [special flag] + --taxprune Enable taxonomic pruning [default] + --no_taxprune Disable taxonomic pruning and perform unpruned rescoring + --taxon_id TAXON_ID Override PGDB NCBI taxon in private inputs; applies to every selected entity + --taxonomic_scope {all,bacteria,archaea,eukaryotes} + Named PGDB taxon override; all means cellular life; euks aliases eukaryotes. Taxonomic pruning is enabled by default. Default: all (cellular life). + --no_transport_inference + Disable TIP transport inference (SIF only) + --entity ENTITY Build only community or the specified MAG ID + --image IMAGE Pathway Tools SIF [registered by build_pt] + +Execution Resources: + --max_cpus MAX_CPUS total CPU budget [local: available CPUs; Slurm: no aggregate cap] + --memory MEMORY memory reservation per task [4 GB] + --max_memory MAX_MEMORY + total memory budget [local: available memory; Slurm: no aggregate cap] + --max_tasks MAX_TASKS + maximum submitted tasks, including queued/running [local: CPU budget; Slurm: 4] + --executor {local,slurm} + execution backend [local]; Slurm uses your logged-in cluster identity + --account ACCOUNT Slurm allocation/account + --partition PARTITION + Slurm partition [cluster default] + --qos QOS Slurm quality of service + --reservation RESERVATION + Slurm reservation + --time_limit TIME_LIMIT + Slurm walltime per task [24h] + --submit_rate SUBMIT_RATE + maximum Slurm submissions per minute [6] + --work_dir WORK_DIR custom Nextflow work directory; retained after the run + --conda_cache CONDA_CACHE + custom Conda cache directory; retained after the run + --keep_work retain automatically allocated work/cache directories +``` + +## report + +```text +usage: metapathways [-h] -o OUTPUT_DIR [--serve] [--no-rebuild] [--port PORT] [--no-browser] + +Build a navigable report from existing MP outputs; no analyses are rerun. + +options: + -h, --help show this help message and exit + -o OUTPUT_DIR, --output_dir OUTPUT_DIR + sample output or parent containing sample outputs + --serve open the local searchable portal and serve until Ctrl-C + --no-rebuild use an existing report snapshot + --port PORT local port [automatically selected] + --no-browser print the local URL without opening a browser +``` diff --git a/docs/commands.md b/docs/commands.md new file mode 100644 index 0000000..523c3a5 --- /dev/null +++ b/docs/commands.md @@ -0,0 +1,113 @@ +# Command cookbook: run only the modules you need + +[Home](index.md) · [Complete CLI reference](cli-reference.md) · [Stage/output reference](workflow.md) + +Use `metapathways COMMAND --help` for your installed revision. `metapathways version` reports the package version; also record the source Git revision. These examples assume an activated environment and writable output paths. Replace `/path/to/MPDB` and input filenames before running them. + +## Choose a command + +| Goal | Command | Meaning of `-o` | +| --- | --- | --- | +| Prepare the bundled test data | `prepare_test` | Workspace containing inputs and reference FASTAs | +| Annotate one or more assemblies | `run` | Parent of sample directories | +| Annotate, map reads, split genomes, infer pathways and report | `analysis_wf` | Parent of sample directories | +| Prepare public/licensed reference indexes and tables | `build_db` | Use `-d` for the MPDB root | +| Build the licensed Pathway Tools image | `build_pt` | Directory holding generated SIF images | +| Split existing annotations by genome | `mag_split` | One existing sample directory | +| Infer community/genome pathways from existing annotations | `ptools` | One existing sample directory | +| Index/view existing results | `report` | One sample or a parent of samples | + +## prepare_test: try the included data + +```bash +metapathways prepare_test -o ~/mp-test +cd ~/mp-test +metapathways build_db --test -d MPDB +``` + +The first command copies the bundled three-sample inputs and reference FASTAs into a writable workspace. The second builds the reference indexes and supporting tables there. Existing identical inputs are retained; changed inputs are never overwritten. Continue with the [test workflow](test.md). + +## build_db: references before analysis + +```bash +metapathways build_db -d /path/to/MPDB --func swissprot -a fast --dryrun +metapathways build_db -d /path/to/MPDB --func swissprot -a fast +``` + +The first command plans; the second downloads/formats. `--func` selects functional references; RNA, enzyme and taxonomy support are also prepared. Public options include SwissProt, CAZy and UniRef50/90. eggNOG requires the documented pre-supplied FASTA rather than an automatic acquisition rule. MetaCyc requires a licensed source; see [Pathway Tools](pathway-tools.md). + +`-a fast` and `-a blast` choose the annotation index format. Match them with `run/analysis_wf --annotation_algorithm FAST` or `BLAST`. Build into a new directory when changing reference versions during an ongoing experiment. Database download/build time is separate from sample-analysis runtime. + +On this command `-t` is the legacy aggregate CPU limit, unlike per-tool `run --threads`. Prefer explicit `--max_cpus` when communicating a shared resource budget. `--dryrun` performs no downloading/indexing. Historical Snakefiles are not the supported execution path. + +## run: annotations and optional abundance + +```bash +metapathways run -i SampleA.fasta -o results -d /path/to/MPDB \ + -1 SampleA_R1.fastq.gz -2 SampleA_R2.fastq.gz \ + --annotation_dbs swissprot metacyc \ + --threads 8 --max_cpus 16 --max_memory '64 GB' +``` + +This writes `results/SampleA/`. For interleaved pairs replace the two read arguments with `-1 SampleA_interleaved.fastq.gz --interleaved`. Single-end uses `-1` alone. No reads means no read abundance. MP passes the actual R2 to CoverM and uses the exact expected CoverM BAM for counting; a leftover sorted BAM is not a replacement for missing mapping output. + +The command prepares Pathway Tools inputs but does not run Pathway Tools. For protein FASTA, standalone `run --input_format fasta-amino` uses compatible annotation stages; the complete nucleotide workflow and DNA/read/genome-splitting examples do not apply unchanged. + +To inspect the plan add `--dryrun`. To rerun only read mapping/counting with the original other arguments, add `--COMPUTE_TPM redo`. Stage defaults, dependencies and outputs are documented in [workflow stages](workflow.md#annotation-stages). Do not use `--force_redo` merely to retry a failed downstream task: it forces annotation-stage recomputation. + +## analysis_wf: one complete run for N samples + +**For PGDBs, complete the [Pathway Tools installation guide](pathway-tools.md) first.** Build/register your licensed SIF with `metapathways build_pt` before running the command below. To omit pathway inference, add `--skip_ptools`; Pathway Tools is then unnecessary. + +```bash +metapathways analysis_wf --manifest /project/samples.tsv \ + -o analysis -d /path/to/MPDB \ + --annotation_dbs swissprot metacyc \ + --taxprune --taxonomic_scope all \ + --threads 8 --max_cpus 32 --max_memory '64 GB' +``` + +Use [automatic directories or a manifest](inputs.md). All samples share one scheduling budget and dependency graph. Independent tasks can overlap, including mapping and PGDB work. `--skip_ptools` omits inference; discovery's `--no_reads` and `--no_mags` explicitly omit those inputs. In a manifest, blank optional input columns express per-sample omissions. + +The workflow validates inputs and writes a resolved manifest before submitting analyses. Required annotation stages cannot be skipped through `skip`; valid completed tasks are reused automatically. `--ptools_memory '8 GB'` changes only each PGDB reservation. `--memory` changes all workflow task reservations, including PGDBs unless `--ptools_memory` is explicitly set. Increasing the total memory budget does not automatically increase either per-task setting. + +## mag_split: reuse community annotation + +```bash +metapathways mag_split -o results/SampleA -m /project/SampleA.tsv \ + --max_cpus 8 --max_memory '32 GB' +``` + +This needs the sample's completed annotation/PF files, original-contig mapping, authoritative feature table, and your headerless contig-to-genome map. It does not perform binning or redo reference searches. It produces MAG-specific Pathway Tools inputs and preserves the original map for full membership reporting. + +Changing assignments can change per-genome pathway inference even if the community annotation is unchanged. Run `ptools` afterward, or let `analysis_wf` manage the dependency. + +## build_pt and ptools: licensed setup, then inference + +```bash +metapathways build_pt -i /path/to/pathway-tools-29.5-linux-64-tier1-install \ + -o ~/mp-containers -d /path/to/MPDB -a fast +metapathways ptools -o results/SampleA \ + --taxprune --taxonomic_scope all --max_cpus 8 --max_memory '32 GB' +``` + +The first builds/registers the image and optionally prepares MetaCyc annotation references. The second uses existing sample annotations to infer community and available MAG pathways. Use `--entity community` or `--entity MAG_001` to select a single entity. The [Pathway Tools chapter](pathway-tools.md) covers installer acquisition, image selection, patches, licensing, taxonomy, TIP, errors and outputs in detail. + +## report: inspect and export without reanalysis + +```bash +metapathways report -o results +metapathways report -o results --serve --no-rebuild +``` + +The first indexes current results. The second serves that snapshot for searching and CSV exports. Omit `--no-rebuild` when you want a refreshed index. Reports can be built from partial outputs, but cannot reconstruct missing resource measurements or repair old biological results. [Explorer instructions](reports-tutorial.md) include SSH access and safe table joins. + +## Resources, failure and restart + +Use the [resource and Slurm guide](resources.md#resources-and-slurm) for CPU/memory budgets and cluster flags. With `--threads 8 --max_cpus 32`, up to four ready eight-CPU tasks can fit, or a mixture of threaded and single-CPU tasks, subject to memory. Local execution is default; Slurm uses your logged-in identity, not credentials passed to MP. + +Repeat the same command and output location after a recoverable failure. MP checks durable receipts and tracked files, reuses successful compatible tasks, and retries failed work. No explicit Nextflow `-resume` is required on the MP CLI. A new output directory is appropriate for a fresh benchmark, not for a simple restart. + +Changing parameters, references, image, inputs or software can change reuse behavior. Output existence alone is not proof of provenance. Preserve `logs/` and `.metapathways/` receipts if you want to resume. See [restart details](execution.md#logs-temporary-files-and-restarting). + +For storage-limited complete workflows, `analysis_wf --compact_results` cleans each completed sample and retains report sources and diagnostics. It is off by default. Read the [retained files and restart rules](benchmarking.md#compact-results-on-limited-storage) before using it; PGDB archives are retained and temporary PGDB trees stay in worker scratch. diff --git a/docs/conda_env.yml b/docs/conda_env.yml deleted file mode 100644 index d77f331..0000000 --- a/docs/conda_env.yml +++ /dev/null @@ -1,5 +0,0 @@ -channels: - - conda-forge -dependencies: - - sphinx - - sphinx_rtd_theme==2.0.0 diff --git a/docs/conf.py b/docs/conf.py new file mode 100644 index 0000000..3a67096 --- /dev/null +++ b/docs/conf.py @@ -0,0 +1,44 @@ +"""Build the guides without importing MP or installing analysis tools.""" +import ast +import os +from pathlib import Path + +project = 'MetaPathways' +author = 'Hallam Lab and MetaPathways contributors' +copyright = '2026, MetaPathways contributors' +metadata = ast.parse((Path(__file__).parents[1] / 'metapathways/_version.py').read_text()) +release = next(ast.literal_eval(n.value) for n in metadata.body + if isinstance(n, ast.Assign) and any(isinstance(t, ast.Name) and t.id == '__version__' for t in n.targets)) +version = release +extensions = ['myst_parser', 'sphinxcontrib.mermaid'] +source_suffix = {'.md': 'markdown', '.rst': 'restructuredtext'} +root_doc = 'index' +exclude_patterns = ['_build', 'build', 'src', 'validation', 'requirements.txt', 'includes'] +myst_heading_anchors = 4 +myst_fence_as_directive = ['mermaid'] +html_theme = 'sphinx_rtd_theme' +html_theme_options = {'collapse_navigation': False, 'navigation_depth': 2} +html_title = f'MetaPathways {release}' +html_baseurl = os.environ.get('READTHEDOCS_CANONICAL_URL', 'https://hallamlab-metapathways.readthedocs.io/en/latest/') +html_static_path = ['assets'] +html_css_files = ['docs.css'] +mermaid_version = '11.12.1' +mermaid_init_config = {'startOnLoad': False, 'theme': 'neutral', 'flowchart': {'htmlLabels': False}} +mermaid_fullscreen = True + +# Preserve the entry points from the previous Sphinx site. +templates_path = ['templates'] + + +def legacy_pages(app): + for old, new in {'quick_start': 'installation', 'install': 'installation', 'usage': 'commands'}.items(): + yield old, {'redirect_target': new + '.html'}, 'redirect.html' + + +def setup(app): + app.connect('html-collect-pages', legacy_pages) + +# The extension runs Mermaid on window load; avoid a second automatic render. +mermaid_light_theme = 'neutral' +mermaid_dark_theme = 'neutral' +mermaid_height = 'auto' diff --git a/docs/containers.md b/docs/containers.md new file mode 100644 index 0000000..799793a --- /dev/null +++ b/docs/containers.md @@ -0,0 +1,71 @@ +# Quay containers: Docker and Apptainer + +[Home and quick start](installation.md) · [Test dataset](test.md) · [Licensed Pathway Tools](pathway-tools.md) + +Use the versioned `quay.io/hallamlab/metapathways:4.0.0` image on Linux x86-64. It includes MP, its workflow dependencies (including MAGSplitter and Camelot), and the three-sample test dataset derived from CAMI II ([Meyer et al., 2022](#cami-references)). Production references and licensed Pathway Tools are supplied separately. + +## Docker three-sample test + +Create a working directory and open a shell in the image. The mount keeps inputs, references and results on your host. Run these commands in your host terminal: + +```bash +mkdir -p ~/mp-test-docker +cd ~/mp-test-docker +docker pull quay.io/hallamlab/metapathways:4.0.0 +docker run --rm -it --network host --user "$(id -u):$(id -g)" \ + -v "$PWD:/work" -w /work quay.io/hallamlab/metapathways:4.0.0 bash +``` + +Inside the container, run: + +```bash +metapathways prepare_test -o . +metapathways build_db --test -d MPDB +metapathways analysis_wf \ + --manifest cami-test/all.tsv -o all -d MPDB \ + --annotation_dbs swissprot_test \ + --rRNA_refdbs SILVA_SSU_test SILVA_LSU_test \ + --skip_ptools --threads 4 --memory '4 GB' --max_tasks 2 +metapathways report -o all --serve --no-browser --port 8765 +``` + +Open the printed URL in your browser. On a remote Linux host, use the [SSH tunnel instructions](reports-tutorial.md#view-a-remote-report-through-ssh). Host networking makes the server's loopback address available on that host. Keep the server running while browsing; Ctrl-C stops it. Type `exit` to close the container shell. Files in the working directory remain on the host and belong to your user. + +To return later, repeat the `docker run` command from the same host directory, then run `metapathways report -o all --serve --no-browser --port 8765`. + +## Apptainer three-sample test + +Create a working directory and pull the image on an internet-connected host: + +```bash +mkdir -p ~/mp-test-apptainer +cd ~/mp-test-apptainer +apptainer pull metapathways.sif docker://quay.io/hallamlab/metapathways:4.0.0 +apptainer exec --bind "$PWD:/work" --pwd /work metapathways.sif bash +``` + +Inside the container, run: + +```bash +metapathways prepare_test -o . +metapathways build_db --test -d MPDB +metapathways analysis_wf \ + --manifest cami-test/all.tsv -o all -d MPDB \ + --annotation_dbs swissprot_test \ + --rRNA_refdbs SILVA_SSU_test SILVA_LSU_test \ + --skip_ptools --threads 4 --memory '4 GB' --max_tasks 2 +metapathways report -o all --serve --no-browser --port 8765 +``` + +Open the printed URL, using an [SSH tunnel](reports-tutorial.md#view-a-remote-report-through-ssh) if remote. Ctrl-C stops the report server; `exit` closes the shell. Your inputs, references and results remain in the host working directory. The SIF stays read-only. Pulling this MP image does not require building a custom image or using `--fakeroot`. + +## From the test to your own data + +The reference build downloads enzyme and taxonomy support files. Complete reference preparation on an internet-connected host before using offline compute nodes. The tiny test references are for testing only. Build production references with `metapathways build_db -d MPDB --func swissprot -a fast` in a separate project directory. + +Mount all inputs, references and outputs into the container and use their container-visible paths in manifests. For Slurm, use the [Mamba installation](installation.md#1-conda-package-with-mamba-preferred) on shared storage so the controller and compute jobs can use the same environment. Follow the [resource and Slurm guide](resources.md#resources-and-slurm) for submission limits. + +The public MP image does not include Pathway Tools or MetaCyc. Licensed users should follow the [Pathway Tools guide](pathway-tools.md) to build a separate SIF from their own installer. That image can be copied to another compatible host and selected with `--image`. + +```{include} includes/cami-references.md +``` diff --git a/docs/data-flow.md b/docs/data-flow.md new file mode 100644 index 0000000..fa24486 --- /dev/null +++ b/docs/data-flow.md @@ -0,0 +1,18 @@ +# Data flow: inputs to connected results + +For the tool-by-tool diagrams and citations, see the [detailed workflow](detailed-workflow.md). + +The arrows show which data products feed each calculation. They do not require independent branches to run serially. MP preserves sample, contig, feature and entity identifiers so the final tables can be joined and explored. + +[![Data flow](assets/diagrams/data-flow.svg)](assets/diagrams/data-flow.svg) + +[Zoom diagram](assets/diagrams/data-flow.svg) · [Mermaid source](diagrams/data-flow.mmd) + +- Without reads, MP does not compute read abundance. +- Without a genome map, community analysis can still proceed; genome splitting is omitted. +- With `--skip_ptools`, annotation, abundance and reports remain available, but PGDB inference is omitted. +- PGDB staging joins annotations back to actual contig sequences and feature coordinates. Both community and genome PGDBs need sequence-backed inputs. + +Abundance is calculated from the reads and features, independently of pathway inference. A pathway–gene association can be joined to abundance through the corresponding feature IDs; these are different measurements, not interchangeable values. Each annotation's taxonomy belongs to that annotation's reference database. See the [result schema](results-schema.md) for table keys and safe joins. + +Follow the [complete workflow](analysis.md) for commands, or the [stage reference](workflow.md) for individual products and dependencies. diff --git a/docs/databases.md b/docs/databases.md new file mode 100644 index 0000000..ff76ee4 --- /dev/null +++ b/docs/databases.md @@ -0,0 +1,13 @@ +# Prepare reference databases + +For a minimal reference database: + +```bash +metapathways build_db -d /data/MPDB --func swissprot -a fast +``` + +SILVA and supporting enzyme/taxonomy files are built alongside functional references. Select multiple functional databases with `--func swissprot cazy uniref50`; `uniref90` is also supported and substantially larger. `-a blast` builds BLAST indexes instead of FAST indexes. Use `--dryrun` to inspect the planned downloads and indexing jobs first. + +The builder downloads public references from their configured upstream locations. These are not all version-pinned. eggNOG requires a local FASTA at `MPDB/functional/eggnog`; the old builder had no working eggNOG acquisition rule. Licensed MetaCyc reference acquisition is separate from this public builder. Existing compatible MPDB installations can be used directly with `-d`. + +Choose annotation databases actually present in your MPDB. The FAST/BLAST choice in `run` must match its indexes. Keep reference release records and checksums; changing a database can change biological results. diff --git a/docs/detailed-workflow.md b/docs/detailed-workflow.md new file mode 100644 index 0000000..514a9f5 --- /dev/null +++ b/docs/detailed-workflow.md @@ -0,0 +1,287 @@ +# Detailed workflow and tool citations + +```{container} mp-primary-workflow +[![Detailed MetaPathways nodal workflow from reference setup through annotation, abundance, pathway inference and reporting](assets/workflow-detailed.svg)](assets/workflow-detailed.svg) +``` + +[Open full-size SVG](assets/workflow-detailed.svg) · [Download PDF](assets/workflow-detailed.pdf) · [Brief overview](index.md) + +Process names appear above the diamonds; tools and major libraries appear below. +The numbered modules group related work. Reference/image setup is reusable; +read abundance and community/genome PGDB tasks follow their own dependencies. +The three diagrams below expand those dependencies and their inputs. + +This page follows a nucleotide assembly through MetaPathways, from reference preparation to connected annotation, abundance and pathway tables. It identifies the software responsible for each calculation and explains how Nextflow schedules it. For a shorter introduction, see the [conceptual overview](overview.md), [local/HPC architecture](architecture.md) and [data-flow overview](data-flow.md). + +**If you want pathway/genome databases (PGDBs), complete the [Pathway Tools installation guide](pathway-tools.md) first.** Annotation, read abundance and reporting can run with `--skip_ptools`. + +## How to cite your analysis + +Cite [MetaPathways](#metapathways), [Nextflow](#nextflow), the biological tools used by your selected stages, and the reference databases you searched. Add [Pathway Tools](#pathway-tools), [MetaCyc](#metacyc), MAGSplitter and Camelot when those components are used. Cite the container runtime and Slurm when applicable. The tables below connect each component to its publication or official project; the [full references](#full-references) include DOI links and websites. + +Download {download}`the software and database bibliography `. Papers describe methods, not the precise software or database versions in your run: also record the MP version, environment, reference releases, image identity, settings and execution logs. See [reproducibility](reproducibility.md). For the bundled test data, use the separate [CAMI and CAMI II citations](cami-references.md). + +## 1. Prepare references and the optional Pathway Tools image + +These are setup commands, run before sample analysis. An existing compatible MPDB and SIF can be reused across samples and computers; they are not rebuilt for each sample. + +[![Detailed workflow 1](assets/diagrams/detailed-workflow-1.svg)](assets/diagrams/detailed-workflow-1.svg) + +[Zoom diagram](assets/diagrams/detailed-workflow-1.svg) · [Mermaid source](diagrams/detailed-workflow-1.mmd) + +`build_db` prepares selected public references and the supporting MPDB structure. FAST uses `fastdb`; BLAST+ uses `makeblastdb`. rRNA searches require nucleotide BLAST indexes. MP's own preparation scripts build the lookup tables consumed by annotation and reporting. Database options and custom references are described in [database construction](databases.md). + +`build_pt` builds and validates a private Apptainer image from the licensed installer. With its optional MPDB destination, MP extracts the matching MetaCyc protein FASTA **and companion flat files**, builds the selected search index, and derives the EC/reaction/pathway mappings. This licensed preparation is separate from the public-reference build. See [MetaCyc preparation](pgdb-workflow.md#metacyc-from-pathway-tools). + +The BLAST+ installation inside the Pathway Tools image supports Pathway Tools' own sequence databases. It is distinct from the FAST/BLAST choice for MP's functional annotation searches. Docker is an alternative way to run the MP application image; it is not required to build the Pathway Tools SIF. + +## 2. Plan and schedule work + +`analysis_wf` validates the manifest or automatically matched input layout, resolves sample IDs, and builds a task dependency graph. The Python controller writes modular Nextflow definitions plus task specifications. Nextflow runs eligible tasks locally or submits them to Slurm; it does not submit every downstream task before its inputs are ready. See [input organization](inputs.md) and [execution](execution.md). + +Each worker checks its task receipt and tracked inputs before either reusing valid outputs or invoking the recorded command. Tasks write biological outputs into the sample's output directory and retain command output, status and resource diagnostics. Nextflow's trace, report and timeline describe the invocation; MP's receipts also distinguish reused work and optional entity outcomes. + +| Scheduling level | What can run together | +| --- | --- | +| Across samples | Ready stages from different samples can overlap. Samples are not processed one at a time. | +| Within an annotation stage | Independent searches or parses for different reference databases can overlap. The following stage waits for that group. | +| After Pathway Tools input preparation | Read abundance, the community PGDB, and MAG splitting can proceed independently. Each MAG PGDB waits for splitting. | +| Within a task | Capable tools receive the requested thread budget. pProdigal and ptRNAscan distribute work across their underlying predictors. Serial stages and each Pathway Tools task reserve one CPU. | +| Across executors | Local execution fits aggregate CPU/memory limits. Slurm uses per-job requests plus submission/queue limits; dependencies remain managed by Nextflow. | + +`--threads` controls capable tools, not the total workflow concurrency. `--max_cpus`, `--max_memory`, `--max_tasks` and Slurm submission settings are explained in [resources](resources.md). Multiple processes shown inside one task, such as CoverM, SAMtools and featureCounts, run as subcommands of that task; they are not separate Slurm jobs. + +## 3. Predict and annotate features + +This diagram follows the **stage order for nucleotide FASTA input**. Boxes grouping multiple MP transformations are abbreviated for readability. The reference searches within a group can run concurrently; this does not imply that all RNA prediction and protein search stages run independently within a sample. Protein-only inputs follow the compatible reduced path described in the [stage reference](workflow.md). + +[![Detailed workflow 2](assets/diagrams/detailed-workflow-2.svg)](assets/diagrams/detailed-workflow-2.svg) + +[Zoom diagram](assets/diagrams/detailed-workflow-2.svg) · [Mermaid source](diagrams/detailed-workflow-2.mmd) + +pProdigal parallelizes [Prodigal](#prodigal) gene prediction. MP derives and filters protein sequences before searching the selected protein references with FAST or [BLAST+](#blast). MP computes sequence-based reference scores and then applies its configured search thresholds and score-ratio rules. FAST is a threaded implementation derived from [LAST](#last); citing FAST does not mean the separate LAST executable was run. + +Barrnap predicts rRNA features; BLAST+ searches their sequences against the chosen rRNA references. ptRNAscan partitions the input and starts single-threaded [tRNAscan-SE](#trnascan-se) workers within the assigned CPU budget. Barrnap and tRNAscan-SE use their own underlying RNA search software/models, including [nhmmer/HMMER](#nhmmer--hmmer) and [Infernal](#infernal), respectively. These are not additional MP-level workflow stages. + +MP combines protein and RNA evidence, performs feature-overlap processing through [pybedtools](#pybedtools)/[BEDTools](#bedtools), and writes annotation tables and annotated sequence products. An annotation row's taxonomy comes from **that row's reference database**. Hit taxonomy and the within-database lowest common ancestor (LCA) remain distinct; MP does not borrow a taxon assignment from another database to fill the row. References without supported taxonomy produce the documented unavailable status. See [annotation interpretation](annotation.md). + +## 4. Measure abundance, build PGDBs and assemble reports + +Solid arrows show the downstream processing paths; dotted arrows connect existing products to reporting. Additional inputs are named inside the relevant boxes. The abundance branch requires reads. The PGDB branch requires Pathway Tools; genome-specific PGDBs additionally require a contig-to-genome map. + +[![Detailed workflow 3](assets/diagrams/detailed-workflow-3.svg)](assets/diagrams/detailed-workflow-3.svg) + +[Zoom diagram](assets/diagrams/detailed-workflow-3.svg) · [Mermaid source](diagrams/detailed-workflow-3.mmd) + +[CoverM](#coverm) generates contig abundance and the BAM used for feature counting. MP invokes `coverm contig` without overriding its mapper, so the mapper follows the installed CoverM version's default. Check that version and its logged command/backend when recording methods; a directory called `bwa/` is **not** evidence that BWA performed the mapping. Backend citations are listed below for use when applicable. + +[SAMtools](#samtools) name-sorts the expected CoverM BAM. [featureCounts](#featurecounts), distributed with Subread, uses the annotated feature coordinates; MP's `abund_calc.py` derives the feature abundance table. Contig-level coverage and feature-level counts are separate measurements. Reads are not required to predict features or infer pathways. + +MAGSplitter reuses the community's existing annotations and the supplied membership map; it does not bin contigs, rerun protein searches or infer genome quality. MP stages sequence-backed inputs for both community and genome entities, preserving feature identities and recording input normalization. This provides Pathway Tools with sequence and coordinate information as well as functional annotations. See [PGDB input preparation](pgdb-workflow.md). + +[Pathway Tools/PathoLogic](#pathway-tools) builds each PGDB using the MetaCyc reference in the selected installation. Taxonomic scope/pruning and transport inference are analysis settings; they do not change MP's upstream taxonomic annotation tables. After construction, Pathway Tools exports the database and Camelot/MP extracts pathway-to-gene associations. Independent containerized entities can run concurrently, with one CPU per Pathway Tools task. + +A required community PGDB failure stops dependent work. Optional MAG PGDB failures are recorded and do not become successful empty pathway predictions. With `--skip_ptools`, the PGDB branch is omitted. With compact results enabled, per-sample cleanup waits for that sample's terminal tasks and retains reporting inputs and diagnostics. See [execution and cleanup](execution.md). + +After successful workflow completion, the MP controller builds the combined report and SQLite index. This is controller-side postprocessing, not a separate Slurm biological job. The explorer follows sample-scoped contig, feature and entity relationships and exports selected tables; it does not rerun annotation or pathway inference. See the [report tutorial](reports-tutorial.md) and [results schema](results-schema.md). + +## Tool and wrapper index + +“MP” in the diagrams means code shipped with MetaPathways. These transformations are covered by the MetaPathways citation. A repository link is supplied where a separate paper has not been established; it is not a claim that a paper does not exist. + +| Component | Role in this workflow | Citation and official project | +| --- | --- | --- | +| MetaPathways | Validation, planning, QC, parsing, LCA, annotation integration, PGDB staging, abundance calculations and reporting | [Publication](#metapathways); [GitHub](https://github.com/hallamlab/MetaPathways) | +| Nextflow | Task dependencies, execution, trace and scheduling | [Publication](#nextflow); [website](https://www.nextflow.io/) | +| pProdigal / Prodigal | Parallel wrapper / protein-coding gene prediction | [pProdigal](https://github.com/sjaenick/pprodigal); [Prodigal paper](#prodigal); [Prodigal](https://github.com/hyattpd/Prodigal) | +| FAST | Protein search and reference indexing (`fastal`, `fastdb`) | [GitHub](https://github.com/hallamlab/FAST); [LAST method ancestry](#last) | +| NCBI BLAST+ | Alternative protein searches, rRNA searches and database indexing | [Publication](#blast); [website](https://blast.ncbi.nlm.nih.gov/) | +| barrnap / nhmmer | rRNA prediction / underlying profile search | [barrnap](https://github.com/tseemann/barrnap); [nhmmer paper](#nhmmer--hmmer); [HMMER](http://hmmer.org/) | +| ptRNAscan / tRNAscan-SE / Infernal | Parallel wrapper / tRNA prediction / underlying RNA search | {download}`MP wrapper source <../dev/ptRNAscan.py>`; [tRNAscan-SE paper](#trnascan-se), [website](https://trna.ucsc.edu/tRNAscan-SE/); [Infernal paper](#infernal), [website](http://eddylab.org/infernal/) | +| pybedtools / BEDTools | Feature interval comparison and overlap processing | [pybedtools paper](#pybedtools), [GitHub](https://github.com/daler/pybedtools); [BEDTools paper](#bedtools), [GitHub](https://github.com/arq5x/bedtools2) | +| CoverM | Read mapping orchestration and contig coverage measurements | [Publication](#coverm); [GitHub](https://github.com/wwood/CoverM) | +| SAMtools | BAM processing for read abundance | [Publication](#samtools); [website](https://www.htslib.org/) | +| featureCounts / Subread | Assign alignments to annotated features | [Publication](#featurecounts); [website](https://subread.sourceforge.net/) | +| MAGSplitter | Split annotated features using contig-to-genome membership | [GitHub](https://github.com/hallamlab/MAGSplitter) | +| Pathway Tools / PathoLogic | Build, save and export pathway/genome databases | [Publication](#pathway-tools); [website](https://www.pathwaytools.org/) | +| Camelot (`camelot-frs`) | Load PGDB flat-file frames and query pathway/gene relationships | [Bitbucket](https://bitbucket.org/tomeraltman/camelot-frs/) | +| Minimap2, BWA, Strobealign | Read-mapping backends; cite the backend actually used by CoverM | [Minimap2](#minimap2), [GitHub](https://github.com/lh3/minimap2); [BWA](#bwa), [GitHub](https://github.com/lh3/bwa); [Strobealign](#strobealign), [version-specific citation guidance](https://github.com/ksahlin/strobealign#citation) | + +The BWA reference below describes the original BWA method; for a run using BWA-MEM, also cite Li (2013), *Aligning sequence reads, clone sequences and assembly contigs with BWA-MEM*, [arXiv:1303.3997](https://arxiv.org/abs/1303.3997). Follow Strobealign's version-specific citation guidance rather than assuming one article covers every backend release. Installed backend packages do not imply that all of them ran. + +## Reference databases + +Choose citations for the references actually used, and report their release dates or versions. Searching a database is not the same as running its authors' annotation software: for example, MP's eggNOG reference search does not invoke eggNOG-mapper. + +| Reference | Use | Citation and official website | +| --- | --- | --- | +| UniProtKB/Swiss-Prot | Protein function and supported taxon identifiers | [UniProt](#uniprot); [uniprot.org](https://www.uniprot.org/) | +| UniRef | Clustered protein reference sequences and supported taxonomy | [UniRef](#uniref); [UniRef](https://www.uniprot.org/uniref) | +| CAZy | Carbohydrate-active enzyme reference annotations | [CAZy](#cazy); [cazy.org](https://www.cazy.org/); [dbCAN download service](https://bcb.unl.edu/dbCAN2/) used by the MP reference builder | +| eggNOG | Orthology-based functional reference, when supplied/selected | [eggNOG](#eggnog); [eggnog.embl.de](http://eggnog.embl.de/) | +| SILVA | Ribosomal RNA reference searches | [SILVA](#silva); [arb-silva.de](https://www.arb-silva.de/) | +| NCBI Taxonomy | Taxon names and lineage tree for supported reference hits/LCA | [NCBI Taxonomy](#ncbi-taxonomy); [NCBI](https://www.ncbi.nlm.nih.gov/taxonomy) | +| ENZYME / ExPASy | EC nomenclature and supporting enzyme tables | [ENZYME](#enzyme); [enzyme.expasy.org](https://enzyme.expasy.org/) | +| MetaCyc | Licensed optional protein reference and Pathway Tools' pathway knowledge base | [MetaCyc](#metacyc); [metacyc.org](https://metacyc.org/) | + +MetaCyc has two roles: selecting it as an MP protein annotation database is optional, while Pathway Tools uses its own installed MetaCyc knowledge base for inference. Omitting the protein search does not remove that inference reference. Custom references require their own provenance and citations. + +## Supporting software and distribution + +These components support execution, parsing, packaging or presentation; they are not all standalone biological tasks. Exact transitive dependencies vary with the resolved environment. The environment specification and exported package list are the version inventory for a particular run. + +| Component | Role and reference | +| --- | --- | +| Apptainer | Private SIF build/execution. [Project](https://apptainer.org/), [GitHub](https://github.com/apptainer/apptainer), [recommended foundational Singularity paper](#apptainer--singularity). The project recommends that paper under its former name; record the actual Apptainer version separately. | +| Slurm | Optional cluster scheduler. [Publication](#slurm), [official documentation](https://slurm.schedmd.com/). | +| Mamba, Conda, Bioconda | Environment/package installation. [Mamba](https://github.com/mamba-org/mamba), [Conda](https://github.com/conda/conda), [Bioconda paper](#bioconda), [Bioconda](https://bioconda.github.io/). | +| Docker and Quay | Alternative application container runtime and image distribution. [Docker](https://www.docker.com/), [Quay](https://quay.io/). | +| Python, Java/OpenJDK, Groovy | MP and Nextflow implementation/runtime stack. [Python](https://www.python.org/), [OpenJDK](https://openjdk.org/), [Groovy](https://groovy-lang.org/). | +| pandas, NumPy, SciPy | Tabular/numerical processing and supporting dependencies. [pandas paper](#pandas), [project](https://pandas.pydata.org/); [NumPy paper](#numpy), [project](https://numpy.org/); [SciPy paper](#scipy), [project](https://scipy.org/). | +| pyfastx, pysam | Sequence access and alignment-format support in the software environment. [pyfastx paper](#pyfastx), [GitHub](https://github.com/lmdu/pyfastx); [pysam](https://github.com/pysam-developers/pysam). | +| sexpdata, html2text | Parse Lisp-style data and convert HTML text during PGDB extraction. [sexpdata](https://github.com/jd-boyd/sexpdata), [html2text](https://github.com/Alir3z4/html2text). | +| tqdm, peppy | Supporting progress/configuration dependencies. [tqdm](https://github.com/tqdm/tqdm), [peppy](https://github.com/pepkit/peppy). | +| SQLite | Disk-backed relational report index through Python's SQLite interface. [sqlite.org](https://www.sqlite.org/). | +| Cython, setuptools, pip | Build/install support. [Cython paper](#cython), [project](https://cython.org/); [setuptools](https://setuptools.pypa.io/), [pip](https://pip.pypa.io/). | +| curl, GNU Wget, urllib3 | Download/HTTP support. [curl](https://curl.se/), [Wget](https://www.gnu.org/software/wget/), [urllib3](https://urllib3.readthedocs.io/). | +| Xvfb/X.Org, xauth, ncurses, OpenSSL, libxml2 | Headless Pathway Tools display and container/runtime support. [X.Org](https://www.x.org/), [ncurses](https://invisible-island.net/ncurses/), [OpenSSL](https://www.openssl.org/), [libxml2](https://gitlab.gnome.org/GNOME/libxml2). | +| Sphinx, MyST, Read the Docs theme, Mermaid | Documentation build and these diagrams. [Sphinx](https://www.sphinx-doc.org/), [MyST](https://myst-parser.readthedocs.io/), [theme](https://sphinx-rtd-theme.readthedocs.io/), [sphinxcontrib-mermaid](https://github.com/mgaitan/sphinxcontrib-mermaid), [Mermaid](https://mermaid.js.org/), [Read the Docs](https://about.readthedocs.com/). | + +Historical Snakefiles, older mapping helpers and bundled legacy executables are not evidence that those routes ran. The supported controller uses Nextflow, and the abundance route above uses CoverM followed by featureCounts. Cite software from the actual commands and environment used for your analysis. + +## Full references + +The references below identify methods and resources; publication versions are not pins for the installed software or MPDB release. Wrappers without a verified separate publication are linked in the tool index above. + + +### MetaPathways + +McLaughlin RJ, Liu TX, Altman T, et al. (2024). *MetaPathways v3.5: Modularity and Scalability Improvements for Pathway Inference from Environmental Genomes*. bioRxiv. [doi:10.1101/2024.06.04.597460](https://doi.org/10.1101/2024.06.04.597460). [Official project](https://github.com/hallamlab/MetaPathways). + +### Nextflow + +Di Tommaso P, Chatzou M, Floden EW, et al. (2017). *Nextflow enables reproducible computational workflows*. Nature Biotechnology 35(4): 316–319. [doi:10.1038/nbt.3820](https://doi.org/10.1038/nbt.3820). [Official project](https://www.nextflow.io/). + +### Prodigal + +Hyatt D, Chen GL, LoCascio PF, et al. (2010). *Prodigal: prokaryotic gene recognition and translation initiation site identification*. BMC Bioinformatics 11(1): 119. [doi:10.1186/1471-2105-11-119](https://doi.org/10.1186/1471-2105-11-119). [Official project](https://github.com/hyattpd/Prodigal). + +### LAST + +Kiełbasa SM, Wan R, Sato K, et al. (2011). *Adaptive seeds tame genomic sequence comparison*. Genome Research 21(3): 487–493. [doi:10.1101/gr.113985.110](https://doi.org/10.1101/gr.113985.110). [Official project](https://gitlab.com/mcfrith/last). + +### BLAST+ + +Camacho C, Coulouris G, Avagyan V, et al. (2009). *BLAST+: architecture and applications*. BMC Bioinformatics 10(1): 421. [doi:10.1186/1471-2105-10-421](https://doi.org/10.1186/1471-2105-10-421). [Official project](https://blast.ncbi.nlm.nih.gov/). + +### nhmmer / HMMER + +Wheeler TJ, Eddy SR. (2013). *nhmmer: DNA homology search with profile HMMs*. Bioinformatics 29(19): 2487–2489. [doi:10.1093/bioinformatics/btt403](https://doi.org/10.1093/bioinformatics/btt403). [Official project](http://hmmer.org/). + +### Infernal + +Nawrocki EP, Eddy SR. (2013). *Infernal 1.1: 100-fold faster RNA homology searches*. Bioinformatics 29(22): 2933–2935. [doi:10.1093/bioinformatics/btt509](https://doi.org/10.1093/bioinformatics/btt509). [Official project](http://eddylab.org/infernal/). + +### tRNAscan-SE + +Chan PP, Lin BY, Mak AJ, et al. (2021). *tRNAscan-SE 2.0: improved detection and functional classification of transfer RNA genes*. Nucleic Acids Research 49(16): 9077–9096. [doi:10.1093/nar/gkab688](https://doi.org/10.1093/nar/gkab688). [Official project](https://trna.ucsc.edu/tRNAscan-SE/). + +### pybedtools + +Dale RK, Pedersen BS, Quinlan AR. (2011). *Pybedtools: a flexible Python library for manipulating genomic datasets and annotations*. Bioinformatics 27(24): 3423–3424. [doi:10.1093/bioinformatics/btr539](https://doi.org/10.1093/bioinformatics/btr539). [Official project](https://github.com/daler/pybedtools). + +### BEDTools + +Quinlan AR, Hall IM. (2010). *BEDTools: a flexible suite of utilities for comparing genomic features*. Bioinformatics 26(6): 841–842. [doi:10.1093/bioinformatics/btq033](https://doi.org/10.1093/bioinformatics/btq033). [Official project](https://github.com/arq5x/bedtools2). + +### CoverM + +Aroney STN, Newell RJP, Nissen JN, et al. (2025). *CoverM: read alignment statistics for metagenomics*. Bioinformatics 41(4): btaf147. [doi:10.1093/bioinformatics/btaf147](https://doi.org/10.1093/bioinformatics/btaf147). [Official project](https://github.com/wwood/CoverM). + +### SAMtools + +Danecek P, Bonfield JK, Liddle J, et al. (2021). *Twelve years of SAMtools and BCFtools*. GigaScience 10(2): giab008. [doi:10.1093/gigascience/giab008](https://doi.org/10.1093/gigascience/giab008). [Official project](https://www.htslib.org/). + +### featureCounts + +Liao Y, Smyth GK, Shi W. (2014). *featureCounts: an efficient general purpose program for assigning sequence reads to genomic features*. Bioinformatics 30(7): 923–930. [doi:10.1093/bioinformatics/btt656](https://doi.org/10.1093/bioinformatics/btt656). [Official project](https://subread.sourceforge.net/). + +### Pathway Tools + +Karp PD, Latendresse M, Paley SM, et al. (2016). *Pathway Tools version 19.0 update: software for pathway/genome informatics and systems biology*. Briefings in Bioinformatics 17(5): 877–890. [doi:10.1093/bib/bbv079](https://doi.org/10.1093/bib/bbv079). [Official project](https://www.pathwaytools.org/). + +### Minimap2 + +Li H. (2018). *Minimap2: pairwise alignment for nucleotide sequences*. Bioinformatics 34(18): 3094–3100. [doi:10.1093/bioinformatics/bty191](https://doi.org/10.1093/bioinformatics/bty191). [Official project](https://github.com/lh3/minimap2). + +### BWA + +Li H, Durbin R. (2009). *Fast and accurate short read alignment with Burrows–Wheeler transform*. Bioinformatics 25(14): 1754–1760. [doi:10.1093/bioinformatics/btp324](https://doi.org/10.1093/bioinformatics/btp324). [Official project](https://github.com/lh3/bwa). + +### Strobealign + +Sahlin K. (2022). *Strobealign: flexible seed size enables ultra-fast and accurate read alignment*. Genome Biology 23(1): 260. [doi:10.1186/s13059-022-02831-7](https://doi.org/10.1186/s13059-022-02831-7). [Official project](https://github.com/ksahlin/strobealign). + +### Apptainer / Singularity + +Kurtzer GM, Sochat V, Bauer MW. (2017). *Singularity: Scientific containers for mobility of compute*. PLOS ONE 12(5): e0177459. [doi:10.1371/journal.pone.0177459](https://doi.org/10.1371/journal.pone.0177459). [Official project](https://github.com/apptainer/apptainer#citing-apptainer). + +### Slurm + +Yoo AB, Jette MA, Grondona M. (2003). *SLURM: Simple Linux Utility for Resource Management*. In *Job Scheduling Strategies for Parallel Processing*. Lecture Notes in Computer Science 2862: 44–60. [doi:10.1007/10968987_3](https://doi.org/10.1007/10968987_3). [Official project](https://slurm.schedmd.com/). + +### Bioconda + +The Bioconda Team, Grüning B, Dale R, et al. (2018). *Bioconda: sustainable and comprehensive software distribution for the life sciences*. Nature Methods 15(7): 475–476. [doi:10.1038/s41592-018-0046-7](https://doi.org/10.1038/s41592-018-0046-7). [Official project](https://bioconda.github.io/). + +### pandas + +McKinney W. (2010). *Data Structures for Statistical Computing in Python*. Proceedings of the Python in Science Conference: 56–61. [doi:10.25080/majora-92bf1922-00a](https://doi.org/10.25080/majora-92bf1922-00a). [Official project](https://pandas.pydata.org/). + +### NumPy + +Harris CR, Millman KJ, van der Walt SJ, et al. (2020). *Array programming with NumPy*. Nature 585(7825): 357–362. [doi:10.1038/s41586-020-2649-2](https://doi.org/10.1038/s41586-020-2649-2). [Official project](https://numpy.org/). + +### SciPy + +Virtanen P, Gommers R, Oliphant TE, et al. (2020). *SciPy 1.0: fundamental algorithms for scientific computing in Python*. Nature Methods 17(3): 261–272. [doi:10.1038/s41592-019-0686-2](https://doi.org/10.1038/s41592-019-0686-2). [Official project](https://scipy.org/). + +### pyfastx + +Du L, Liu Q, Fan Z, et al. (2021). *Pyfastx: a robust Python package for fast random access to sequences from plain and gzipped FASTA/Q files*. Briefings in Bioinformatics 22(4): bbaa368. [doi:10.1093/bib/bbaa368](https://doi.org/10.1093/bib/bbaa368). [Official project](https://github.com/lmdu/pyfastx). + +### Cython + +Behnel S, Bradshaw R, Citro C, et al. (2011). *Cython: The Best of Both Worlds*. Computing in Science & Engineering 13(2): 31–39. [doi:10.1109/mcse.2010.118](https://doi.org/10.1109/mcse.2010.118). [Official project](https://cython.org/). + +### UniProt + +The UniProt Consortium, Bateman A, Martin MJ, et al. (2023). *UniProt: the Universal Protein Knowledgebase in 2023*. Nucleic Acids Research 51(D1): D523–D531. [doi:10.1093/nar/gkac1052](https://doi.org/10.1093/nar/gkac1052). [Official project](https://www.uniprot.org/). + +### UniRef + +Suzek BE, Huang H, McGarvey P, et al. (2007). *UniRef: comprehensive and non-redundant UniProt reference clusters*. Bioinformatics 23(10): 1282–1288. [doi:10.1093/bioinformatics/btm098](https://doi.org/10.1093/bioinformatics/btm098). [Official project](https://www.uniprot.org/uniref). + +### CAZy + +Drula E, Garron ML, Dogan S, et al. (2022). *The carbohydrate-active enzyme database: functions and literature*. Nucleic Acids Research 50(D1): D571–D577. [doi:10.1093/nar/gkab1045](https://doi.org/10.1093/nar/gkab1045). [Official project](https://www.cazy.org/). + +### eggNOG + +Huerta-Cepas J, Szklarczyk D, Heller D, et al. (2019). *eggNOG 5.0: a hierarchical, functionally and phylogenetically annotated orthology resource based on 5090 organisms and 2502 viruses*. Nucleic Acids Research 47(D1): D309–D314. [doi:10.1093/nar/gky1085](https://doi.org/10.1093/nar/gky1085). [Official project](http://eggnog.embl.de/). + +### SILVA + +Quast C, Pruesse E, Yilmaz P, et al. (2013). *The SILVA ribosomal RNA gene database project: improved data processing and web-based tools*. Nucleic Acids Research 41(D1): D590–D596. [doi:10.1093/nar/gks1219](https://doi.org/10.1093/nar/gks1219). [Official project](https://www.arb-silva.de/). + +### NCBI Taxonomy + +Schoch CL, Ciufo S, Domrachev M, et al. (2020). *NCBI Taxonomy: a comprehensive update on curation, resources and tools*. Database 2020: baaa062. [doi:10.1093/database/baaa062](https://doi.org/10.1093/database/baaa062). [Official project](https://www.ncbi.nlm.nih.gov/taxonomy). + +### ENZYME + +Bairoch A. (2000). *The ENZYME database in 2000*. Nucleic Acids Research 28(1): 304–305. [doi:10.1093/nar/28.1.304](https://doi.org/10.1093/nar/28.1.304). [Official project](https://enzyme.expasy.org/). + +### MetaCyc + +Caspi R, Billington R, Keseler IM, et al. (2020). *The MetaCyc database of metabolic pathways and enzymes - a 2019 update*. Nucleic Acids Research 48(D1): D445–D453. [doi:10.1093/nar/gkz862](https://doi.org/10.1093/nar/gkz862). [Official project](https://metacyc.org/). diff --git a/docs/diagram-requirements.txt b/docs/diagram-requirements.txt new file mode 100644 index 0000000..984b515 --- /dev/null +++ b/docs/diagram-requirements.txt @@ -0,0 +1 @@ +playwright==1.63.0 diff --git a/docs/diagrams/architecture.mmd b/docs/diagrams/architecture.mmd new file mode 100644 index 0000000..81f60a0 --- /dev/null +++ b/docs/diagrams/architecture.mmd @@ -0,0 +1,25 @@ +%%{init: {"theme":"base","fontFamily":"Times New Roman, Times, serif","themeVariables":{"fontFamily":"Times New Roman, Times, serif","fontSize":"16px","primaryColor":"#CCCCCC","primaryTextColor":"#111111","primaryBorderColor":"#666666","secondaryColor":"#DAE8FC","tertiaryColor":"#F5F5F5","lineColor":"#808080","edgeLabelBackground":"#FFFFFF","background":"#FFFFFF","defaultLinkColor":"#808080","textColor":"#111111"},"flowchart":{"htmlLabels":false,"curve":"linear"}}}%% +flowchart TB + U[User command and manifest] --> MP["MP controller
Validate and plan"] + MP --> NF["Nextflow
Dependencies and task scheduling"] + NF --> L["Local executor
Server CPU and memory budget"] + NF --> S["Slurm executor
Submission and queue limits"] + L --> LW[Local worker processes] + S --> HW[Compute-node jobs] + LW --> T[Annotation and mapping tools] + HW --> T + LW --> P{"Optional Pathway Tools
Private SIF instances"} + HW --> P + T --> O[Output files, logs and task receipts] + P --> O + O --> R[MP report and explorer] + classDef module fill:#CCCCCC,stroke:#111111,stroke-width:1.5px,color:#111111; + classDef compute fill:#F5F5F5,stroke:#666666,stroke-width:2px,color:#111111; + classDef input fill:#DAE8FC,stroke:#6C8EBF,stroke-width:2px,color:#111111; + classDef output fill:#D5E8D4,stroke:#82B366,stroke-width:2px,color:#111111; + classDef data fill:#FFFFFF,stroke:#666666,stroke-width:1.5px,color:#111111; + class U input; + class MP,NF,L,S,LW,HW module; + class T,P compute; + class O data; + class R output; diff --git a/docs/diagrams/data-flow.mmd b/docs/diagrams/data-flow.mmd new file mode 100644 index 0000000..dfe9930 --- /dev/null +++ b/docs/diagrams/data-flow.mmd @@ -0,0 +1,33 @@ +%%{init: {"theme":"base","fontFamily":"Times New Roman, Times, serif","themeVariables":{"fontFamily":"Times New Roman, Times, serif","fontSize":"16px","primaryColor":"#CCCCCC","primaryTextColor":"#111111","primaryBorderColor":"#666666","secondaryColor":"#DAE8FC","tertiaryColor":"#F5F5F5","lineColor":"#808080","edgeLabelBackground":"#FFFFFF","background":"#FFFFFF","defaultLinkColor":"#808080","textColor":"#111111"},"flowchart":{"htmlLabels":false,"curve":"linear"}}}%% +flowchart TB + A[Assembly FASTA] --> QC[Preprocessed contigs] + QC --> F[Predicted CDS and RNA features] + F --> AN[Functional and taxonomic annotations] + DB[MPDB reference sequences and mappings] --> AN + READS[Optional reads] --> MAP[Read mapping and feature counting] + QC --> MAP + F --> MAP + MAP --> AB[Contig and feature abundance] + AN --> PI[Community Pathway Tools inputs] + PI --> SPLIT[Genome-specific inputs] + GM[Optional contig-to-genome map] --> SPLIT + QC --> SEQ[Sequence-backed PGDB staging] + PI --> SEQ + SPLIT --> SEQ + SEQ --> PT[Optional licensed Pathway Tools inference] + PT --> PW[Pathway, reaction and gene tables] + AN --> REPORT[Report tables and relational index] + AB --> REPORT + GM --> REPORT + PW --> REPORT + REPORT --> PORTAL[Explorer searches, subsets and CSV exports] + classDef module fill:#CCCCCC,stroke:#111111,stroke-width:1.5px,color:#111111; + classDef compute fill:#F5F5F5,stroke:#666666,stroke-width:2px,color:#111111; + classDef input fill:#DAE8FC,stroke:#6C8EBF,stroke-width:2px,color:#111111; + classDef output fill:#D5E8D4,stroke:#82B366,stroke-width:2px,color:#111111; + classDef data fill:#FFFFFF,stroke:#666666,stroke-width:1.5px,color:#111111; + class A,DB,READS,GM input; + class QC,F,AN,PI,AB,PW data; + class MAP,SPLIT,SEQ,PT compute; + class REPORT module; + class PORTAL output; diff --git a/docs/diagrams/detailed-nodal-workflow.json b/docs/diagrams/detailed-nodal-workflow.json new file mode 100644 index 0000000..c52a35f --- /dev/null +++ b/docs/diagrams/detailed-nodal-workflow.json @@ -0,0 +1,491 @@ +{ + "title": "MetaPathways", + "controller": "MetaPathways + Nextflow • Detailed nucleotide-assembly workflow", + "execution": "Numbered modules group operations • Dependencies govern execution • Optional branches use additional inputs", + "inputs": [ + "Assemblies, sample configuration and MPDB reference indexes", + "Optional: reads, contig-to-genome map and licensed Pathway Tools SIF" + ], + "rows": [ + { + "title": "Public references", + "nodes": [ + { + "label": "Selected public\nreference databases", + "tool": "", + "kind": "input" + }, + { + "label": "Download and normalize\nreference records", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Build protein / rRNA\nsearch indexes", + "tool": "FAST / BLAST+", + "kind": "compute" + }, + { + "label": "Map identifiers,\ntaxonomy and ECs", + "tool": "MP", + "kind": "compute" + }, + { + "label": "MPDB indexes\nand lookup tables", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Licensed image (optional)", + "nodes": [ + { + "label": "Licensed Pathway\nTools installer", + "tool": "", + "kind": "input" + }, + { + "label": "Build private\nSIF image", + "tool": "Apptainer", + "kind": "compute" + }, + { + "label": "Validate startup\nand sequence search", + "tool": "Pathway Tools", + "kind": "compute" + }, + { + "label": "Prepare matching\nMetaCyc references", + "tool": "MP / Pathway Tools", + "kind": "compute" + }, + { + "label": "Registered SIF\nand MetaCyc indexes", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Plan and schedule", + "nodes": [ + { + "label": "Sample manifest,\nconfiguration and MPDB", + "tool": "", + "kind": "input" + }, + { + "label": "Validate inputs;\nresolve sample IDs", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Schedule dependency\ngraph and resources", + "tool": "Nextflow / Slurm", + "kind": "compute" + }, + { + "label": "Check receipts;\nreuse or run tasks", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Per-sample tasks\nand execution records", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Sequence preparation", + "nodes": [ + { + "label": "Assembly\nFASTA sequences", + "tool": "", + "kind": "input" + }, + { + "label": "Validate format;\nnormalize sequences", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Filter contigs\nby configured criteria", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Map identifiers;\ncalculate sequence stats", + "tool": "MP", + "kind": "compute" + }, + { + "label": "QCed contigs\nand identifier map", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Coding features", + "nodes": [ + { + "label": "QCed nucleotide\ncontigs", + "tool": "", + "kind": "input" + }, + { + "label": "Predict coding\nregions and coordinates", + "tool": "pProdigal", + "kind": "compute" + }, + { + "label": "Derive CDS and\namino-acid sequences", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Filter predicted\nprotein sequences", + "tool": "MP", + "kind": "compute" + }, + { + "label": "CDS coordinates\nand filtered proteins", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Functional search", + "nodes": [ + { + "label": "Filtered proteins\nand MPDB indexes", + "tool": "", + "kind": "input" + }, + { + "label": "Search selected\nprotein references", + "tool": "FAST / BLAST+", + "kind": "compute" + }, + { + "label": "Compute sequence\nreference scores", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Apply thresholds;\nparse hits per database", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Database-specific\nfunctional evidence", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "RNA features", + "nodes": [ + { + "label": "QCed contigs and\nrRNA reference indexes", + "tool": "", + "kind": "input" + }, + { + "label": "Predict rRNA\ncoordinates", + "tool": "barrnap", + "kind": "compute" + }, + { + "label": "Search rRNA\nreference sequences", + "tool": "BLAST+", + "kind": "compute" + }, + { + "label": "Predict tRNA\ncoordinates and types", + "tool": "ptRNAscan-SE", + "kind": "compute" + }, + { + "label": "rRNA / tRNA features\nand reference hits", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Feature integration", + "nodes": [ + { + "label": "Protein evidence\nand RNA features", + "tool": "", + "kind": "input" + }, + { + "label": "Combine functional\nand RNA annotations", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Resolve overlapping\nsequence features", + "tool": "pybedtools / BEDTools", + "kind": "compute" + }, + { + "label": "Integrated features\nand annotated GFF", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Annotation tables", + "nodes": [ + { + "label": "Integrated features\nand reference taxonomy", + "tool": "", + "kind": "input" + }, + { + "label": "Map functional\nidentifiers and names", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Assign per-database\nhit taxonomy / LCA", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Write annotation\nand reference tables", + "tool": "MP / pandas", + "kind": "compute" + }, + { + "label": "Functional / taxonomic\nannotation tables", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Annotated exports", + "nodes": [ + { + "label": "Contigs, features\nand annotation tables", + "tool": "", + "kind": "input" + }, + { + "label": "Export annotated\nsequence records", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Prepare PF features\nand coordinate maps", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Build EC / reaction\nand ORF mappings", + "tool": "MP", + "kind": "compute" + }, + { + "label": "GenBank and\nPathway Tools inputs", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Read abundance (optional)", + "nodes": [ + { + "label": "Reads, contigs\nand annotated features", + "tool": "", + "kind": "input" + }, + { + "label": "Map reads;\nmeasure contig coverage", + "tool": "CoverM", + "kind": "compute" + }, + { + "label": "Name-sort the\nexact mapping BAM", + "tool": "SAMtools", + "kind": "compute" + }, + { + "label": "Count features;\nnormalize abundances", + "tool": "featureCounts / MP", + "kind": "compute" + }, + { + "label": "Contig coverage, counts\nand RPKM / TPM", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Community / MAG inputs (optional)", + "lanes": [ + [ + { + "label": "Community features\nand contig sequences", + "tool": "", + "kind": "input" + }, + { + "label": "Stage sequence-backed\ncommunity inputs", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Community\nentity input files", + "tool": "", + "kind": "data" + } + ], + [ + { + "label": "Existing annotations\nand contig-to-MAG map", + "tool": "", + "kind": "input" + }, + { + "label": "Partition existing\nfeatures by genome", + "tool": "MAGSplitter", + "kind": "compute" + }, + { + "label": "Stage genome sequences\nand coordinate maps", + "tool": "MP", + "kind": "compute" + } + ] + ], + "result": "Prepared community\nand genome entities" + }, + { + "title": "Build PGDBs (optional)", + "nodes": [ + { + "label": "Prepared entities\nand licensed SIF", + "tool": "", + "kind": "input" + }, + { + "label": "Start private\ninstance per entity", + "tool": "Pathway Tools", + "kind": "compute" + }, + { + "label": "Infer metabolic\npathways", + "tool": "PathoLogic", + "kind": "compute" + }, + { + "label": "Save / export\npathway databases", + "tool": "Pathway Tools", + "kind": "compute" + }, + { + "label": "Community / genome\nPGDB archives", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Pathway extraction (optional)", + "nodes": [ + { + "label": "Exported PGDB\nflat files", + "tool": "", + "kind": "input" + }, + { + "label": "Extract pathways\nand gene associations", + "tool": "Camelot / MP", + "kind": "compute" + }, + { + "label": "Preserve feature IDs;\nwrite linked tables", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Pathway and\npathway-to-ORF tables", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Results integration", + "nodes": [ + { + "label": "Available annotation,\nabundance and\npathway tables", + "tool": "", + "kind": "input" + }, + { + "label": "Join scoped sample,\nfeature and genome IDs", + "tool": "MP / SQLite", + "kind": "compute" + }, + { + "label": "Record task outcomes\nand resource summaries", + "tool": "MP / Nextflow", + "kind": "compute" + }, + { + "label": "Build run report\nand results index", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Run report and\nSQLite results index", + "tool": "", + "kind": "output" + } + ] + }, + { + "title": "Explore and export", + "nodes": [ + { + "label": "Results index\nand linked tables", + "tool": "", + "kind": "input" + }, + { + "label": "Browse samples, features\nand pathways", + "tool": "MP", + "kind": "compute" + }, + { + "label": "Search, filter\nand select records", + "tool": "MP / JavaScript", + "kind": "compute" + }, + { + "label": "Export selected\nresult tables", + "tool": "MP / SQLite", + "kind": "compute" + }, + { + "label": "EDA portal\nand selected CSVs", + "tool": "", + "kind": "output" + } + ] + } + ] +} diff --git a/docs/diagrams/detailed-workflow-1.mmd b/docs/diagrams/detailed-workflow-1.mmd new file mode 100644 index 0000000..7971a7e --- /dev/null +++ b/docs/diagrams/detailed-workflow-1.mmd @@ -0,0 +1,22 @@ +%%{init: {"theme":"base","fontFamily":"Times New Roman, Times, serif","themeVariables":{"fontFamily":"Times New Roman, Times, serif","fontSize":"16px","primaryColor":"#CCCCCC","primaryTextColor":"#111111","primaryBorderColor":"#666666","secondaryColor":"#DAE8FC","tertiaryColor":"#F5F5F5","lineColor":"#808080","edgeLabelBackground":"#FFFFFF","background":"#FFFFFF","defaultLinkColor":"#808080","textColor":"#111111"},"flowchart":{"htmlLabels":false,"curve":"linear"}}}%% +flowchart TB + PUBLIC["Selected public references
Proteins, SILVA, taxonomy and EC data"] --> BUILD["build_db through Nextflow
Download and prepare reference files"] + BUILD --> INDEX["fastdb or makeblastdb
Protein indexes and BLAST nucleotide indexes"] + BUILD --> TABLES["MP reference preparation
Identifiers, taxonomy and EC mappings"] + INDEX --> MPDB["MPDB
Searchable references and supporting tables"] + TABLES --> MPDB + INSTALLER["User-supplied licensed installer"] --> PTBUILD["build_pt through Nextflow
Apptainer build and official SRI patches"] + PTBUILD --> CHECK["Validate Pathway Tools startup
and BLAST database creation/search"] + CHECK --> SIF["Registered private SIF
Pathway Tools and bundled MetaCyc"] + SIF -. optional matching reference preparation .-> META["MP MetaCyc preparation
protseq.fsa and companion flat files"] + META --> METAINDEX["FAST or BLAST protein index
EC, reaction and pathway mappings"] + METAINDEX --> MPDB + classDef module fill:#CCCCCC,stroke:#111111,stroke-width:1.5px,color:#111111; + classDef compute fill:#F5F5F5,stroke:#666666,stroke-width:2px,color:#111111; + classDef input fill:#DAE8FC,stroke:#6C8EBF,stroke-width:2px,color:#111111; + classDef output fill:#D5E8D4,stroke:#82B366,stroke-width:2px,color:#111111; + classDef data fill:#FFFFFF,stroke:#666666,stroke-width:1.5px,color:#111111; + class PUBLIC,INSTALLER input; + class BUILD,PTBUILD module; + class INDEX,TABLES,CHECK,META,METAINDEX compute; + class MPDB,SIF output; diff --git a/docs/diagrams/detailed-workflow-2.mmd b/docs/diagrams/detailed-workflow-2.mmd new file mode 100644 index 0000000..5374b58 --- /dev/null +++ b/docs/diagrams/detailed-workflow-2.mmd @@ -0,0 +1,30 @@ +%%{init: {"theme":"base","fontFamily":"Times New Roman, Times, serif","themeVariables":{"fontFamily":"Times New Roman, Times, serif","fontSize":"16px","primaryColor":"#CCCCCC","primaryTextColor":"#111111","primaryBorderColor":"#666666","secondaryColor":"#DAE8FC","tertiaryColor":"#F5F5F5","lineColor":"#808080","edgeLabelBackground":"#FFFFFF","background":"#FFFFFF","defaultLinkColor":"#808080","textColor":"#111111"},"flowchart":{"htmlLabels":false,"curve":"linear"}}}%% +flowchart TB + ASM["Assembly FASTA
Original contig identifiers"] --> QC["PREPROCESS_INPUT - MP
Sequence filtering and identifier map"] + QC --> ORF["ORF_PREDICTION
pProdigal wrapping Prodigal"] + ORF --> AA["ORF_TO_AMINO and FILTER_AMINOS - MP
CDS coordinates, nucleotide and protein sequences"] + AA --> SEARCH["FUNC_SEARCH - one task per selected database
FAST fastal or BLAST+ blastp"] + DB["MPDB protein indexes"] -. reference input .-> SEARCH + SEARCH --> SCORE["COMPUTE_REFSCORES - MP
Sequence-based reference scores for hit filtering"] + SCORE --> PARSE["PARSE_FUNC_SEARCH - MP
Per-database thresholds and parsed hits"] + PARSE --> RRNA["SCAN_rRNA - barrnap
rRNA coordinates and sequences"] + QC -. contig sequences .-> RRNA + RRNA --> SILVA["SCAN_rRNA - BLAST+ blastn and MP
Search selected rRNA references and summarize"] + RDB["MPDB rRNA indexes
For example SILVA SSU and LSU"] -. reference input .-> SILVA + SILVA --> TRNA["SCAN_tRNA - ptRNAscan wrapping tRNAscan-SE
tRNA coordinates and identities"] + QC -. contig sequences .-> TRNA + TRNA --> ANN["ANNOTATE_ORFS - MP
Combine protein hits and RNA features
pybedtools / BEDTools overlap processing"] + ANN --> REPORTS["CREATE_ANNOT_REPORTS - MP
Functional assignments and per-database taxonomy / LCA"] + TAX["Reference taxon identifiers
and NCBI taxonomy tree"] -. lookup data .-> REPORTS + REPORTS --> GBK["GENBANK_FILE - MP
Annotated sequence export"] + GBK --> PI["PATHOLOGIC_INPUT - MP
PF features, feature coordinates, ORF map and EC / reaction map"] + PI --> NEXT["Continue to abundance and optional PGDBs"] + classDef module fill:#CCCCCC,stroke:#111111,stroke-width:1.5px,color:#111111; + classDef compute fill:#F5F5F5,stroke:#666666,stroke-width:2px,color:#111111; + classDef input fill:#DAE8FC,stroke:#6C8EBF,stroke-width:2px,color:#111111; + classDef output fill:#D5E8D4,stroke:#82B366,stroke-width:2px,color:#111111; + classDef data fill:#FFFFFF,stroke:#666666,stroke-width:1.5px,color:#111111; + class ASM,DB,RDB,TAX input; + class QC,AA,SCORE,PARSE,ANN,REPORTS,GBK,PI module; + class ORF,SEARCH,RRNA,SILVA,TRNA compute; + class NEXT output; diff --git a/docs/diagrams/detailed-workflow-3.mmd b/docs/diagrams/detailed-workflow-3.mmd new file mode 100644 index 0000000..e361c54 --- /dev/null +++ b/docs/diagrams/detailed-workflow-3.mmd @@ -0,0 +1,30 @@ +%%{init: {"theme":"base","fontFamily":"Times New Roman, Times, serif","themeVariables":{"fontFamily":"Times New Roman, Times, serif","fontSize":"16px","primaryColor":"#CCCCCC","primaryTextColor":"#111111","primaryBorderColor":"#666666","secondaryColor":"#DAE8FC","tertiaryColor":"#F5F5F5","lineColor":"#808080","edgeLabelBackground":"#FFFFFF","background":"#FFFFFF","defaultLinkColor":"#808080","textColor":"#111111"},"flowchart":{"htmlLabels":false,"curve":"linear"}}}%% +flowchart TB + PI["Completed PATHOLOGIC_INPUT
Community features and feature-to-contig relationships"] --> COV["COMPUTE_TPM - CoverM
Reads plus sample contigs
Mapping and contig coverage"] + COV --> BAM{"SAMtools
Name-sort the exact CoverM BAM"} + BAM --> FC{"featureCounts
BAM plus MP-generated GTF
Counts for annotated features"} + FC --> AB["Abundance outputs
Contig coverage and feature counts
MP abund_calc.py normalization"] + COV -. contig statistics .-> AB + PI --> COM["Community PGDB staging - MP
Sample sequences, feature coordinates
and translation tables"] + PI --> SPLIT{"MAGSplitter
Use contig-to-genome map
Partition existing features"} + SPLIT --> MAG["Per-genome PGDB staging - MP
Corresponding contig sequences
and feature coordinates"] + COM --> PT["Pathway Tools / PathoLogic
Licensed SIF with MetaCyc
One private instance per entity
Build and save PGDB"] + MAG --> PT + PT --> EXPORT["Pathway Tools export then Camelot / MP
Read flat files and extract pathway-to-gene associations"] + EXPORT --> PW["PGDB archive
Pathway and pathway-to-ORF tables"] + AB --> DONE["Sample terminal tasks finish
Optional compact-results cleanup"] + PW --> DONE + DONE --> REPORT["MP report controller
Import tables into SQLite with scoped identifiers"] + AN["Annotation, taxonomy and membership tables
Sample statistics and execution records"] -. report sources .-> REPORT + REPORT --> HTML["MP_run_report.html
Run details, outcomes and resource summaries"] + REPORT --> EDA["EDA_portal.html
Samples to features, annotations and pathways
Search, subset and export CSV"] + classDef module fill:#CCCCCC,stroke:#111111,stroke-width:1.5px,color:#111111; + classDef compute fill:#F5F5F5,stroke:#666666,stroke-width:2px,color:#111111; + classDef input fill:#DAE8FC,stroke:#6C8EBF,stroke-width:2px,color:#111111; + classDef output fill:#D5E8D4,stroke:#82B366,stroke-width:2px,color:#111111; + classDef data fill:#FFFFFF,stroke:#666666,stroke-width:1.5px,color:#111111; + class PI input; + class COV,BAM,FC,COM,SPLIT,MAG,PT,EXPORT compute; + class AB,PW,AN data; + class DONE,REPORT module; + class HTML,EDA output; diff --git a/docs/diagrams/figures.json b/docs/diagrams/figures.json new file mode 100644 index 0000000..66be0e9 --- /dev/null +++ b/docs/diagrams/figures.json @@ -0,0 +1,46 @@ +{ + "canvas_width": 2240, + "mermaid_scale": 1.5, + "figures": [ + { + "name": "readme", + "page": "README.md", + "index": 0 + }, + { + "name": "results-schema", + "page": "docs/results-schema.md", + "index": 0 + }, + { + "name": "architecture", + "page": "docs/architecture.md", + "index": 0 + }, + { + "name": "overview", + "page": "docs/overview.md", + "index": 0 + }, + { + "name": "detailed-workflow-1", + "page": "docs/detailed-workflow.md", + "index": 0 + }, + { + "name": "detailed-workflow-2", + "page": "docs/detailed-workflow.md", + "index": 1 + }, + { + "name": "detailed-workflow-3", + "page": "docs/detailed-workflow.md", + "index": 2 + }, + { + "name": "data-flow", + "page": "docs/data-flow.md", + "index": 0 + } + ] +} diff --git a/docs/diagrams/overview.mmd b/docs/diagrams/overview.mmd new file mode 100644 index 0000000..39a5df5 --- /dev/null +++ b/docs/diagrams/overview.mmd @@ -0,0 +1,20 @@ +%%{init: {"theme":"base","fontFamily":"Times New Roman, Times, serif","themeVariables":{"fontFamily":"Times New Roman, Times, serif","fontSize":"16px","primaryColor":"#CCCCCC","primaryTextColor":"#111111","primaryBorderColor":"#666666","secondaryColor":"#DAE8FC","tertiaryColor":"#F5F5F5","lineColor":"#808080","edgeLabelBackground":"#FFFFFF","background":"#FFFFFF","defaultLinkColor":"#808080","textColor":"#111111"},"flowchart":{"htmlLabels":false,"curve":"linear"}}}%% +flowchart LR + A[Assemblies] --> B[Functional and taxonomic annotation] + R[Reads] --> C[Read abundance] + A --> C + B --> D[Community and genome pathways] + G[Genome assignments] --> D + P[Licensed Pathway Tools] --> D + B --> E[Reports and explorer] + C --> E + D --> E + E --> F[Filtered tables and CSV exports] + classDef module fill:#CCCCCC,stroke:#111111,stroke-width:1.5px,color:#111111; + classDef compute fill:#F5F5F5,stroke:#666666,stroke-width:2px,color:#111111; + classDef input fill:#DAE8FC,stroke:#6C8EBF,stroke-width:2px,color:#111111; + classDef output fill:#D5E8D4,stroke:#82B366,stroke-width:2px,color:#111111; + classDef data fill:#FFFFFF,stroke:#666666,stroke-width:1.5px,color:#111111; + class A,R,G,P input; + class B,C,D,E module; + class F output; diff --git a/docs/diagrams/readme.mmd b/docs/diagrams/readme.mmd new file mode 100644 index 0000000..6fc5d39 --- /dev/null +++ b/docs/diagrams/readme.mmd @@ -0,0 +1,20 @@ +%%{init: {"theme":"base","fontFamily":"Times New Roman, Times, serif","themeVariables":{"fontFamily":"Times New Roman, Times, serif","fontSize":"16px","primaryColor":"#CCCCCC","primaryTextColor":"#111111","primaryBorderColor":"#666666","secondaryColor":"#DAE8FC","tertiaryColor":"#F5F5F5","lineColor":"#333333","edgeLabelBackground":"#FFFFFF","background":"#FFFFFF"},"flowchart":{"htmlLabels":false,"curve":"linear"}}}%% +flowchart LR + A[Assemblies] --> B[Functional and taxonomic annotation] + R[Reads] --> C[Read abundance] + A --> C + B --> D[Community and genome pathways] + G[Genome assignments] --> D + P[Licensed Pathway Tools] --> D + B --> E[Reports and explorer] + C --> E + D --> E + E --> F[Filtered tables and CSV exports] + classDef module fill:#CCCCCC,stroke:#111111,stroke-width:1.5px,color:#111111; + classDef compute fill:#F5F5F5,stroke:#666666,stroke-width:2px,color:#111111; + classDef input fill:#DAE8FC,stroke:#6C8EBF,stroke-width:2px,color:#111111; + classDef output fill:#D5E8D4,stroke:#82B366,stroke-width:2px,color:#111111; + classDef data fill:#FFFFFF,stroke:#666666,stroke-width:1.5px,color:#111111; + class A,R,G,P input; + class B,C,D,E module; + class F output; diff --git a/docs/diagrams/results-schema.mmd b/docs/diagrams/results-schema.mmd new file mode 100644 index 0000000..9f600a8 --- /dev/null +++ b/docs/diagrams/results-schema.mmd @@ -0,0 +1,15 @@ +%%{init: {"theme":"base","fontFamily":"Times New Roman, Times, serif","themeVariables":{"fontFamily":"Times New Roman, Times, serif","fontSize":"16px","primaryColor":"#CCCCCC","primaryTextColor":"#111111","primaryBorderColor":"#666666","secondaryColor":"#DAE8FC","tertiaryColor":"#F5F5F5","lineColor":"#808080","edgeLabelBackground":"#FFFFFF","background":"#FFFFFF","defaultLinkColor":"#808080","textColor":"#111111"},"flowchart":{"htmlLabels":false,"curve":"linear"},"themeCSS":".edgeLabel .background { fill: #FFFFFF !important; opacity: 1 !important; } .edgeLabel text { fill: #111111 !important; }"}}%% +erDiagram + samples ||--o{ contigs : contains + contigs ||--o{ orfs : contains + orfs ||--o{ annotations : has + annotations ||--o{ annotation_terms : describes + samples ||--o{ entities : contains + entities ||--o{ contig_mags : assigns + contigs ||--o{ contig_mags : belongs + entities ||--o{ entity_orfs : supplies + orfs ||--o{ entity_orfs : supplies + entities ||--o{ pathways : infers + pathways ||--o{ pathway_orfs : supports + orfs ||--o{ pathway_orfs : participates + orfs ||--o{ orf_groups : groups diff --git a/docs/documentation.md b/docs/documentation.md new file mode 100644 index 0000000..57b80b9 --- /dev/null +++ b/docs/documentation.md @@ -0,0 +1,48 @@ +# Maintaining the documentation + +The Markdown files in `docs/` are the source for Read the Docs. Keep the README limited to the original abstract, installation/test essentials, the conceptual diagram and links to the complete guide. Edit detailed instructions here; do not maintain separate copies on the website. + +## Build and review locally + +Use a separate documentation environment: + +```bash +mamba create -n metapathways-docs -c conda-forge python=3.11 pip +conda activate metapathways-docs +python -m pip install -r docs/requirements.txt +python -m sphinx -W --keep-going -b html docs docs/_build/html +python -m http.server 8000 --directory docs/_build/html +``` + +Open `http://localhost:8000`. The documentation build does not install MP, reference databases, or Pathway Tools. It uses the committed CLI reference; after changing CLI arguments, regenerate that page from an MP development environment with `python scripts/generate_cli_docs.py`. + +Mermaid sources live in `docs/diagrams/`; GitHub and Read the Docs display the committed SVG previews at a shared scale. Keep the conceptual diagram in the README consistent with `overview.md`. The GitHub documentation job builds with warnings treated as errors. + +## Connect Read the Docs + +In the existing **metapathways** project, connect `hallamlab/MetaPathways`. The configuration file is `.readthedocs.yaml`, the Sphinx configuration is `docs/conf.py`, and dependencies are in `docs/requirements.txt`. + +Activate the branch you want to preview under **Versions** and trigger a build. Keep the public default on the intended main/development branch; switch it to the updated branch after the documentation is merged. Enable pull-request previews if desired. Branch previews allow review before changing the site's default version. + +The project address stays **https://hallamlab-metapathways.readthedocs.io/**. Repository changes supply the build configuration; linking the project, version activation and default-version settings are managed in Read the Docs. Confirm a successful build there before announcing updated hosted pages. + +The previous `quick_start.html`, `install.html` and `usage.html` addresses redirect to the corresponding new pages. + +## Shared diagram scale + +ASPIRE and MetaPathways use a shared 2,240-unit-wide white canvas for documentation previews. MP’s main workflow in the README and documentation home page uses its cropped SVG at full available width for readability. Smaller diagrams are centered without stretching; Mermaid diagrams use a common 1.5× scale to bring their 16-pixel labels close to the publication figures’ typography. Preview width is responsive, but relative scale stays consistent across pages. Click a diagram to open its SVG for closer inspection. Original publication SVG/PDF downloads stay tightly cropped. + +Edit Mermaid sources under `docs/diagrams/`, or the original publication SVGs under `docs/assets/`. To rebuild the committed previews from the repository root, in the documentation environment: + +```bash +python -m pip install -r docs/diagram-requirements.txt +python -m playwright install chromium +python scripts/render_workflow_diagrams.py +``` + +Chromium requires its usual Linux system libraries. Rendering downloads the pinned Mermaid bundle; ordinary Sphinx builds need neither Chromium nor network access for diagrams. `docs/diagrams/figures.json` records the source mapping and shared canvas size. If a future diagram needs a wider canvas, update both projects together. Do not hand-edit generated files in `docs/assets/diagrams/`. + +The expanded nodal figure at the top of `detailed-workflow.md` is generated from +`docs/diagrams/detailed-nodal-workflow.json` by `scripts/render_detailed_workflow.py`. +The standard rebuild command above regenerates its SVG, PDF and shared-scale +preview alongside the three supporting diagrams. diff --git a/docs/execution.md b/docs/execution.md new file mode 100644 index 0000000..a2a6a7b --- /dev/null +++ b/docs/execution.md @@ -0,0 +1,34 @@ +# Logs, temporary files and restarting + +Every command with an output location saves its CLI transcript under `OUTPUT/logs/cli/`. Workflow invocations also retain: + +```text +OUTPUT/logs/COMMAND/RUN_ID/ +├── console.log # Controller and labeled tool output +├── tasks.json # Commands, dependencies, resources and paths +├── tasks/ # Individual output logs and status records +├── summary.json # Invocation outcome and per-task results +├── nextflow.log +├── nextflow_tasks/ # Archived .command.* diagnostics +├── trace.tsv +├── report.html +├── timeline.html +├── main.nf +└── nextflow.config +``` + +Console output is displayed and saved. Reports link the retained Nextflow records. Original sample logs remain available as well. Historical workflows without trace files cannot retrospectively gain CPU/RAM measurements by generating a report. + +Pathway Tools containers place their private `/tmp` and `/var/tmp` on the disk +backing the task state directory, rather than Apptainer's limited in-memory +session filesystem. In compact mode this follows the selected task scratch +location (normally `SLURM_TMPDIR` on Slurm). Each invocation remains isolated; +this controller setting does not require rebuilding an existing SIF. Shared +JSON receipts and image-digest caches use unique temporary files and atomic +replacement, so concurrent publishers do not share a temporary filename. + +Default Nextflow work, session cache and Conda package/environment caches live under `OUTPUT/.metapathways/COMMAND/tmp/RUN_ID`. After a completed invocation, including an ordinary task failure, diagnostics are archived and these temporary directories are removed. Interrupted invocations retain them because cancellation may still be in progress. + +Use `--keep_work` to retain the defaults. Explicit `--work_dir` and `--conda_cache` paths are retained automatically; MP does not delete a user-provided shared cache. Nextflow's installed runtime and MP's Conda environment are not per-run caches and are retained. + +Durable receipts remain under `OUTPUT/.metapathways/COMMAND/receipts`. Repeating a command checks tracked inputs, outputs and command signatures, permitting reuse even after temporary cleanup. Never remove final outputs expecting receipts alone to recover them. To force one annotation stage, use its `redo` flag; `--force_redo` reruns all annotation stages. Old read-mapping output without a current receipt is deliberately recomputed once. diff --git a/docs/getting-started.md b/docs/getting-started.md new file mode 100644 index 0000000..abe457d --- /dev/null +++ b/docs/getting-started.md @@ -0,0 +1,84 @@ +# Getting started: from a terminal to your first result + +[Home and reading order](index.md) · Next: [test walkthrough](test.md) + +## What you will do + +Install MP on a Linux computer, activate its software environment, run a small included example, and open a report. You do not need a Pathway Tools license for the first example. Add pathway inference after the example works. + +If you already have a working MP installation, go directly to the [test walkthrough](test.md). If your lab manages the software centrally, ask for the activation command and MPDB path rather than installing a second copy. + +## Before typing commands + +A **terminal** accepts commands. A **working directory** is the folder relative paths refer to. `pwd` displays it; `ls` lists its contents; `cd DIRECTORY` changes it. `~` means your home directory. `/data/project/sample.fasta` is an absolute path; `sample.fasta` means a file in the working directory. Linux distinguishes `SampleA` from `samplea`. + +Copy commands inside the code blocks, without Markdown backticks or a shell prompt. A backslash at the end of a line continues the command on the next line: do not put spaces after it. Values such as `/path/to/MPDB` are placeholders; replace them with real paths. Use simple project and sample names without spaces or shell punctuation, because some legacy tool wrappers construct shell commands. + +An **environment** is a collection of compatible software. Activating it makes your terminal find the intended MP and tool executables. Activate it again in each new terminal. It does not require setting an MP-specific environment variable. + +On a remote server, the analysis commands run in the SSH terminal on that server. Your web browser normally runs on your own computer. The [SSH report instructions](reports-tutorial.md#view-a-remote-report-through-ssh) connect the two. For a long run, use your site's persistent terminal/session practice so disconnecting SSH does not interrupt your controller. + +## Install on Linux x86-64 + +Choose one route, in this order: + +1. **[Conda package with Mamba](installation.md#1-conda-package-with-mamba-preferred)** — preferred for local servers and HPC. Install the package and activate the environment before each session. Workflow helpers are included. +2. **[Quay Docker or Apptainer](containers.md)** — use a versioned image and the container-specific three-sample instructions. References and outputs must be writable; the public MP image does not contain licensed Pathway Tools. +3. **[Local installation from GitHub](installation.md#3-local-installation-from-github)** — build the supporting Mamba environment, then install MP with pip from your checkout. Use this route to install from a local checkout or make changes to MP. + +Windows and macOS users can connect to a Linux server. These instructions do not claim a tested native Windows, Apple Silicon, or WSL installation. If Mamba is not installed, follow the official [Miniforge instructions](https://github.com/conda-forge/miniforge) or your HPC's software setup instructions. No MP-specific environment variable is required. + +The source installation makes a fixed copy of the checked-out code. After updating the checkout, reinstall with `python -m pip install .`. Record `git rev-parse HEAD` alongside `metapathways version`. Developers may explicitly choose an editable installation with `-e .`. Do not update an installation while it is running an analysis. + +Confirm the installation: + +```bash +which python +which metapathways +metapathways version +metapathways analysis_wf --help +nextflow -version +java -version +apptainer --version +magsplitter --help +``` + +`which` should point into your activated environment, or to an intentional site-managed executable. `analysis_wf --help` should list manifest and resource options. A help command does not run an analysis. If a command is missing, first confirm you activated the correct environment. On hosts that restrict user namespaces, Apptainer setup may require an administrator; see [Pathway Tools build prerequisites](pathway-tools.md#host-prerequisites). + +## First installation check + +Follow the **[three-sample test walkthrough](test.md)**. It uses the same `analysis_wf` command as a real analysis, with small bundled references, assemblies, paired reads and genome maps. It exercises annotation, read abundance, genome splitting and the reports without requiring Pathway Tools. Database preparation downloads enzyme and taxonomy support records, so internet access is required. + +The older `run --test` K12 example remains available for compatibility, but it does not exercise the complete workflow and is not the complete workflow test. + +## Your first real assembly + +The tiny test references are only for the example. Build or obtain a production MPDB before interpreting your own data: + +```bash +metapathways build_db -d ~/MPDB --func swissprot -a fast +metapathways run -i /path/to/sample.fasta -o ~/mp-results -d ~/MPDB \ + --threads 8 --max_cpus 8 --max_memory '32 GB' +``` + +The assembly goes through quality control, gene/RNA prediction, reference searches, and annotation. Without reads, there is no measured read abundance. Without a separate `ptools` step, there are no inferred PGDB pathways. A single `analysis_wf` command can combine these steps once the inputs and licensed container are ready. + +Continue with the [test walkthrough](test.md), [command cookbook](commands.md), and [input organization](inputs.md). + +## Terms used in the guides + +| Term | Meaning in MP | +| --- | --- | +| Assembly / contig | DNA sequences assembled from reads; each FASTA record is a contig | +| FASTA / FASTQ | Sequence file / read file with per-base quality scores | +| Paired / interleaved | Two mate files / one file with alternating mate records | +| ORF | Predicted protein-coding region; RNA features are also retained in relevant outputs | +| MAG / genome bin | A group of contigs assigned to a genome; MP consumes assignments, not a binning algorithm | +| MPDB | MP's reference sequences, indexes, taxonomy and functional mapping tables | +| MetaCyc | Reference pathways/reactions and associated proteins; not a sample's pathway predictions | +| PGDB | Pathway/Genome Database inferred for one community or genome bin | +| SIF | An Apptainer image containing the licensed Pathway Tools installation | +| Nextflow / task | Scheduler behind MP / one scheduled unit of work | +| Manifest | A TSV listing sample IDs and exact input paths | +| Receipt | MP's record used to decide whether a completed task can be reused | +| TSV / CSV | Table with tab-separated / comma-separated fields | diff --git a/docs/includes/cami-references.md b/docs/includes/cami-references.md new file mode 100644 index 0000000..b978d33 --- /dev/null +++ b/docs/includes/cami-references.md @@ -0,0 +1,5 @@ +## CAMI references + +- **CAMI:** Sczyrba, A., Hofmann, P., Belmann, P., et al. (2017). *Critical Assessment of Metagenome Interpretation—a benchmark of metagenomics software*. **Nature Methods 14**(11), 1063–1071. [DOI: 10.1038/nmeth.4458](https://doi.org/10.1038/nmeth.4458). [CAMI project website](https://cami-challenge.org/). +- **CAMI II:** Meyer, F., Fritz, A., Deng, Z.-L., et al. (2022). *Critical Assessment of Metagenome Interpretation: the second round of challenges*. **Nature Methods 19**(4), 429–440. [DOI: 10.1038/s41592-022-01431-4](https://doi.org/10.1038/s41592-022-01431-4). [CAMI project website](https://cami-challenge.org/). +- **Source dataset for the MP test subset:** CAMI II multi-sample human microbiome dataset. [Dataset DOI: 10.4126/FRL01-006425518](https://doi.org/10.4126/FRL01-006425518). The bundled inputs are selected and cropped subsets of this collection; their exact transformations and file hashes are recorded in the bundle provenance. diff --git a/docs/index.md b/docs/index.md new file mode 100644 index 0000000..1cca57a --- /dev/null +++ b/docs/index.md @@ -0,0 +1,76 @@ +# MetaPathways + +Annotate metagenomes, measure read abundance, infer community and genome pathways, and explore the results. Run on a local server or submit work through Slurm using the same commands. + +**New users:** start with [installation and the three-sample test](installation.md). The small input dataset and workflow helpers are included. **Want PGDBs? Follow the [Pathway Tools installation guide](pathway-tools.md) before running the complete workflow.** + +```{container} mp-primary-workflow +[![MetaPathways workflow: six conceptual modules for preprocessing, feature prediction, annotation, optional pathways and read abundance, and integrated reports and explorer, orchestrated by Nextflow locally or on Slurm.](assets/workflow-main.svg)](assets/workflow-main.svg) +``` + +Arrows between numbered modules trace the conceptual flow of results; independent tasks and optional branches follow the dependencies in the detailed workflow. Optional reads add abundance; genome maps add genome-specific analysis. The report and explorer connect available results for searching, subsetting and CSV export. [View the SVG](assets/workflow-main.svg) · [Detailed nodal workflow and citations](detailed-workflow.md). + +## Start simple + +```{toctree} +:maxdepth: 1 +:caption: Getting started + +overview +installation +getting-started +containers +test +test-bundle +cami-references +``` + +## Run your analysis + +```{toctree} +:maxdepth: 1 +:caption: Analysis guides + +inputs +databases +pathway-tools +analysis +annotation +pgdb-workflow +commands +resources +execution +troubleshooting +``` + +## Understand and explore + +```{toctree} +:maxdepth: 1 +:caption: Results and reference + +reports-tutorial +reports-reference +results-schema +architecture +data-flow +detailed-workflow +workflow +cli-reference +benchmarking +reproducibility +``` + +## Maintain and contribute + +```{toctree} +:maxdepth: 1 +:caption: Development + +pr-testing +releasing +release-readiness +documentation +``` + +Source code and documentation: [GitHub](https://github.com/hallamlab/MetaPathways). Questions, bugs and feature requests: [GitHub issues](https://github.com/hallamlab/MetaPathways/issues). See the [README](https://github.com/hallamlab/MetaPathways#team-support-and-citation) for contributors and citation. diff --git a/docs/inputs.md b/docs/inputs.md new file mode 100644 index 0000000..30ca6b8 --- /dev/null +++ b/docs/inputs.md @@ -0,0 +1,107 @@ +# Organize assemblies, reads and genome assignments + +[Home](index.md) · [Command cookbook](commands.md) · [Manifest reference](analysis.md#custom-analysis-manifest) + +**Want PGDBs? Complete the [Pathway Tools installation guide](pathway-tools.md) first**, including building and registering your licensed SIF with `metapathways build_pt`. Then prepare your inputs and run `analysis_wf`. To run without pathway inference, add `--skip_ptools`. + +## Decide what each sample means + +A sample is one assembly with its own reads and optional contig-to-genome assignments. The assembly is required. Reads add abundance measurements. A genome map adds genome-level splitting and PGDBs. MP does not assemble reads or infer bins itself. + +Choose sample IDs that begin with a letter and contain only letters, digits and underscores, for example `Oral_15`. Keep them stable between reruns. IDs such as `sample`, `Sample` and `SAMPLE` are distinct. Do not use reserved directory names such as `reports`, `logs`, `inputs`, `assemblies`, `reads` or `mag_maps`. + +`run -i ASSEMBLY_DIRECTORY` annotates multiple files, but it is not the automatic per-sample read/map matcher. Use `analysis_wf` for that complete workflow. It requires nucleotide FASTA; standalone `run` also has a protein-input mode. + +## Automatic discovery + +```text +my_inputs/ +├── assemblies/ +│ ├── SampleA.fasta +│ └── SampleB.fasta.gz +├── reads/ +│ ├── SampleA_R1.fastq.gz +│ ├── SampleA_R2.fastq.gz +│ └── SampleB_interleaved.fastq.gz +└── mag_maps/ + ├── SampleA.tsv + └── SampleB.tsv +``` + +Use `.fa`, `.fna` or `.fasta` for assemblies, optionally `.gz`. For reads use `.fq` or `.fastq`, optionally `.gz`. The suffixes `_R1`, `_R2`, `_interleaved` and `_single` specify the read layout. Matching is by sample ID, not file ordering or fuzzy similarity. Maps end in `.tsv`. + +```bash +metapathways analysis_wf -i my_inputs -o analysis -d /path/to/MPDB \ + --annotation_dbs swissprot metacyc \ + --taxprune --taxonomic_scope all \ + --threads 8 --max_cpus 32 --max_memory '64 GB' +``` + +Every sample needs one unambiguous layout and one map by default. MP stops on missing, unmatched or ambiguous inputs instead of guessing. Do not place both paired files and an interleaved file for the same sample in this discovery directory. Use `--no_reads` or `--no_mags` to omit that input type for the entire discovered dataset. `--no_mags` still allows community pathway inference; `--skip_ptools` skips all PGDB construction. + +If your directories are already separate: + +```bash +metapathways analysis_wf \ + -i /project/assemblies --reads_dir /project/reads --mag_maps_dir /project/maps \ + -o /project/analysis -d /project/MPDB \ + --taxprune --taxonomic_scope all +``` + +## Organize with links, without copying large reads + +A symbolic link gives an existing file another path without duplicating its data. Use absolute targets so moving the working directory does not change what they point to: + +```bash +mkdir -p my_inputs/assemblies my_inputs/reads my_inputs/mag_maps +ln -s /data/original/anonymous_gsa.fasta my_inputs/assemblies/SampleA.fasta +ln -s /data/original/anonymous_reads.fq.gz my_inputs/reads/SampleA_interleaved.fastq.gz +ln -s /data/original/contig_to_genome.tsv my_inputs/mag_maps/SampleA.tsv +``` + +The original files must remain readable throughout the workflow. The same assembly or read file cannot be reused under multiple manifest samples via hard/symbolic aliases; this catches accidental duplicated inputs. Independent test fixture copies are intentional and documented separately. + +## Custom manifest: keep every file where it is + +Use a tab-separated file with this exact header: + +```text +sample_id assembly read_layout reads_1 reads_2 mag_map +``` + +The columns are real tabs, not the literal characters `\t`. A spreadsheet can save “tab-delimited text”; do not save an Excel workbook and rename it `.tsv`. Blank cells must remain blank, not `None` or `NA`. + +| Column | What to put in it | +| --- | --- | +| `sample_id` | Your chosen stable sample name; independent of assembly basename | +| `assembly` | Path to that sample's nucleotide FASTA | +| `read_layout` | `paired`, `interleaved`, `single` or `none` | +| `reads_1` | R1, the interleaved file, or the single-end file; blank for `none` | +| `reads_2` | R2 only for `paired`; blank otherwise | +| `mag_map` | Headerless contig-to-bin TSV, or blank to omit bins for that sample | + +Relative paths are resolved against the manifest's directory. Absolute paths make files easier to locate across terminal sessions; on a cluster, they must also work on compute nodes. Different samples can use different read layouts and omit different optional inputs through the manifest. + +```bash +metapathways analysis_wf --manifest /project/samples.tsv \ + -o /project/analysis -d /project/MPDB \ + --annotation_dbs swissprot metacyc \ + --taxprune --taxonomic_scope all \ + --threads 8 --max_cpus 32 --max_memory '64 GB' +``` + +Do not combine `--manifest` with discovery flags, `-1`, `-2` or `--interleaved`. The manifest carries those choices per sample. The resolved manifest and planned entities are saved in the output root as `inputs.resolved.tsv` and `inputs.entities.json`. + +## Contig-to-genome map format + +MAGSplitter needs exactly two tab-separated columns and **no header**: + +```text +contig_001 GenomeA +contig_002 GenomeA +contig_003 GenomeB +``` + +The first column is the original FASTA record ID: the token after `>` up to the first whitespace. Do not use MP's renamed `Sample-C123` IDs here. The second column is the bin ID. Contigs may be absent if unbinned, but every supplied contig must belong to the assembly and must occur at most once in the map. Periods in bin names become underscores; avoid names that collide after normalization. Use names beginning with a letter and consisting of letters, numbers or underscores; `community` and names containing `non_binned` are reserved. + +A contig removed by assembly QC cannot contribute features downstream. A mapped genome with no usable annotated features may have no generated PGDB input. Therefore the number of genome IDs in a map is an input inventory, not a promise of that many successful PGDBs. diff --git a/docs/installation.md b/docs/installation.md new file mode 100644 index 0000000..a2848bf --- /dev/null +++ b/docs/installation.md @@ -0,0 +1,65 @@ +# Installation and first run + +Install MetaPathways, try the included three-sample dataset, then use your own data. Linux x86-64 is supported. Choose one installation method below; **Mamba is recommended**. + +## 1. Conda package with Mamba (preferred) + +```bash +mamba create -n metapathways --override-channels --strict-channel-priority \ + -c hallamlab -c conda-forge -c bioconda metapathways=4.0.0 +conda activate metapathways +``` + +This installs MP and its workflow dependencies, including MAGSplitter and Camelot. Prepare the included data, build the small reference database, and run all three samples: + +```bash +metapathways prepare_test -o ~/mp-test +cd ~/mp-test +metapathways build_db --test -d MPDB +metapathways analysis_wf \ + --manifest cami-test/all.tsv -o all -d MPDB \ + --annotation_dbs swissprot_test \ + --rRNA_refdbs SILVA_SSU_test SILVA_LSU_test \ + --skip_ptools --threads 4 --memory '4 GB' --max_tasks 2 +metapathways report -o all --serve --no-browser --port 8765 +``` + +Open the URL printed by the report server. On a remote server, use an [SSH tunnel](reports-tutorial.md#view-a-remote-report-through-ssh). The **2.4 MiB input dataset is included** in the package and is derived from CAMI II ([Meyer et al., 2022](#cami-references)); database preparation downloads enzyme and taxonomy support records. The test covers annotation, paired-read abundance, genome splitting, reports and exploration. Pathway inference is skipped because it requires your own Pathway Tools license. + +## 2. Quay: Docker or Apptainer + +```bash +# Docker +docker pull quay.io/hallamlab/metapathways:4.0.0 + +# Or Apptainer +apptainer pull metapathways.sif docker://quay.io/hallamlab/metapathways:4.0.0 +``` + +The image includes the same workflow dependencies and test data. Follow the [Docker three-sample test](containers.md#docker-three-sample-test) or [Apptainer three-sample test](containers.md#apptainer-three-sample-test) to run the commands with your working directory mounted for persistent results. Licensed Pathway Tools is a separate image. + +## 3. Local installation from GitHub + +```bash +git clone https://github.com/hallamlab/MetaPathways.git +cd MetaPathways +mamba env create -f docker/conda_base.yml +mamba run -n metapathways python -m pip install . +conda activate metapathways +``` + +Then run the **same three-sample commands under option 1**, starting with `metapathways prepare_test -o ~/mp-test`. MP installs its Python workflow helpers automatically. The data comes from the installed package; the test does not depend on your checkout location. + +## Try your own data + +Build a production reference database, then annotate an assembly: + +```bash +metapathways build_db -d ~/MPDB --func swissprot -a fast +metapathways run -i /path/to/assembly.fasta -o results -d ~/MPDB --threads 8 +``` + +For assemblies with reads and genome maps, follow the [complete workflow](inputs.md). **If you want PGDBs, complete the [Pathway Tools installation guide](pathway-tools.md) before starting that workflow.** The small test references are for testing only. + +```{include} includes/cami-references.md +``` diff --git a/docs/overview.md b/docs/overview.md new file mode 100644 index 0000000..6ca008d --- /dev/null +++ b/docs/overview.md @@ -0,0 +1,17 @@ +# What MetaPathways does + +For the tool-by-tool diagrams and citations, see the [detailed workflow](detailed-workflow.md). + +The [main workflow figure](index.md) and the diagrams below share the appnote's gray modules, blue inputs, green outputs and serif typography. Diamonds denote compute steps, not decisions; the schema diagram retains its relationship notation. + +MetaPathways connects assembly annotations, read abundance and optional pathway inference in a shared set of sample and genome results. It accepts one sample or a collection and runs locally or through Slurm. + +[![Overview](assets/diagrams/overview.svg)](assets/diagrams/overview.svg) + +[Zoom diagram](assets/diagrams/overview.svg) · [Mermaid source](diagrams/overview.mmd) + +Assemblies are required. Reads add abundance measurements; genome assignments let MP split community annotations into genome-specific inputs. MP does not assemble reads or perform genome binning. Reports remain available when optional inputs are omitted. + +**For PGDBs, complete the [Pathway Tools installation guide](pathway-tools.md) first.** Use `--skip_ptools` to run without pathway inference. The [data-flow diagram](data-flow.md) shows these dependencies in more detail. + +Start with [installation and the included three-sample test](installation.md), then follow the [complete workflow](analysis.md). diff --git a/docs/pathway-tools.md b/docs/pathway-tools.md new file mode 100644 index 0000000..2737723 --- /dev/null +++ b/docs/pathway-tools.md @@ -0,0 +1,283 @@ +# Pathway Tools: licensing, image builds, databases and inference + +[Home](index.md) · [Test walkthrough](test.md) · [Complete flags](cli-reference.md) + +## Defaults and explicit choices + +| Setting | MP default | Your alternative | +| --- | --- | --- | +| Reaction compatibility screening when building MetaCyc | Enabled; successful full screens automatically publish an MPDB compatibility list | `--skip_pt_screen` on `build_pt` or `build_db` | +| Taxonomic pruning for PGDB inference | Enabled; avoids the unpruned rescoring pass | `--no_taxprune` | +| Organism taxonomic scope | `all`: NCBI 131567, cellular life; broad enough for mixed-domain inputs | `--taxonomic_scope bacteria`, `archaea`, `eukaryotes`, or `--taxon_id ID` | +| Transport inference in SIF runs | Enabled | `--no_transport_inference` | +| CPUs per PGDB | One; concurrency uses independent PGDBs | Control concurrency through the workflow resource flags | + +These defaults apply to `ptools` and `analysis_wf`. Taxonomic scope guides PGDB +inference without changing the gene-level taxonomy in MP annotation tables. +The compatibility list removes only unsafe **explicit reaction assignments**, +not genes, annotations or sequences; the same reaction may still be inferred +later. The sections below explain the evidence, limits and diagnostic records. + +## Understand the three different databases + +| Name | What it contains | How you use it | +| --- | --- | --- | +| MPDB | Reference protein/RNA sequences, search indexes, taxonomy and mapping tables | `-d /path/to/MPDB` on annotation/workflow commands | +| MetaCyc in Pathway Tools | Reference biochemical knowledge used for pathway inference | Bundled in an appropriate licensed Pathway Tools distribution | +| Your PGDB | Inferred pathways/reactions and gene associations for a community or genome bin | Created by `ptools` or `analysis_wf` under the sample output | + +Adding MetaCyc to **MPDB** enables another protein annotation search. Pathway Tools can infer pathways from SwissProt-derived annotations without that search: it still consults its own bundled MetaCyc knowledge base. Therefore “without MetaCyc annotation” does not mean “without MetaCyc pathway knowledge.” + +EcoCyc describes E. coli; MetaCyc is a multi-organism reference; BioCyc is the collection of organism PGDBs. Compare sample pathway counts with the appropriate kind of database and counting unit. Base pathways, superpathways, and signaling pathways are not interchangeable totals. + +## Get the installer + +Start at the official [Pathway Tools site](https://www.pathwaytools.org/) and its licensing/download link. SRI provides separate academic and commercial licensing routes. The [academic license page](https://bioinformatics.ai.sri.com/ptools/licensing/ptools-academic-license.shtml) describes eligibility and the request process; SRI sends download instructions to the approved technical contact. Use the terms supplied by SRI for your organization. + +Download the **Linux x86-64 installer**, even if your browser is on a Mac or Windows laptop: MP builds a Linux container on the analysis server. Choose a licensed distribution containing MetaCyc; the vendor describes its editions in the [installation guide](https://www.pathwaytools.org/installation-guide/released/index.html). Retain the installer filename, version, download records and checksum. Download addresses provided in license correspondence need not be public. + +If it is on your laptop, copy it to the server using your normal file-transfer method. For example, in a local terminal, replacing the username and host: + +```bash +scp ~/Downloads/pathway-tools-29.5-linux-64-tier1-install user@server:~/Downloads/ +``` + +Create the destination directory on the server first if needed. Do not unpack or manually run the installer before giving it to `build_pt`. MP performs the unattended installation within the build. Version 29.5 is a tested example, not a claim that it is the newest version. Consult the vendor's [release notes](https://bioinformatics.ai.sri.com/ptools/release-notes.html) for newer releases and verify their installer compatibility separately. + +## Host prerequisites + +Activate the MP environment and check `apptainer --version` and `nextflow -version`. Building requires network access for the container base, operating-system packages and official patches, plus writable space for the installer, temporary build and finished SIF. A build can temporarily use substantially more disk than the final image. Use `df -h` to inspect free space. + +The host must permit Apptainer's unprivileged/fakeroot build path. Administrators may need to configure subordinate UID/GID ranges and the `newuidmap`/`newgidmap` helpers (`uidmap` on Ubuntu). Installing a Conda executable cannot override a host security policy. Ask your administrator if a user-namespace or fakeroot error appears. Docker is not required. + +Use shared storage for a finished image that compute nodes need to read. Local scratch can accelerate building on slow network filesystems. `APPTAINER_TMPDIR` is an optional Apptainer troubleshooting setting, not required to run MP. See your site's Apptainer setup when choosing scratch; it must have enough space. + +## Build and register once + +Run on the server, with the environment active: + +```bash +metapathways build_pt \ + -i ~/Downloads/pathway-tools-29.5-linux-64-tier1-install \ + -o ~/mp-containers +``` + +MP uses Nextflow to build the image, download official release-specific SRI patches, validate the installation, and register the resulting SIF. Validation checks Pathway Tools startup and a small BLAST database/search. Wait for the build command to complete successfully before launching dependent analyses. + +`-o` is the image directory. Without it, the default is `~/.local/share/metapathways/containers` (or its XDG equivalent). Registration is stored at `~/.config/metapathways/ptools.json` (or its XDG equivalent). The adjacent `.sif.json` records provenance, including hashes and patch records. `--ptools_version 29.5` supplies a version when MP cannot recognize a renamed installer. `-t 2` controls compression CPUs, not the number of CPUs used by later PGDB tasks. + +An explicit new build creates a new image and preserves older images. Automatic patch downloads are disabled during analyses, so runs use the patch snapshot built into that image. There are no MP-authored edits to licensed Pathway Tools code or MetaCyc frames. A successful image validation does not prove that every possible genome will infer successfully. + +Subsequent `ptools` and `analysis_wf` commands use the registered SIF automatically. No image environment variable is necessary. To pin a particular image for a comparison, add: + +```text +--image /absolute/path/to/pathway-tools-VERSION-HASH.sif +``` + +Explicit `--image` takes precedence over the environment override and registration. Do not delete a SIF referenced by a running task. Rebuilding an image or changing the selected path can invalidate PGDB task reuse. On Slurm the selected path must be readable at the same location on all compute nodes. + +## Build the MPDB MetaCyc annotation reference at the same time + +Prepare public references first, then build the licensed image and add its MetaCyc reference: + +```bash +metapathways build_db -d ~/MPDB --func swissprot -a fast +metapathways build_pt \ + -i ~/Downloads/pathway-tools-29.5-linux-64-tier1-install \ + -o ~/mp-containers -d ~/MPDB -a fast +``` + +The second command exports the bundled MetaCyc protein sequences and biochemical flat files from the SIF, constructs protein-to-reaction mappings and pathway/compound/ontology tables, and indexes the proteins for FAST. It does not refresh SwissProt or the other existing references. The [installed-file table](pgdb-workflow.md#metacyc-from-pathway-tools) lists the exact MPDB destinations. + +To add or rebuild MetaCyc later using the registered image: + +```bash +metapathways build_db -d ~/MPDB --func metacyc -a fast +``` + +To choose a different licensed source: + +```bash +metapathways build_db -d ~/MPDB --func metacyc -a fast \ + --metacyc_source /path/to/pathway-tools.sif +``` + +`--metacyc_source` also accepts a local complete exported MetaCyc `data/` directory. It needs `protseq.fsa`, `proteins.dat`, `enzrxns.dat`, `reactions.dat`, `pathways.dat`, `compounds.dat`, and `classes.dat`. A `protseq.fsa` alone is insufficient even if its sequence statistics match another FASTA. Some native installs keep most knowledge-base data in the executable; the SIF export obtains the missing flat files. + +Use `-a blast` if subsequent annotation will use `--annotation_algorithm BLAST`. MetaCyc rebuilding replaces its existing reference and removes obsolete index files, including the other aligner's indexes. **Do not rebuild a reference directory while analyses are reading it.** Use a separate MPDB for a concurrent reference-version comparison. + +The ordinary public-reference build does not acquire licensed MetaCyc. You supply the authorized installer or data. MP does not grant permission to redistribute the installer, SIF, patches or reference files; preserve them according to the applicable vendor terms. + +## Run pathway inference + +For an annotated sample at `results/SampleA`: + +```bash +metapathways ptools -o results/SampleA \ + --entity community --taxprune --taxonomic_scope all \ + --max_cpus 8 --max_memory '32 GB' +``` + +For all available entities, omit `--entity`: + +```bash +metapathways mag_split -o results/SampleA -m /path/to/SampleA.tsv +metapathways ptools -o results/SampleA \ + --taxprune --taxonomic_scope all \ + --max_cpus 8 --max_memory '32 GB' +``` + +`--entity MAG_001` selects one actual normalized MAG ID. There is no `--entity MAGS` keyword. Omitting the flag includes the community and available genome bins; successful unchanged entities can be reused. + +`-o` here is the **sample directory**, unlike `run` and `analysis_wf`, whose `-o` is the parent output directory. A wrong level can make MP look for missing `ptools/0.pf` inputs. + +Every PGDB task uses one CPU. Concurrency comes from multiple independent PGDB tasks; assigning eight threads does not make one Pathway Tools process eight times faster. `ptools --memory` controls each standalone PGDB reservation; `analysis_wf --ptools_memory` optionally overrides PGDB reservations within the complete workflow; without it PGDBs inherit `--memory`. An explicit `--max_memory` additionally bounds scheduled reservations. Local reservations are not hard memory ceilings. + +SIF tasks receive private home, data and temporary state. MP also isolates X-display sockets; independent PGDBs can run together without the native installation's shared state. Native Pathway Tools remains a legacy serialized option for standalone `ptools`; the complete `analysis_wf` requires the SIF route. The old `--container` flag is not a synonym for `--image`; use the image route shown here. + +## Choose a taxonomic scope + +`--taxprune` enables taxonomic pruning. `--taxonomic_scope` selects the organism taxon assigned to the staged PGDB inputs: + +| Scope | NCBI taxon | Use | +| --- | ---: | --- | +| `all` | 131567 | Broad cellular-life scope for a mixed community | +| `bacteria` | 2 | A bacterial community or bacterial genome | +| `archaea` | 2157 | An archaeal community or archaeal genome | +| `eukaryotes` (`euks`) | 2759 | Eukaryotic inputs | + +`all` includes multicellular eukaryotes; it is not a microbial-only or unicellular filter. There is no supported `prokaryotes` union scope. `--taxon_id NCBI_ID` allows a more specific positive numeric taxon and is mutually exclusive with the convenience scope flag. + +The scope applies to all selected entities in that invocation. It guides pathway inference; it does not remove contigs or rewrite MP's gene-level taxonomic annotations. To use different scopes for different MAGs, run separate entity-specific `ptools` commands. + +**Defaults: taxonomic pruning enabled, scope `all` (cellular life).** You can omit both flags for that behavior. Use `--no_taxprune` to disable pruning, `--taxonomic_scope bacteria`, `archaea` or `eukaryotes` to narrow the scope, or `--taxon_id` for a specific taxon. Pruning constrains inference using the chosen taxon; it does not disable other pathway-selection rules. Record these settings in your methods. Pathway Tools 29.5 can fail during its unpruned rescore pass; MP retains diagnostics rather than treating partial output as successful. + +See the vendor User Guide's batch PathoLogic discussion and [MP's diagnostic detail](workflow.md#pathway-tools-failure-diagnostics). The installed PDF is authoritative for the installed version. + +## Explicit reaction blacklist + +For SIF runs, MP applies a matching MPDB compatibility list to private staged community and MAG PGDB inputs. Its bundled fallback, `metapathways/resources/ptools_reaction_blacklist.json`, is restricted to the exact SIF tested for those entries. Legacy native runs use the bundled known-trigger list because they have no SIF fingerprint. It removes only listed `METACYC` reaction assignments; original annotation tables, feature IDs, sequences, EC assignments and function names are retained. Each attempt records removed assignments and reasons in `ptools-reaction-filter.json` alongside its input diagnostics (native runs save the audit in the entity output directory). Compact SIF runs include this audit in the diagnostic archive. + +The bundled known triggers are `TRANS-RXN8J2-121` and `RXN8J2-204`. For `TRANS-RXN8J2-121` in Pathway Tools 29.5, supplying it explicitly imports sublancin precursor proteins before name matching; an imported protein has no input raw-gene record, causing a NIL structure error. A one-ORF reproducer confirmed the failure. Omitting the explicit assignment allowed the build to finish and name matching recovered the same reaction later. This filter prevents the known early-import failure; it does not forbid later inference of the reaction or certify all Pathway Tools inputs. Changes to the blacklist invalidate affected PGDB checkpoints. + +## Transport inference and sequence-backed inputs + +Transport inference (TIP) is enabled by default for SIF runs. It uses annotations to identify probable transport proteins and associate or create transport reactions. High- and low-confidence predictions are reported separately. Low-confidence or ambiguous annotations are not equivalent to experimentally demonstrated transport. + +`--no_transport_inference` disables TIP for a controlled comparison. It is accepted by `ptools` and `analysis_wf`. Record it when comparing pathway/reaction totals; TIP can change reactions without changing the number of base pathways. + +MP stages real per-contig DNA sequences, feature coordinates, strands and available Prodigal genetic codes before invoking Pathway Tools. This allows Pathway Tools to derive its protein BLAST database. The BLAST tools **inside the SIF** serve that function; they are separate from FAST/BLAST annotation indexes **in MPDB**. + +Annotations with multiple EC assignments are written as separate EC entries. Provisional EC identifiers are retained and may be rejected by Pathway Tools. Recognized tRNA display names separate amino-acid labels from anticodons, while stable feature IDs are preserved. Some ID-parsing warnings can remain. Neither formatting step invents biological assignments. + +## Outputs, warnings and failures + +Look under `results/SampleA/results/pgdb/community/` and `.../MAGs/MAG_ID/` for the archive, pathway table and pathway-to-ORF table. Each attempt retains `diagnostics/ATTEMPT/execution.json` and available logs. `input/pathologic.log` contains the internal inference messages; `build-xvfb.log` and `export-xvfb.log` contain display-wrapper diagnostics in the corrected version. + +| Symptom | What to check or do | +| --- | --- | +| No registered image | Complete `build_pt`, or specify an existing readable `--image` | +| Installer version not recognized | Keep its original filename or supply `--ptools_version` | +| `debconf: delaying package configuration` | This message alone is normal in a minimal image; inspect the final build status | +| Missing compound / inverse-link failure | Check `pathologic.log`, image/patch provenance and scope flags; do not delete reference compounds to suppress it | +| Protein BLAST DB cannot be created | Check sequence staging and the internal log; installing BLAST alone cannot supply missing sequence | +| `Done` followed by failure | Check the recorded phase and wrapper logs; `Done` is not a substitute for successful export/archive | +| Provisional EC or ambiguous transport warning | Keep the evidence; a formatting change cannot establish a missing biological assignment | +| Zero inferred pathways | A valid empty result produces header-only pathway and pathway-to-ORF tables. MP verifies the empty final pathway report and inference evidence list before handling the Pathway Tools empty-export message; raw PGDB files are preserved | +| MAG lacks selected genes | The complete workflow can skip a bin with no generated PF input; distinguish this from zero inferred pathways | +| Nonzero exit during export | Preserve the saved PGDB and diagnostics; do not label the result complete from build messages alone | + +Community failures fail the workflow. MAG failures may be optional, allowing other entities to finish, but remain failures in task records. `SUCCESS` at the overall scheduler level can coexist with optional failures. Never use it alone to claim all MAGs succeeded. The report distinguishes output availability from latest task outcome. + +Failed on-disk PGDBs are retained under `diagnostics/ATTEMPT/failed-pgdbs/`. MP does not automatically certify or publish partial recovery. Retry with the same parameters after fixing the cause, and inspect the new execution record. [Restart guidance](execution.md#logs-temporary-files-and-restarting) explains reuse and targeted reruns. + +## Screen reaction compatibility (maintainers) + +After building a licensed Pathway Tools image, maintainers can screen the explicit +reaction IDs in an MPDB against that image: + +```bash +metapathways screen_pt -d MPDB -o reaction-screen --max_tasks 1 +``` + +The command uses the image registered by `build_pt`; use `--image /path/to/ptools.sif` +to select another image. Each container has a private home and temporary PGDB. +Publication downloads are disabled. `--scratch_dir /path/to/local/scratch` places +these temporary builds on local storage; only inputs, diagnostic logs and receipts +are retained in the output. This is a local maintainer command, not a Slurm workflow. + +The screen first builds a no-reaction baseline, then tests batches of 100 reaction +IDs. Failed batches are split until individual triggers are isolated. A candidate +requires two isolated failures and a successful no-reaction control. Timeouts and +uncertain results are recorded as inconclusive. Batch failures whose two halves +pass are recorded separately as possible interactions. + +For a targeted check: + +```bash +metapathways screen_pt -d MPDB -o reaction-check \ + --reactions TRANS-RXN8J2-121 +``` + +Repeat the same command to reuse completed attempts and retry interrupted or +inconclusive attempts; earlier diagnostics are preserved. The image, mapping table and +screen settings must match the saved checkpoint; use a new output directory when +changing them. `--max_tasks` can be changed on resume. Each build defaults to a +30-minute timeout (`--timeout`, in seconds). + +Review `summary.json`, `blacklist-candidates.json` and each attempt's logs before +adding any entry to the shipped blacklist. Standalone candidates are not activated unless you explicitly use `--publish`. +Publication requires a full screen with no inconclusive or interaction failures. Synthetic explicit-ID screening does not certify all enzyme names, +taxonomic contexts or combinations of reactions. + +### Automatic compatibility screening during MetaCyc builds + +`build_pt -i INSTALLER -d MPDB` and `build_db -d MPDB --func metacyc` +now screen reactions **by default**, after preparing MetaCyc. No additional user +command is needed. The screen tests the licensed SIF and database together, bisects +failed batches, and requires repeated isolated failures plus successful controls. +It adds time to the build and retains screen evidence under +`MPDB/.metapathways/ptools-screens/`. Screening honors `--max_tasks`, capped by the CPU and memory budgets. Each +PTools container uses one CPU; `--memory` is its memory reservation. For example, +`--max_tasks 32 --memory '8 GB'` can screen 32 batches concurrently when 32 CPUs +and 256 GB of memory are available. Local runs default to available resources; +Slurm defaults to four screening containers if `--max_tasks` is omitted. +The outer Nextflow screening task reserves the aggregate CPU and memory for +those containers. On Slurm they run together inside one job allocation, not +as separate cluster jobs. Planning prints the effective concurrency and +reservation. Failed-batch splitting proceeds sequentially within each batch. + +Use `--skip_pt_screen` on either build command to opt out. When importing a +MetaCyc data directory instead of a SIF, provide `--screen_image /path/to/ptools.sif` +or first register an image with `build_pt`; directory data alone cannot run the +compatibility checks. An unresolved screen fails the screening stage while +retaining the prepared database and diagnostic checkpoints. + +A completed default screen publishes +`MPDB/functional_categories/ptools_reaction_compatibility.json`. MP finds it using +the reference database recorded in the sample run log and checks its mapping and +SIF fingerprints before using it for PGDB builds. It records the selected list in +`ptools-reaction-filter.json`. A mismatch uses only fallback entries confirmed for the selected SIF, rather +than applying an unrelated database-specific list. If neither matches, MP does +not assume other versions share the same defects; rebuild MetaCyc with screening +for the selected image. Compatibility-list changes +invalidate PGDB checkpoints. + +To publish an already completed standalone full screen, repeat its command with +`--publish`; completed attempts are reused. + +## Intermittent container startup failures + +PGDB builds automatically retry an explicit Apptainer container-creation mount +failure up to two times (three attempts total), waiting 5 seconds and then 15 +seconds. This applies only before the container shell starts. Input-validation +errors, missing images, and failures during container setup, pathway inference, +export or archiving are not retried by this mechanism. A persistent mount error +still fails the task and needs investigation of the cluster/container runtime. + +Each attempt streams to the task log and is saved as `container-attempt-N.log`. +`execution.json` records each attempt's exit code, stage, elapsed seconds and any +retry delay. Compact mode preserves these diagnostics in its existing archives. +Task resource/runtime measurements include all attempts and delays; retain these +records when separating successful execution from infrastructure retry overhead. +No SIF rebuild or additional command-line flag is required. diff --git a/docs/pgdb-workflow.md b/docs/pgdb-workflow.md new file mode 100644 index 0000000..f50903e --- /dev/null +++ b/docs/pgdb-workflow.md @@ -0,0 +1,91 @@ +# MAGs and pathway inference + +PGDB inference defaults to **taxonomic pruning enabled, scope `all` (cellular life)**. +MetaCyc builds default to reaction compatibility screening. See [defaults and +opt-outs](pathway-tools.md#defaults-and-explicit-choices) before changing these +analysis settings. + +## Build Pathway Tools once + +Follow the [Pathway Tools licensing and installer guide](pathway-tools.md#get-the-installer) to obtain your licensed Linux x86-64 installer, then provide it to MP: + +```bash +metapathways build_pt \ + -i ./pathway-tools-29.5-linux-64-tier1-install \ + -o ./containers +``` + +Nextflow runs the Apptainer build, validates Pathway Tools startup, and registers the finished SIF in `~/.config/metapathways/ptools.json`. Subsequent `ptools` commands use that image automatically. Without `-o`, images go under `~/.local/share/metapathways/containers`; the XDG equivalents are honored. `--image` on `ptools` or `METAPATHWAYS_PTOOLS_IMAGE` overrides registration. + +Apptainer must support unprivileged builds on your host. Where subordinate UID/GID mappings are configured, its `newuidmap` and `newgidmap` helpers must be installed by the system administrator (Ubuntu provides them in `uidmap`). Building needs internet access for the base image and operating-system packages. Docker is not required. The installer itself is supplied by you; MP does not obtain a license. + +`build_pt -t` controls compression CPUs, default two. If shared storage makes package installation slow, set `APPTAINER_TMPDIR` to sufficiently large local scratch. Startup validation has been exercised with 29.5; do not assume every installer release has the same unattended interface. If the installer was renamed, supply `--ptools_version 29.5` explicitly. + +Every explicit `build_pt` invocation fetches the installer release's official Linux patches from SRI over HTTPS, following the [vendor's patch installation instructions](https://bioinformatics.ai.sri.com/ptools/faq.html). There is no prompt and no custom Pathway Tools code or MetaCyc data patch. Failed patch downloads stop the build. The build loads these patches before freezing the SIF; analysis invocations disable further patch downloads. Each invocation creates a separate image, preserving existing images. The adjacent `.sif.json` records installer/image/recipe hashes, patch source URLs and individual checksums, snapshot hash, and startup validation output; the patch manifest is also embedded in the image. Keep this private image to reproduce its exact patch set. + +New images include NCBI BLAST+ and its configuration. Before registration, validation creates and searches a tiny synthetic protein database, then checks Pathway Tools startup. This BLAST installation supports Pathway Tools' own sequence databases and hole filling; it is separate from the FAST/BLAST annotation indexes in MPDB. Installing official patches or BLAST does not establish that any particular Pathway Tools inference failure has been fixed. + +## MetaCyc from Pathway Tools + +To build the SIF and prepare its bundled MetaCyc reference in an existing MPDB with one command: + +```bash +metapathways build_pt \ + -i ./pathway-tools-29.5-linux-64-tier1-install \ + -o ./containers -d /path/to/MPDB -a fast +``` + +Use `-a blast` for BLAST annotation indexes instead. Nextflow first builds and validates the SIF, then exports its bundled MetaCyc flat files alongside `protseq.fsa` in private staging. It runs the repository's `metacyc_mapping_build.py` and `metacyc_build_ont.py`, checks the reference files and tables, indexes the proteins, and installs: + +| Location relative to MPDB | Contents | +| --- | --- | +| `functional/metacyc` | Protein FASTA from the selected MetaCyc release | +| `functional/formatted/metacyc.*` | Selected FAST or BLAST indexes | +| `functional/formatted/metacyc-names.txt` | Protein identifiers and descriptions | +| `functional_categories/MetaCyc-monomer-rxn-pairs.tsv` | Protein-to-reaction mappings, including containing complexes | +| `functional_categories/MetaCyc-PWY-RXN-CMP-map.tsv` | Pathway, reaction, enzyme and primary compound mappings | +| `functional_categories/MetaCyc_PWY_Ontology.tsv` | Pathway ontology | +| `functional_categories/MetaCyc_reldate.txt`, `MetaCyc_provenance.json` | Release, source/image hashes, input checksums and mapping counts | + +The MetaCyc preparation task uses one CPU. Other MPDB references are not refreshed. Preparation replaces the existing MetaCyc reference and removes obsolete MetaCyc index files, including indexes for the other aligner, so run it when no analyses are reading that MPDB. Annotation runs must use the selected index format. Proteins without reaction mappings are counted in provenance; absence of a reaction assignment is permitted. + +To add MetaCyc later using the registered SIF: + +```bash +metapathways build_db -d /path/to/MPDB --func metacyc -a fast +``` + +Select a particular SIF with `--metacyc_source /path/to/pathway-tools.sif`, or provide a complete licensed flat-file `data/` directory. That directory must contain `protseq.fsa`, `proteins.dat`, `enzrxns.dat`, `reactions.dat`, `pathways.dat`, `compounds.dat`, and `classes.dat`. A FASTA alone cannot supply the reaction and pathway mappings. Native installations may contain only sequence files in `data/` because their remaining reference data are built into the executable; the SIF export handles this. Files supplied over SFTP must first be copied locally. + +MetaCyc is opt-in and is not downloaded by the ordinary default `build_db` command. The user supplies the licensed installation/data and remains responsible for its permitted use; MP does not grant a license or publish the installer, reference data, patches, or SIF. `--dryrun` plans either route without exporting references, fetching patches, indexing, or registering an image. + +## Community pathways + +```bash +metapathways ptools -o results/sample --taxprune --taxonomic_scope all +``` + +These examples use broad cellular-life taxonomic pruning. Choose the scope to match your analysis and record it in your methods; see [scope choices](pathway-tools.md#choose-a-taxonomic-scope). + +Each SIF task gets private Pathway Tools data, home and temporary state, allowing concurrent isolated instances. Pathway Tools uses one CPU per PGDB. A community PGDB failure fails the command. Native Pathway Tools is still supported when no image is selected, but serialized to protect shared state. The legacy `--container` flag retains its original meaning and bypasses automatic SIF selection. + +## Add MAG pathways without reannotating + +Create a tab-separated map with **no header**, containing original assembly contig identifiers and MAG identifiers: + +```text +original_contig_1 MAG_001 +original_contig_2 MAG_001 +original_contig_3 MAG_002 +``` + +Then run: + +```bash +metapathways mag_split -o results/sample -m contig_to_mag.tsv +metapathways ptools -o results/sample --taxprune --taxonomic_scope all +``` + +MAG splitting reuses the community annotation and contig mapping. It also preserves the supplied map at `magsplitter/contig_to_mag.tsv`, allowing the report to connect all retained MAG contigs and their ORFs. MAG Pathway Tools inputs are a smaller, selected set and are shown separately. + +Pathway Tools can fail for individual MAGs. These failures are recorded and allowed to remain optional; they do not fail an otherwise successful community workflow. Missing or failed inference is not equivalent to “zero pathways.” Inspect `entities` and execution history in the portal. `--taxprune` is enabled by default with `--taxonomic_scope all` (cellular life). Use `--no_taxprune` to opt out. diff --git a/docs/pr-testing.md b/docs/pr-testing.md new file mode 100644 index 0000000..82e7b38 --- /dev/null +++ b/docs/pr-testing.md @@ -0,0 +1,19 @@ +# PR tester checklist + +[Home](index.md) · [Test commands](test.md) · [Release process](releasing.md) + +Test the exact proposed commit before merging to `dev`. Record outstanding checks in the PR. Production promotion is a separate `dev` → `main` PR, as described in the [release process](releasing.md). Use a clean installation and a fresh output directory. Record `git rev-parse HEAD`, the platform, and the installed dependency versions. A version number alone does not identify the tested commit. + +1. Follow [source installation](getting-started.md). Check out the PR's exact commit after cloning; do not test an older editable installation by accident. Record the resolved dependency versions, including the pinned MAGSplitter and Camelot helpers installed with MP. +2. Follow the [CAMI II test walkthrough](test.md) ([Meyer et al., 2022](#cami-references)), running its single-sample and two-sample commands with `--skip_ptools`. Check required task outcomes, nonempty abundance outputs, the three genome-bin assignments per sample, and distinct paired read paths. +3. Repeat the two-sample command unchanged. All successful workflow tasks, including `COMPUTE_TPM`, should report `ALREADY_COMPUTED`. Report rebuilding may still run. A repeat is not a fresh performance measurement. +4. Run all three inputs through automatic discovery with a new output directory. Confirm the same sample IDs and input associations as `all.tsv`. Test an intentionally mismatched filename in a separate copy: planning must stop with an actionable input-layout error. +5. Serve the report and open `EDA_portal.html`. Filter each sample, search annotations, and export a CSV. Confirm sample/entity IDs and absence of mixed sample records. Sparse annotations and empty RNA/pathway tables may be expected with the tiny reference fixture; skipped pathways must not appear as successful inference. +6. If licensed, build a Pathway Tools image from your installer. Repeat a small workflow into a fresh output with `--taxprune --taxonomic_scope all` and without `--skip_ptools`. Confirm community/MAG status, sequence-backed inputs, archives and pathway exports. Keep licensed inputs/images private. Record expected no-pathway outcomes separately from failures. +7. If a cluster is available, test a small submission using the documented Slurm settings. Confirm scheduler resource requests and bounded concurrent submissions. Otherwise explicitly mark Slurm untested. +8. Have a maintainer manually run the Release workflow on this branch with **blank `release_tag` and `source_run_id`**, optionally enabling `test_containers`. Review the Conda installed-package integration, container validation, and security reports. No release tags or public uploads are needed for this step. + +Attach the tested commit, commands, task summaries, dependency versions, and relevant logs to the PR, with local credentials and private input paths removed as needed. State which optional checks were not performed. A maintainer should resolve failures and obtain sign-off on the final commit before merging; subsequent code changes require retesting the affected behavior. + +```{include} includes/cami-references.md +``` diff --git a/docs/release-readiness.md b/docs/release-readiness.md new file mode 100644 index 0000000..9b543b9 --- /dev/null +++ b/docs/release-readiness.md @@ -0,0 +1,41 @@ +# MetaPathways 4.0.0 release readiness + +[Release instructions](releasing.md) · [Tester checklist](pr-testing.md) + +The production branch is `main`; `dev` is the integration branch. Installation +examples use version 4.0.0. A version update or branch push does not publish a +package, image, GitHub release or DOI. + +## Validation evidence + +The corrected 49-sample HPC benchmark completed under version 3.5.2 with 6,519 +successful uncached tasks, 49 community PGDBs and 5,392 population PGDBs. The +4.0.0 release must retain its own tested source commit and package/container +validation; changing the version does not relabel that historical benchmark. +The earlier three-sample installation audit is retained as +[historical validation evidence](validation/reviewer-2026-10-02.json). + +The Conda recipe builds pinned MAGSplitter and Camelot sources into the package. +The Quay image uses that same validated Conda artifact. The bundled test data +are included; licensed Pathway Tools, its images and production reference data +are excluded from public packages. + +## Source validation (2026-10-09) + +- 170 runtime tests and 20 release-control tests passed. +- Strict documentation build, links/anchors in 40 rendered pages, and citation metadata passed. +- Version 4.0.0 source distribution and Linux/Python 3.11 wheel built successfully. +- Installed-wheel assets, version, bundled test preparation, MAGSplitter CLI and Camelot import passed outside the source checkout. +- Final Conda integration and Docker/Apptainer validation remain release-build checks. + +## Release checks + +1. Pass unit, release-control, documentation, source and wheel checks. +2. Build the final committed Conda package and run its installed workflow test. +3. Build/test Docker and Apptainer artifacts and review the vulnerability scan. +4. Retain checksums, dependency exports and the tested commit for each artifact. +5. Create the matching version tag after production validation. Manually select + Anaconda and Quay publication only when ready; both default to off. +6. Publish a GitHub release separately when DOI archiving is intended. + +See the [release process](releasing.md) for exact commands and recovery steps. diff --git a/docs/releasing.md b/docs/releasing.md index ed15991..09cda26 100644 --- a/docs/releasing.md +++ b/docs/releasing.md @@ -1,34 +1,45 @@ # Releasing MetaPathways -The source version in `metapathways/_version.py` is authoritative. An installed -`3.5.0.dev64` package is an older distribution; changing GitHub does not upgrade -that environment. The first stable release from this checkout is **3.5.0**. -Use **3.5.1**, **3.5.2**, etc. for subsequent patch releases, or **3.5.1rc1** -for a preview. Tags have a leading `v`; package versions do not. +[User documentation](index.md) · [Reproducibility](reproducibility.md) + +This chapter is for maintainers preparing and publishing packages. Every release +must pass package and workflow validation before publication. + +The source version in `metapathways/_version.py` is authoritative. The current +release series is **4.0**, with package version **4.0.0** and tag **v4.0.0**. +Use subsequent patch versions such as **4.0.1**, or **4.0.1rc1** for a preview. +Tags have a leading `v`; package versions do not. Updating a repository does not +upgrade an already installed environment. The supported release package targets **Linux x86-64 / Python 3.11**. -It includes the core annotation pipeline and bundled FAST/metacount executables. -MAGSplitter, camelot-frs, and licensed Pathway Tools remain separately installed -optional dependencies. The release build does not download moving Git branches -and silently include them in the Conda package. +It includes the annotation pipeline, bundled FAST/metacount executables, test +data, MAGSplitter and Camelot. Helper sources are pinned to exact commits with +SHA256 checksums. Licensed Pathway Tools is obtained separately by the user. ## Normal release using GitHub CI -From a normal checkout of `dev`, using your existing Git SSH/HTTPS credentials: +Use the branch sequence: **feature branch → `dev` → `main`**. `dev` is the integration branch; `main` is the production branch. GitHub's default branch and the Read the Docs `latest` version are separate settings; neither setting promotes a release. + +Open feature work as a draft PR targeting `dev` while benchmarks or test checks are pending. Record the tested commit and outstanding checks in the PR. After benchmark review, independent tester sign-off and passing CI, mark it ready and merge into `dev`. Then open a separate promotion PR from `dev` into `main`, review the complete diff, and validate the resulting production commit before tagging. An older `main` can include changes accumulated on `dev` before the feature PR; review those too. Do not reset or force-push production to bypass that review. + +Prepare any version changes on the feature branch before final PR review: ```bash -python scripts/release.py prepare 3.5.0 +python scripts/release.py prepare 4.0.0 git diff -git add README.md metapathways/_version.py conda_recipe/meta_template.yaml -git commit -m "Prepare MetaPathways 3.5.0" -python scripts/release.py publish +git add README.md CITATION.cff metapathways/_version.py conda_recipe/meta_template.yaml +git commit -m "Prepare MetaPathways 4.0.0" ``` Commit the release script, workflow, tests, and other implementation files as well when installing this workflow for the first time. If `prepare` changes nothing, -there is no version-only commit to make. +there is no version-only commit to make. After the promotion PR is merged, the production commit is validated and release approval is given, use a clean, up-to-date checkout of `main` with your existing Git SSH/HTTPS credentials: -`publish` requires a clean checkout on `dev`. It creates an annotated tag and +```bash +python scripts/release.py publish --branch main +``` + +`publish` requires a clean checkout on the selected release branch. Always pass `--branch main` for production publication; the helper defaults to `main` and also accepts `dev` for explicitly selected integration releases. It creates an annotated tag and pushes the branch and tag atomically to `hallamlab/MetaPathways`. It never force pushes, moves a published tag, or automatically commits your work. GitHub CLI login is unnecessary on your machine. @@ -42,23 +53,49 @@ The tag workflow: 5. Runs database preparation and the included K12 integration test. 6. Requires all 16 named stages, nonempty key outputs, and no error markers. 7. Records logs, dependency exports, revision, and SHA-256 checksums. -8. Publishes a GitHub release with those downloadable assets. -9. Optionally uploads the same validated Conda package to Anaconda.org. +8. Retains the validated assets in GitHub Actions; tag pushes never publish. + +To publish, open **Actions → Release → Run workflow**, select the matching +version tag, and select the destinations you intend to publish. All switches +default to off: **publish_anaconda**, **publish_quay**, and **publish_github**. +Anaconda and Quay publish independently; neither requires a GitHub release. +The GitHub option may trigger archiving through an existing Zenodo integration. +There is no separate Zenodo API upload or checkbox. Only successful full builds are publishable. Release candidates are marked as GitHub prereleases and use the `hallamlab/label/rc` Conda channel; stable versions use `hallamlab` / label `main`. No wheel is advertised as platform-independent. To test CI without publishing, select **Actions → Release → Run workflow** on -the `dev` branch. Assets are retained as an Actions artifact. Dispatching the -workflow on a release tag also enables publishing. +a reviewed `dev` or `main` commit. Leave `release_tag` and `source_run_id` blank; optionally enable `test_containers` to build and scan Docker/SIF artifacts too. This does not publish packages, registry images, a GitHub release, or a DOI. Assets are retained as an Actions artifact. Selecting a tag alone does not enable publishing; destination checkboxes must be selected explicitly. + +## Shared release controls (MetaPathways and SCARAB) + +Both projects use the same manual **publish_anaconda** and **publish_quay** +checkboxes, unchecked by default. Both require a matching version tag and +successful validation for the selected destination. Store secrets in the +GitHub **release** environment (repository Actions secrets also work). +Optional environment approval rules are respected. + +No variables are required. `ANACONDA_OWNER` defaults to `hallamlab`; +`QUAY_REPOSITORY` defaults to `quay.io/hallamlab/metapathways` here and +`quay.io/hallamlab/scarab` in SCARAB. These optional variables override registry +destinations only. The old `PUBLISH_CONDA` and `PUBLISH_QUAY` variables are +ignored and can be removed. + +Each registry has its own publishing job. Use **Re-run failed jobs** to retry +a failed upload with the original tested artifacts while they remain available, +without rebuilding or repeating a successful upload to the other registry. +Do not use **Re-run all jobs** for an upload retry. No force-overwrite is used +for Conda packages. If an upload actually succeeded before its job failed, +inspect the registry before retrying. ## Enable Anaconda.org uploads In GitHub repository **Settings → Secrets and variables → Actions**: - Add the secret **ANACONDA_API_TOKEN**, with upload access to `hallamlab`. -- Set the repository variable **PUBLISH_CONDA** to `true`. +- Select **publish_anaconda** in the manual release form when ready to upload. GitHub releases work without this configuration. GitHub authentication does not authenticate Anaconda.org. The token is passed only to the upload step through @@ -71,8 +108,8 @@ build/test environment: ```bash mamba create -n metapathways-build -c conda-forge \ - python=3.10 conda-build conda-index anaconda-client pyyaml \ - pip wheel 'setuptools<81' + python=3.11 conda-build conda-index anaconda-client pyyaml \ + pip wheel 'setuptools>=83,<85' conda activate metapathways-build python -m pip install build @@ -117,10 +154,9 @@ Anaconda upload, rerun only the failed Anaconda job. `make release-prepare VERSION=3.5.0`, `make release-build`, and `make release-publish` wrap these commands. `make full-build` now builds, validates, and uploads to Anaconda using this controller; it requires the -packaging environment above and does not publish a GitHub tag. Publishing -containers, PyPI wheels, and licensed Pathway Tools runs is outside this release -workflow. The K12 example validates installation and core operation; it does -not reproduce the manuscript's CAMI2 performance benchmark. Reference downloads +packaging environment above and does not publish a GitHub tag. Publishing PyPI wheels and running licensed Pathway Tools are outside this release +workflow; Docker and Apptainer validation is described below. The K12 example validates installation and core operation; it does +not reproduce the manuscript's CAMI II performance benchmark ([Meyer et al., 2022](#cami-references)). Reference downloads are moving resources, and dependency specifications are not a complete lock. The recorded Conda export includes a temporary local URL for MetaPathways itself; when recreating it, replace that URL with the downloaded release package. @@ -137,6 +173,7 @@ After committing and pushing the workflow fix to `dev`, open **Actions → Relea - `release_tag`: `v3.5.0` - `source_run_id`: `36760751744` +- `publish_github`: checked only when ready to resume publication This reuses the successful Conda build from that run and checks its source commit against the original tag. Do not move the tag or run `publish` again for it. @@ -144,13 +181,13 @@ An ordinary “Re-run jobs” on the old run still uses the old workflow. ## Docker, Apptainer, and Quay -Every tagged release (including manual recovery with `release_tag`) builds a +Selecting **test_containers**, **publish_quay**, or **publish_github** builds a Linux amd64 Docker image from the exact validated Conda package and explicit dependency export. It runs the core integration test in Docker, requires all 16 stages and six nonempty outputs, then converts that same image into an Apptainer SIF. Apptainer is checked for the exact version and CLI startup; its read-only filesystem is not used for the bundled database-writing integration test. -The SIF, container checksums, and validation logs are attached to GitHub. +The SIF, container checksums, and validation logs are attached to GitHub only when **publish_github** is selected. They are also retained in the workflow's `container-assets` artifact. In Quay, open the **hallamlab organization → Robot Accounts → Create Robot @@ -161,7 +198,7 @@ In GitHub **Settings → Secrets and variables → Actions**, configure: | --- | --- | --- | | Secret | `QUAY_USERNAME` | Full robot username, e.g. `hallamlab+github_releases` | | Secret | `QUAY_PASSWORD` | Robot token generated by Quay | -| Variable | `PUBLISH_QUAY` | `true` | +| Optional variable | `QUAY_REPOSITORY` | Defaults to `quay.io/hallamlab/metapathways` | | Optional secret | `QUAY_API_TOKEN` | Quay OAuth token with `repo:write` access | The robot credentials upload images. The OAuth token updates the repository @@ -206,10 +243,10 @@ Licensed Pathway Tools is not included in public release containers. ## Security rebuilds without changing the application version -The security rebuild of MetaPathways 3.5.1 uses Python 3.11, Snakemake minimal +The historical security rebuild of MetaPathways 3.5.1 used Python 3.11, Snakemake minimal 9.27.0, urllib3 >=2.8.0, and setuptools >=83. The minimal Snakemake distribution provides the local CLI used for database preparation without the legacy stopit -runtime dependency. Container builds refresh the base image and apply Debian +runtime dependency. This records that release configuration; the current development workflow uses Nextflow, as described in the user guide. Container builds refresh the base image and apply Debian package updates before installing the validated Conda environment. A nonzero Conda build number produces a separate Git tag and release, for example @@ -219,7 +256,7 @@ original tag and published files. Prepare it with: ```bash python scripts/release.py prepare 3.5.1 --build-number 1 # Commit the reviewed changes, then: -python scripts/release.py publish +python scripts/release.py publish --branch main ``` Quay receives `3.5.1-build1` and `v3.5.1-build1` tags as well as updated `3.5.1`, @@ -237,3 +274,55 @@ the runner and blocks publication when it finds a high/critical vulnerability for which a fix is available. Both the full scan and filtered gate reports are retained as `container-security-report`. Unfixed distribution advisories need separate review; passing this gate does not mean the image has no vulnerabilities. + +## Feature-branch review before publication + +The intended sequence is **feature branch → single-server and HPC benchmarks plus tester review → PR merge to `dev` → promotion PR to `main` → production validation → release tag → publication**. Pushing a branch or opening a PR does not publish a release. Smoke checks run for PRs targeting `main` or `dev`, and pushes to those branches or `feat/**`. They validate the citation file, unit tests, docs, source/wheel builds, installed CLI entry points, and packaged test/report assets. + +Use the [PR tester checklist](pr-testing.md) for independent installation, the single- and two-sample CAMI workflows, restart behavior, optional licensed Pathway Tools, and report exports. Record the tested Git commit. Record the source commit as well as the package version when testing a candidate. + +The release version is **4.0.0**. Preparing this version does not publish it; branch and PR testing still happen before a release tag is pushed. Do not reuse 3.5.1 or its existing tags. Change the version with `prepare` before final release validation; it also updates the citation version and resets the Conda build number for a new version. Avoid editing runtime code/version files while a benchmark is executing from an editable checkout. + +Before merging, manually dispatch the Release workflow on the feature branch with blank publication inputs. Select `test_containers` to test the exact Conda artifact inside Docker and its SIF conversion. Download `release-assets`, `container-assets`, and the security reports from Actions for review. Core integration uses explicit 2 GB task reservations, a 4 GB memory budget, and two CPUs so the small fixture fits CI runners. These limits are not production metagenome sizing recommendations. + +The Conda package includes the Nextflow controller, reporting assets, test data, MAGSplitter and Camelot. Helper sources and SHA256 checksums are pinned in `requirements-workflow.txt` and fetched during the package build. The Quay image installs that same artifact; users do not install helpers separately. Source installations resolve the same pinned helpers through MP’s Python package metadata. The public build gate exercises the core K12 workflow; **passing that gate does not certify `analysis_wf` with MAGs, Slurm, nested Apptainer, or licensed Pathway Tools**. Those require the tester checks. A container containing Apptainer does not guarantee the host permits nested container execution. Test licensed inference with the recommended host Conda installation and your own image first. + +## Zenodo and software citation + +`CITATION.cff` supplies the application note's nine authors in manuscript order, their explicit manuscript affiliations, the software title, repository, and license. Release builds reject missing citation metadata, duplicate or incomplete author entries, mismatched citation versions, and an overriding `.zenodo.json`. Confirm any later author or affiliation changes against the manuscript before publication. No DOI or release date has been invented. `prepare` adds/updates the selected software version. We maintain one citation metadata file; Zenodo prioritizes `.zenodo.json` over CFF if both exist, so do not add a conflicting second file. See [Zenodo's supported metadata](https://help.zenodo.org/docs/github/describe-software/). + +Zenodo deposits are **manual and separate from GitHub releases**. In +[Zenodo's GitHub settings](https://zenodo.org/account/settings/github/), turn +**off** the switches for `hallamlab/MetaPathways` and `hallamlab/SCARAB`. +This is an account-side step: repository workflow edits cannot disable an +existing Zenodo webhook. Until it is disabled, publishing a GitHub release can +still create a Zenodo record. Neither workflow calls the Zenodo API or needs +a Zenodo token. + +When a release is approved for archiving, download its tested source archive +and checksums, then upload it through Zenodo's dashboard. For an existing +software record, use **New version** to preserve the version relationship and +concept DOI; do not create an unrelated record for each release. Review the +version, repository/tag link, license, creators, affiliations, and ORCIDs in the +draft before clicking **Publish**. Keep drafts unpublished while testing. + +Do not infer the software author list or affiliations from GitHub contributor +accounts. Use the reviewed `CITATION.cff` and confirm any affiliations with the +authors. Older tags without citation metadata may have been archived with +GitHub-derived creator details. Fix creator names or affiliations on an +existing published record using **Edit**; a metadata correction does not need +a new software version. See [editing published records](https://help.zenodo.org/docs/deposit/manage-records/). + +## Account-side release checklist + +| Service | Repository preparation | Maintainer verification before publishing | +| --- | --- | --- | +| GitHub | PR smoke checks, manual package/container validation, tag workflow | Protect the target branch, require smoke checks and tester approval; choose a new version | +| Anaconda.org | Recipe, installed-package integration, checksums, RC/main labels | `ANACONDA_API_TOKEN` has upload access to `hallamlab`; manual `publish_anaconda` selection | +| Quay | Build from validated package, Docker test, SIF conversion, vulnerability gate | Repository exists; robot has Write permission; `QUAY_USERNAME`, `QUAY_PASSWORD`, manual `publish_quay` selection | +| Zenodo | CFF metadata and documented release process | Automatic GitHub integration disabled; manually reviewed deposit and DOI verified | + +Secret values must never be committed or printed. Repository files alone cannot verify the current account permissions or secret configuration. A local source build is not proof of a successful Conda solve, registry upload, or Zenodo deposit. + +```{include} includes/cami-references.md +``` diff --git a/docs/reports-reference.md b/docs/reports-reference.md new file mode 100644 index 0000000..088f037 --- /dev/null +++ b/docs/reports-reference.md @@ -0,0 +1,44 @@ +# Reports and the EDA portal + +Successful `run`, `mag_split` and `ptools` commands refresh the report automatically. When a sample belongs to an existing parent report, the parent report is refreshed so all samples remain accessible. Report indexing runs on the controller after Nextflow completes, uses SQLite on disk, and does not repeat annotation, mapping or pathway inference. + +For existing or partial results, build a report independently: + +```bash +metapathways report -o results +``` + +`-o` can be one sample output directory or a parent containing sample directories as immediate children. Reports are saved under: + +```text +results/reports/ +├── MP_run_report.html # Accounting, import notes, files and execution links +├── EDA_portal.html # Search and export interface +├── results.sqlite # Relational data for all included samples +├── schema.json # Columns, row counts, keys and relationships +├── output_inventory.tsv # Paths, sizes and modification times +├── portal.js +└── portal.css +``` + +Open `MP_run_report.html` directly for navigation. To search, subset and export: + +```bash +metapathways report -o results --serve --no-rebuild +``` + +This opens a browser and serves the existing snapshot on loopback only. Omit `--no-rebuild` to refresh changed results first. `--no-browser` prints the URL; `--port 8765` chooses a fixed local port. Ctrl-C stops the server. All assets are local, and the explorer makes no third-party requests. Copy the complete output directory to a workstation to retain file links; no cloud upload is involved. + +Protein taxonomy supports SwissProt (including the small test database), UniRef and eggNOG. Each annotation row reports the taxon of its own reference hit in `taxonomy`; `lca_taxonomy` is computed from score-qualified hits in that same database, with support counted independently for each database. The primary ORF row follows its selected annotation's `reference_db` and target. No database priority or cross-database fallback is used. These describe reference-hit evidence, not a definitive organism assignment for the query. Unsupported databases report `Not computed`; missing or unknown taxon IDs in supported databases report `Unclassified`. An actual LCA of `root` remains `root`. SILVA rRNA taxonomy is separate. Existing outputs need annotation-table regeneration to gain these fields. + +The workflow in the portal is: + +1. Choose a table: ORFs/taxonomy, functional annotations, pathways, pathway genes, MAG membership, abundance or file inventory. +2. Search text, add column filters, or filter ORF-based tables by linked EC/reaction, reference database and pathway/entity identifiers. +3. Follow a row's related-result buttons. Sample, pathway and entity keys remain attached, avoiding cross-sample collisions. +4. Select export columns, apply filters, and export **all matching rows** as CSV. Pagination does not truncate an export. +5. Save the query definition or bookmark its URL to record your selection. + +The portal is a table explorer: it adds no biological plots or interpretation. The [schema guide](results-schema.md) explains row units, relation keys, safe joins and the distinction between missing output and biological absence. Large joins remain in SQLite; CSV is streamed rather than assembled entirely in browser memory. Broad scans can still take time. Each query has a two-minute execution budget, and at most four queries execute concurrently. + +The full inventory links remaining RNA, sequence, alignment, statistics and raw PGDB products. The relational views cover the recognized formats documented in the schema guide; an arbitrary legacy table is not silently assumed to fit that schema. Missing primary annotations leave explicit placeholder ORFs and import notes; malformed recognized tables stop the rebuild and preserve the previous database. diff --git a/docs/reports-tutorial.md b/docs/reports-tutorial.md new file mode 100644 index 0000000..1ac6652 --- /dev/null +++ b/docs/reports-tutorial.md @@ -0,0 +1,88 @@ +# Explore results, follow links and export tables + +[Home](index.md) · [Exact schema and SQL](results-schema.md) · [Benchmark statistics](benchmarking.md) + +## Open the report or start the explorer + +The report and explorer serve different purposes. `MP_run_report.html` provides accounting and navigation to results/logs. `EDA_portal.html` queries the SQLite index so you can search, filter, follow related records and export CSVs. The explorer requires its local Python server; double-clicking the explorer HTML alone does not start that server. + +After a completed workflow: + +```bash +metapathways report -o /path/to/analysis --serve --no-rebuild +``` + +Keep the terminal open. MP prints a local address and normally opens your browser. Ctrl-C stops the server. To refresh the index after results change, omit `--no-rebuild`. Report generation reads existing output; it does not rerun annotation or infer missing pathways. + +Use the parent directory containing sample folders to combine samples in one report. Using a single sample directory limits that report to the sample. Report files live under the selected root's `reports/` directory. + +## View a remote report through SSH + +On the **remote server**, activate the MP environment and run: + +```bash +metapathways report -o /remote/path/to/analysis \ + --serve --no-rebuild --no-browser --port 8765 +``` + +On **your local computer**, in a second terminal, replacing `user@server` with the same SSH destination you normally use: + +```bash +ssh -N -L 8765:127.0.0.1:8765 user@server +``` + +The tunnel command may appear idle; that is normal. Leave both terminals running. Open these addresses in your local browser: + +- Explorer: +- Run report: + +`-L` forwards the local port to the server's loopback port, and `-N` asks SSH to provide forwarding without a remote shell. Use your site's SSH host alias or jump-host configuration when required. MP listens only on loopback; you do not need to expose the server publicly or open a firewall port. + +If local port 8765 is busy, forward another local port with `-L 8766:127.0.0.1:8765`, then browse to `localhost:8766`. If the remote port is busy, choose a different `--port` and update the tunnel's destination port too. A page that loads but cannot query may be an old static copy or an HTML file opened without the server. + +## Start with accounting + +Open the run report before interpreting tables. Review which samples and products were found, source/import notes, and task execution records. Optional MAG failures can coexist with overall workflow success. A file from an earlier successful attempt can coexist with a failed newer attempt; output availability and latest task status answer different questions. + +Check a few known sample IDs and original contig IDs. If expected tables are empty, inspect the inventory and task logs before interpreting zeros. Missing output does not mean the organism lacks a function. + +## Follow one annotation through the results + +1. Choose **ORFs and taxonomy** and filter to a sample. +2. Search for an ORF ID or product term you recognize. +3. Follow its related annotation records to inspect database accessions and EC/reaction assignments. +4. Follow pathway associations, keeping both sample and entity IDs selected. +5. Export the filtered table with identifying columns included. + +A primary ORF annotation is not identical to all database hits. **Functional annotations** retains reference records; a gene can have more than one. **Pathway genes** contains pathway/ORF associations, so a gene in several pathways appears several times. The [schema](results-schema.md) describes each row unit and join key. + +## Select useful subsets + +| Question | Starting table and filters | +| --- | --- | +| Which SwissProt annotations mention kinase in this sample? | Functional annotations: sample, reference database, product text | +| Which genes support a particular MAG pathway? | Pathways: sample, entity, pathway ID; follow Pathway genes | +| What are all reported genes on contigs assigned to a MAG? | MAG ORFs: sample and entity | +| Which genes were supplied to MAG pathway inference? | MAG input genes: sample and entity | +| What abundance measurements exist for an ORF? | Abundance: sample, feature type, feature ID, measurement | +| Where is the raw GFF, RNA output, BAM or PGDB archive? | File inventory / run-report output links | + +**MAG ORFs** uses the full contig map. **MAG input genes** is the smaller set selected for Pathway Tools; it is not the total gene content of the bin. Similarly, input genome-bin counts need not equal successful PGDB counts. + +The portal provides text searches and column/related-result filters, not biological plotting. Choose the columns needed for downstream work and export all matching rows as CSV. Pagination limits what is displayed, not the full filtered export. Retain the query definition or bookmarked URL and record the report snapshot used for an analysis. + +## Avoid accidental double counting + +Always keep `sample_id`. Keep `entity_id` for pathways and genome results. The same ORF-looking identifier can occur in different samples, and the same pathway can occur in the community and many bins. + +Do not sum abundance after joining each gene to all its annotations and all its pathways: many-to-many joins multiply rows. Decide whether you need unique genes, annotation records or pathway associations before aggregating. Pathway scores are reported inference values, not universal probabilities of biological truth. See [safe SQL examples](results-schema.md#sql-with-explicit-joins). + +## Share or archive a result + +Filtered CSVs can be shared independently. To retain the explorer and original-file navigation, preserve the complete output root, including its `reports/`, sample results and logs, and serve it with a compatible MP installation. `results.sqlite` and `schema.json` also support direct read-only analysis in Python/R/SQLite. + +Not every raw product becomes a relational table: the inventory links additional sequences, alignments, RNA reports and PGDB products. Import notes identify missing or unsupported sources. The portal does not silently infer missing data, recalculate abundance or validate historical mapping provenance. + +### Starting at the sample level + +A fresh explorer URL opens **Samples**. Use a sample row's related-results buttons to inspect its ORFs, then follow annotations, pathways and abundance. Saved query URLs retain their selected table and filters. The footer shows the recorded run version, sample count, command, executor and status when available, plus the report update time and a GitHub link for bug reports and feature requests. For older outputs without a recorded run version, the footer labels the report-builder version instead. Diagnostic details remain available in the Import notes and Execution history tables. diff --git a/docs/reproducibility.md b/docs/reproducibility.md new file mode 100644 index 0000000..0dd1d10 --- /dev/null +++ b/docs/reproducibility.md @@ -0,0 +1,62 @@ +# Reproducibility and validation + +[User guide](index.md) · [Release process](releasing.md) + +## Record the actual revision and inputs + +MetaPathways is distributed under the {download}`MIT license <../LICENSE>`; bundled third-party sources retain their notices. GitHub tags, Conda packages and container images are separate publications. A tag alone does not update the other distributions. Record the source Git commit, package build or immutable container digest used for each analysis. + +Preserve: + +- Assembly/read identifiers, layout (single, paired or interleaved), checksums and sample naming. +- Original contig-to-MAG maps, including the binning version/method and QC decisions. +- MP commands, workflow parameters, tool versions and environment exports. +- Reference releases, acquisition dates and file checksums; public downloads can change. +- The Pathway Tools installer/image checksum, version and taxonomic-pruning setting. +- Final outputs, run logs, Nextflow trace/report/timeline and report `sources`/`schema.json`. + +Example environment records: + +```bash +mamba list --explicit > conda-explicit.txt +metapathways version > metapathways-version.txt +``` + +For a source installation, also record `git rev-parse HEAD` from the checkout and `python -m pip freeze` from its activated environment. + +The environment specification in `docker/conda_base.yml` is not an exact lock. MAGSplitter and Camelot revisions are pinned in `requirements-workflow.txt`; the release package includes both helpers from those exact sources. Record the MP artifact checksum along with the environment. The source revision's setup checks used Nextflow 26.04.6, Python 3.11 and Apptainer 1.5.4, with Pathway Tools 29.5 startup validation. This does not establish end-to-end biological equivalence for the orchestration migration. + +## Installation example versus biological validation + +`prepare_test -o WORKSPACE` copies the bundled three-sample inputs and reference FASTAs. From that workspace, `build_db --test -d MPDB` formats the fixtures and downloads current ExPASy/NCBI support records. The workflow test is neither offline nor fully pinned; record the downloaded reference dates. Its default workflow skips Pathway Tools. + +The test suite includes synthetic/unit checks for read-layout arguments, task ordering/resources, resume/cleanup, container isolation interfaces, report joins and export behavior. Real synthetic Nextflow tasks exercise scheduling and logging; concurrent Pathway Tools startup checks exercise private container state. These checks do not substitute for real annotations, PGDB construction, or execution on an actual Slurm cluster. The small test walkthrough and any user-run benchmarks must be evaluated through their retained results and task records. Do not infer full biological validation from unit-test success. See the [test protocol](test.md) and [benchmark measurement guide](benchmarking.md). + +Run the software checks from the checkout: + +```bash +python -m unittest discover -s tests -p 'test_*.py' +python -m unittest discover -s tests/release -p 'test_*.py' +python scripts/check_docs.py +``` + +The report HTTP tests require permission to bind a temporary loopback socket. Biological integration checks require the full Conda environment, appropriate references and an explicit decision to run them. Reports created from existing outputs perform indexing only. + +## Historical benchmark provenance + +The repository's previous reproducibility notes describe the [v1 preprint](https://doi.org/10.1101/2024.06.04.597460), section 3.1, as reporting 15 CAMI II human-microbiome metagenomes ([Meyer et al., 2022](#cami-references)) and 622 MAGs from five body sites, SwissProt 2023_05, MetaCyc 27.1, MetaBAT2 2.15 and Pathway Tools 27.0, with 16 cores on Ubuntu 20.04.6. These are historical settings, not the defaults or a newly verified reproduction of this source revision. + +The [CAMI2 study](https://doi.org/10.1038/s41592-022-01431-4) identifies the human collection at [PUBLISSO, DOI 10.4126/FRL01-006425518](https://doi.org/10.4126/FRL01-006425518). A collection DOI does not identify the exact assembly/read/bin selections used by a particular benchmark. The K12 installation fixture does not substitute for those records. + +A fresh benchmark needs a complete sample/input manifest, contig maps, reference/software versions, resource limits, commands and retained task traces. Historical output affected by the paired-read mapping bug must have mapping corrected before abundance is used. Expected MAG Pathway Tools failures should be recorded as such; they are not evidence that no pathways exist. + +Only measured trace fields should support resource plots. A report rebuilt over old outputs cannot recreate missing peak RAM, CPU time or stage runtime. Reused tasks, failed tasks, incomplete output and successful fresh tasks must be distinguished when making supplementary tables or benchmark claims. Report database row counts are data-accounting units, not performance measurements. + +## Documentation and release records + +Documentation is maintained in this README-led GitHub tree. CI checks local Markdown links and generated CLI reference freshness; no Sphinx/Read the Docs deployment is required. Git history retains the former RST documentation. + +No new version DOI or archive has been assigned by this feature work. Publish and archive a validated release before claiming a corresponding version DOI in a manuscript. Use [maintainer release instructions](releasing.md) for package/container publication. + +```{include} includes/cami-references.md +``` diff --git a/docs/requirements.txt b/docs/requirements.txt index 4fedba9..1bea3d7 100644 --- a/docs/requirements.txt +++ b/docs/requirements.txt @@ -1 +1,4 @@ -sphinx_rtd_theme==2.0.0 +Sphinx==8.2.3 +myst-parser==4.0.1 +sphinx-rtd-theme==3.0.2 +sphinxcontrib-mermaid==2.0.2 diff --git a/docs/resources.md b/docs/resources.md new file mode 100644 index 0000000..1e9be50 --- /dev/null +++ b/docs/resources.md @@ -0,0 +1,45 @@ +# Resources and Slurm + +## Local execution is the default + +| Setting | Default / meaning | +| --- | --- | +| `run -t / --threads` | Eight CPUs per threaded tool, capped by total CPU budget | +| Threaded tools | pProdigal, barrnap, ptRNAscan, FAST, BLAST, CoverM, featureCounts, and samtools sort | +| Serial stages | One CPU each, including Pathway Tools and Python parsers; native math pools are capped too | +| `--max_cpus` | CPUs available to the process, including affinity/cgroup limits | +| `--memory` | Per-task reservation: normally 16 GB; standalone `ptools` defaults to 4 GB | +| `analysis_wf --ptools_memory` | Optional PGDB-only override; otherwise inherits `--memory` | +| `--max_memory` | Currently available host memory, bounded by cgroups | +| `--max_tasks` | Additional concurrency cap; CPU/memory reservations still apply | +| `build_db -t` | Legacy total CPU budget; omitted means available CPUs | +| `build_pt -t` | Image-compression CPUs | + +For example, permit a total of 16 CPUs while allowing eight threads per capable tool: + +```bash +metapathways run -i sample.fasta -o results -d /data/MPDB \ + -t 8 --max_cpus 16 --max_memory '64 GB' +``` + +Independent reference searches may run together, and Nextflow schedules as many eligible tasks as fit their reservations. Memory reservations are scheduling requests, not measured peaks; local execution does not impose a container memory ceiling. Each MP invocation has its own budget. `--threads` controls each threaded task, while `--max_cpus` controls the total: for example, a 16-CPU budget can fit two eight-CPU jobs, one eight-CPU job plus eight single-CPU jobs, or sixteen single-CPU jobs when dependencies and memory permit. ptRNAscan uses that many single-threaded tRNAscan-SE workers; samtools receives one fewer additional thread so its main thread fits the reservation. Thread counts are upper bounds: small inputs and serial portions of a tool may use fewer CPUs. These settings apply when planning a new invocation; they do not resize running jobs. + +## Submit from a Slurm headnode + +MP explicitly requests one node (`--nodes=1`) for every Slurm task. Threaded tools use their allocated CPUs on that node; concurrency across nodes comes from separate jobs. No node-count flag is needed in the MP command. + +Run from an authenticated login/head node where `sbatch`, `squeue` and `scancel` are available: + +```bash +metapathways run -i /shared/sample.fasta -o /shared/results -d /shared/MPDB \ + --executor slurm --account my_project --partition compute \ + -t 8 --memory '64 GB' --max_tasks 100 --time_limit 24h +``` + +MP uses your existing Slurm identity; it does not take a password or SSH private key. `--partition` is optional: omitted means the cluster default. Run `sinfo` to list partitions; these are named node groups/queues with access and time limits. `--qos` and `--reservation` are optional. Inputs, outputs, references, work/cache paths, the MP installation and its environment must have the same absolute paths on compute nodes. Pathway Tools on Slurm requires a SIF. The report's HTML explorer runs locally; it is not a Slurm service. + +Slurm defaults to at most four submitted jobs, six submissions per minute and 24 hours per task. `--max_tasks 100` directly permits up to 100 queued/running jobs; Slurm decides actual running concurrency. There are no implicit aggregate CPU or memory caps on Slurm. If you explicitly provide `--max_cpus` or `--max_memory`, MP conservatively reduces the job cap using the largest task request. `--submit_rate` independently throttles submissions; for example, `--submit_rate 60` permits one per second. There are no automatic task retries. MP uses an explicit Nextflow configuration so ambient profiles do not silently override these limits. + +The basic interface is the same locally and on HPC: `--threads 8 --memory '64 GB' --max_tasks 100`. Serial tasks still request one CPU. In `analysis_wf`, the memory request also applies to PGDBs unless overridden with `--ptools_memory`. Locally, detected available CPUs and memory automatically limit execution; on Slurm, the scheduler determines capacity. Aggregate maxima are optional additional controls, not numbers the user must calculate. + +Live Slurm execution still needs validation at your site. Configuration and synthetic local scheduling tests do not establish compatibility with every cluster's authentication, filesystem or resource policies. diff --git a/docs/results-schema.md b/docs/results-schema.md new file mode 100644 index 0000000..e3ef06e --- /dev/null +++ b/docs/results-schema.md @@ -0,0 +1,135 @@ +# Results schema and EDA portal + +[User guide](reports-reference.md#reports-and-the-eda-portal) · [CLI reference](cli-reference.md) + +For step-by-step browsing and SSH setup, start with the [explorer tutorial](reports-tutorial.md). This page defines the exact table meanings and relationships. + +## What the report does + +`metapathways report -o OUTPUT` reads existing outputs and writes `OUTPUT/reports/`. It never launches annotation, read mapping, MAG splitting or Pathway Tools. Successful analysis commands also refresh the report. The report follows ASPIRE/BASINS' accounting-and-navigation approach: sample inventory, available products, import limitations, source records and links to execution diagnostics. It adds no biological interpretation or new plots. + +The HTML report is readable directly from disk. The EDA portal's queries and CSV exports require `metapathways report -o OUTPUT --serve --no-rebuild`. A loopback-only server reads `results.sqlite`; the browser receives one page at a time. HTML, CSS and JavaScript are packaged with MP. There are no CDN dependencies or third-party requests. + +The schema version is recorded in `schema.json`, together with generation time, source root, table/view columns, row counts, primary keys and foreign keys. SQLite is also usable from Python, R or the `sqlite3` CLI. All paths in the inventory and source tables are relative to the selected output root. + +## Relationships + +[![Results schema](assets/diagrams/results-schema.svg)](assets/diagrams/results-schema.svg) + +[Zoom diagram](assets/diagrams/results-schema.svg) · [Mermaid source](diagrams/results-schema.mmd) + +Always scope contig and ORF identifiers by `sample_id`. Scope a pathway by `(sample_id, entity_id, pathway_id)`: the same pathway in a community and a MAG is two inference records. `community` is the reserved community entity identifier. MAG identifiers come from output directories or the preserved contig map. As in MAGSplitter, periods in original MAG names become underscores in entity IDs; `contig_mags.original_mag_id` retains the supplied identifier. Sources use database-local integer `source_id` values; these IDs may change on rebuild and are not global identifiers. + +## Tables and their row units + +| Table | One row represents | Main keys and provenance | +| --- | --- | --- | +| `samples` | A sample output directory | `sample_id`; output path relative to the report root | +| `contigs` | A retained contig | `(sample_id, contig_id)`; original ID and length from `preprocessed/*.mapping.txt` | +| `orfs` | An ORF | `(sample_id, orf_id)`; contig, coordinates, strand, primary target/product and taxonomy from `*.functional_and_taxonomic_table.txt` | +| `annotations` | One reference annotation record | `annotation_id`; sample/ORF, database, accession, product, score, original EC/reaction strings, `source_id` | +| `annotation_terms` | One term on one annotation | `(annotation_id, term_type, term)`; pipe-separated EC/reaction values are split without creating EC×reaction combinations | +| `entities` | A community, MAG or unbinned entity | `(sample_id, entity_id)`; type, pathway output availability, latest recorded Pathway Tools task status | +| `contig_mags` | An explicit contig-to-MAG assignment | `(sample_id, contig_id, entity_id)`; source is preserved `magsplitter/contig_to_mag.tsv` | +| `entity_orfs` | A gene explicitly present in a MAG Pathway Tools input | `(sample_id, entity_id, orf_id)` from `magsplitter/results/*/0.pf`; **not full MAG gene membership** | +| `orf_groups` | A representative/member association | `(sample_id, representative_orf_id, member_orf_id)` from `ptools/orf_map.txt`; includes the representative itself | +| `pathways` | An entity-specific pathway inference | Composite pathway key; common name, reported score/reaction counts/ORF count and source | +| `pathway_orfs` | An explicitly reported pathway/ORF association | Composite pathway key plus `orf_id`; duplicates in an ORF list are collapsed | +| `abundance_explorer` | One feature per sample | Wide read-abundance table: `length_bp`, `count`, `mean_coverage`, `coverage_variance`, `trimmed_mean_coverage`, `rpkm`, `tpm`; ORF/contig links and source | +| `abundance` | One original measurement for one feature | `(sample_id, feature_type, feature_id, measurement)`; numeric value and source | +| `execution` | A retained task invocation | Run identifier, command, label, outcome, duration, error and summary file; contains reruns/cache hits too | +| `sources` | A parsed source file | Relative path, size, SHA-256 and role | +| `files` | An inventoried output file | Relative path, size and modification time; large raw files are not all checksummed | +| `issues` | An import limitation or discrepancy | Sample, source and explanation | + +Protein taxonomy is preserved in `*.annotation_taxonomy.tsv`, keyed by ORF ID, reference database and target accession. `taxid` and `taxonomy` describe that hit; `lca_taxonomy` summarizes score-qualified hits to that ORF **within the same database**, using independent per-database minimum-support counts. SwissProt uses its `OX` NCBI taxon ID, UniRef its `TaxID`, and eggNOG the numeric prefix of the target identifier. Unknown/missing IDs produce `Unclassified`; unsupported databases produce `Not computed`. Hit taxa do not by themselves identify the query organism. The primary ORF table includes `reference_db` and `lca_taxonomy` for its selected annotation. No taxonomy is borrowed from another reference database. + +Report schema version 2 adds the `annotation_taxonomy` table and joins functional rows by sample, ORF, database **and target**. Older results without this file show `Not computed` for functional-row taxonomy instead of borrowing the primary ORF taxonomy. Rebuilding the report alone does not compute missing taxonomy; regenerate annotation tables first. + +The primary annotations and the EC/reaction mapping have different meanings. The primary table contains MP's selected target/product and reported taxonomy. `annotations` retains individual database records. It uses `*.EC_RXN_map.tsv` when available, otherwise `*.1.txt`; importing both would duplicate hits. Blank repeated ORF cells in the compact `.1.txt` format are forward-filled within that file. + +ORF lengths and coordinates are copied in their source units/conventions; the importer does not recompute them. Taxonomy strings remain as reported, rather than being converted into assumed ranks. Reference scores are not relabeled as universal confidence probabilities. + +Abundance retains the original measurement names, including read-file labels in CoverM's contig table. ORF `Count`, `RPKM` and `TPM` are separate measurements. The `(feature_type, sample_id, feature_id)` relation identifies a contig or ORF; SQLite cannot express this polymorphic relation as one ordinary foreign key. Abundance values are not recalculated, and historical incorrect mate mapping is not repaired by importing it. + +## Ready-made explorer views + +| Portal table / SQL view | Grain and joins | +| --- | --- | +| ORFs and taxonomy / `orf_explorer` | One ORF with original contig ID and length | +| Functional annotations / `annotation_explorer` | One reference annotation with contig, hit taxonomy and same-database LCA | +| Pathways / `pathway_explorer` | One pathway inference with entity type and count of distinct explicit ORF links | +| Pathway genes / `pathway_gene_explorer` | One pathway/ORF link with primary product, taxonomy and contig | +| MAG ORFs / `mag_orf_explorer` | All reported ORFs on explicitly mapped contigs; requires the full contig map | +| MAG input genes / `mag_gene_explorer` | Only the selected genes in MAG Pathway Tools input files | + +These views do not join all annotation hits onto all pathway memberships. Such a join creates a many-to-many expansion and makes naive counts or abundance sums wrong. The browser's related filters use `EXISTS` to select matching ORFs/annotations without multiplying rows. EC, reaction and reference-database restrictions in one related filter must match the same annotation record; pathway and entity restrictions must match the same pathway association. + +An entity-only related filter selects the explicit MAG Pathway Tools input genes. For complete MAG membership, open **MAG ORFs** and filter `entity_id`; then follow the ORF's annotation/pathway links. This distinction is deliberate when old results lack the full contig map. + +## Subsetting examples + +### Taxon and function + +1. Choose **Functional annotations**. +2. Add `taxonomy contains Bacteria` and `product contains kinase`. +3. Add `reference_db equals swissprot` if only that database is desired. +4. Select columns, click **Apply filters**, then export CSV. + +### Genes supporting a pathway in one MAG + +1. Choose **Pathways** and filter `sample_id`, `entity_id` and `pathway_id` by exact equality. +2. Click **Pathway genes** on that row. All three identifiers carry into the next query. +3. Filter further by taxonomy or product, and export the matching associations. +4. Use **ORF annotations** to see the gene's individual reference hits and EC/reaction terms. + +### All reported ORFs assigned to a MAG + +Choose **MAG ORFs** and set exact sample and entity filters. This uses the contig membership map, including ORFs absent from the selected Pathway Tools input. If the map was not preserved in old outputs, the report leaves this table empty rather than inferring membership from pathway genes. You can supply the original headerless two-column map as `SAMPLE/magsplitter/contig_to_mag.tsv` and rebuild, without rerunning biological analyses. + +### SQL with explicit joins + +```python +import sqlite3 +from pathlib import Path + +uri = Path('results/reports/results.sqlite').resolve().as_uri() + '?mode=ro' +with sqlite3.connect(uri, uri=True) as db: + rows = db.execute(''' + SELECT g.sample_id, g.entity_id, g.pathway_id, + g.orf_id, o.contig_id, o.taxonomy + FROM pathway_orfs AS g + JOIN orfs AS o USING (sample_id, orf_id) + WHERE g.sample_id = ? AND g.entity_id = ? AND g.pathway_id = ? + ''', ('sample', 'MAG_001', 'PWY-6167')).fetchall() +``` + +To count genes use `COUNT(DISTINCT orf_id)` **within a sample**, or distinct `(sample_id, orf_id)` pairs across samples. A gene participating in several pathways remains one gene. Pathway-associated ORFs are not automatically expanded through `orf_groups`: expanding a representative across MAG boundaries without checking membership would invent associations. + +## Missing, partial and historical outputs + +- Missing source tables produce import notes and empty/partially populated views. Referenced ORFs absent from the primary table are retained with `annotation_present=0` and null unknown fields. +- `pathway_status=available` means a pathway TSV exists, including a valid empty table. `unavailable` means no recognized pathway TSV was found. `last_task_status` records the latest retained Pathway Tools task outcome when known; an old output file can coexist with a failed latest attempt. +- Failed MAG inference is optional in MP, but not converted into a confident biological absence. Historical outputs without Nextflow summaries have unknown task status. +- The reported pathway ORF count is preserved separately from the count of unique parsed links. Disagreement is recorded in `issues`. +- Contig assignments missing from the retained contig map are counted in import notes; QC can remove original contigs. +- Malformed recognized tables or duplicate primary keys abort a rebuild. The prior database is retained. Input files are never edited by reporting. +- Reports are snapshots. `--no-rebuild` deliberately shows the old snapshot even if files have changed since generation. Rebuild when you want updated data. + +## Outputs that stay as source files + +The file inventory links RNA tables, FASTA/GFF/GenBank files, alignment results, run statistics, Pathway Tools flat files/archives and redundant annotation/pathway exports. Their arbitrary formats are not silently normalized into the biological core schema. The canonical pathway associations come from `*_pwy.tsv`; the denormalized `*_pwy2orf.tsv` remains accessible as its original file. + +The inventory omits hidden runtime state, report products, symlinked files/directories, and standard work/cache directories. The source importer rejects resolved paths outside the selected output tree. Report generation parses result tables and hashes its indexed source files; it does not read large sequence/aligner files merely to checksum every byte. + +## Exports and local service + +CSV export streams every matching row with the selected columns, independent of the visible page. Null values are blank. Text that would look like a spreadsheet formula is prefixed with an apostrophe; original values remain unchanged in SQLite. A query-definition JSON records filters, columns, ordering, schema version and report timestamp. The URL hash also records the query and is bookmarkable while the corresponding report snapshot remains available. + +The server binds only to `127.0.0.1`, uses a random URL prefix, checks Host/Origin, opens SQLite read-only and accepts only declared views/columns/operators. It does not execute arbitrary submitted SQL. Source-file links are restricted to the report tree and inventoried output paths. Do not expose it as a public web service. For a remote workstation, copy the results or use your site's approved SSH forwarding of a chosen loopback port. + +### Wide read-abundance table (schema version 3) + +The explorer's **Read abundance** table and its CSV export show measurements side by side, one row per sample and feature (`feature_type` is `orf` or `contig`). Filename prefixes are removed from known CoverM metric names. `count` copies ORF `Count` or contig `Read Count`; the counting conventions and normalization remain those of featureCounts and CoverM respectively, so these are not interchangeable units. `length_bp` copies the abundance file's length, rather than the annotation table's length. Coverage statistics unavailable for ORFs are blank, not zero. No measurements are recalculated. The separate **Read abundance: raw measurements** table retains original column names and values, including unrecognized measurements. Source links identify the original TSVs. Rebuild the report to update an existing explorer; mapping and annotation do not need rerunning. + +Known tRNA/rRNA abundance features are retained as linked gene records without a missing-protein-annotation warning. Missing CDS or unidentified referenced features still produce an import note. Routine abundance imports do not produce a warning. diff --git a/docs/reviewer-bundle.md b/docs/reviewer-bundle.md new file mode 100644 index 0000000..df34a6d --- /dev/null +++ b/docs/reviewer-bundle.md @@ -0,0 +1,7 @@ +--- +orphan: true +--- + +# Test bundle + +See the [test bundle documentation](test-bundle.md). diff --git a/docs/reviewer-test.md b/docs/reviewer-test.md new file mode 100644 index 0000000..5dd5c05 --- /dev/null +++ b/docs/reviewer-test.md @@ -0,0 +1,7 @@ +--- +orphan: true +--- + +# Test walkthrough + +The installation and workflow test is for everyone. Continue to the [test walkthrough](test.md). diff --git a/docs/src/conf.py b/docs/src/conf.py deleted file mode 100644 index a4b49de..0000000 --- a/docs/src/conf.py +++ /dev/null @@ -1,59 +0,0 @@ -# Configuration file for the Sphinx documentation builder. -# -# This file only contains a selection of the most common options. For a full -# list see the documentation: -# https://www.sphinx-doc.org/en/master/usage/configuration.html - -# -- Path setup -------------------------------------------------------------- - -# If extensions (or modules to document with autodoc) are in another directory, -# add these directories to sys.path here. If the directory is relative to the -# documentation root, use os.path.abspath to make it absolute, like shown here. -# -#import os -#import sys -#sys.path.insert(0, os.path.abspath('..')) - - -# -- Project information ----------------------------------------------------- - -project = 'MetaPathways' -copyright = '2024, BCB2 Developers' -author = 'BCB2' - - -# -- General configuration --------------------------------------------------- - -# Add any Sphinx extension module names here, as strings. They can be -# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom -# ones. -extensions = [ -] - -# Add any paths that contain templates here, relative to this directory. -templates_path = ['_templates'] - -# List of patterns, relative to source directory, that match files and -# directories to ignore when looking for source files. -# This pattern also affects html_static_path and html_extra_path. -exclude_patterns = ['_*', 'Thumbs.db', '.DS_Store'] - - -# -- Options for HTML output ------------------------------------------------- - -# The theme to use for HTML and HTML Help pages. See the documentation for -# a list of builtin themes. -# -html_theme = 'sphinx_rtd_theme' -# source_suffix = ['.rst', '.md'] -source_suffix = '.rst' - -# The master toctree document. -master_doc = 'index' - -# Add any paths that contain custom static files (such as style sheets) here, -# relative to this directory. They are copied after the builtin static files, -# so a file named "default.css" will overwrite the builtin "default.css". -html_static_path = ['static'] - -numfig = True diff --git a/docs/src/index.rst b/docs/src/index.rst deleted file mode 100644 index 92e6ec8..0000000 --- a/docs/src/index.rst +++ /dev/null @@ -1,31 +0,0 @@ -.. MetaPathways documentation master file, created by - sphinx-quickstart on Wed Sep 23 13:09:29 2020. - You can adapt this file completely to your liking, but it should at least - contain the root `toctree` directive. - - -MetaPathways -============ - -.. toctree:: - :maxdepth: 2 - :caption: Documentation - - overview - quick_start - install - usage - reproducibility - -.. Indices and tables -.. ================== - -.. * :ref:`genindex` -.. * :ref:`modindex` -.. * :ref:`search` - -.. Contact -.. ======= - -.. :ref:`contact` - diff --git a/docs/src/install.rst b/docs/src/install.rst deleted file mode 100644 index 9e7cc9b..0000000 --- a/docs/src/install.rst +++ /dev/null @@ -1,34 +0,0 @@ -Installation -************ - -Metapathways can be installed with conda or obtained as a container. -The fastest install is via Mamba (recommended) or by pulling down the container. - -.. note:: - - Metapathways has been built and tested on Linux OS. - Lightly tested on Mac OS. - And not currently supported on native Windows. - - -Mamba (recommended) -------------------- - -.. code-block:: bash - - mamba create -n metapathways_env -c hallamlab -c bioconda -c conda-forge metapathways - -Conda ------ - -.. code-block:: bash - - conda create -n metapathways_env -c hallamlab -c bioconda -c conda-forge metapathways - -Docker ---------- - -.. code-block:: bash - - docker pull quay.io/hallamlab/metapathways - diff --git a/docs/src/overview.rst b/docs/src/overview.rst deleted file mode 100644 index b7c8d94..0000000 --- a/docs/src/overview.rst +++ /dev/null @@ -1,108 +0,0 @@ -Overview -******** - -MetaPathways [1]_ is a meta'omic analysis pipeline for the annotation and analysis for environmental sequence information. -MetaPathways include metagenomic or metatranscriptomic sequence data in one of several file formats -(nucleotide FASTA or amino-acid FASTA). The pipeline consists of five operational stages including - -Pipeline -~~~~~~~~ - -.. figure:: static/glyph.png - :align: center - :alt: alternate text - :figclass: align-center - -MetaPathways is composed of four general stages, encompassing a number of analytical or data handling steps **(Figure 1)**: - -.. |nbsp| unicode:: 0xA0 - :trim: - -|nbsp| - -#. **Quality Control**: - Basic quality control (QC) is performed with includes filtering out sequences below a set - length threshold (default 180bp). - -#. **Feature Prediction**: - Several sequence features can be predicted on the QC'ed contigs. Open-reading frames - (ORFs) are predicted by default and (optionally) ribosomal subunits (rRNAs) and - transfer RNAs (tRNAs) can be predicted. To improve the runtime and efficiency, Prodigal - [2]_ is run through a parallel version (pProdigal) [3]_ and - tRNAscan-SE [4]_ is run using a wrapper script that allows for more - efficient multi-threading. BARRNAP [5]_ is used for the prediction of rRNAs - including: 16S, 23S, and 5S. MetaPathways provides an overlap-aware identification so - that users can make informed decisions about features when they overlap each other. - Additionally, users can define the minimum length of ORFs to keep for downstream analysis. - -#. **Functional Annotation**: - Using a seed-and-extend homology search algorithm, either BLAST [6]_ or FAST [7]_, - users can conduct searches against both functional and taxonomic (optional) databases. - Currently supported databases include: Uniprot SwissProt [8]_, Uniprot UniRef90 - [9]_, MetaCyc [10]_, CAZymes [11]_ for functional annotations, and SILVA [12]_ for - taxonomic. However, users can create custom databases for any preferred databases. - Optionally, reads can be used to calculate abundance information at both the contig - and ORF-level. - -#. **Pathway Inference**: - MetaPathways then predicts `MetaCyc pathways `_ using the - `Pathway Tools software `_ and its pathway prediction - algorithm PathoLogic [13]_, resulting in the creation of a community-level environmental - Pathway/Genome Database (ePGDB), an integrative data structure of sequences, genes, pathways, - and literature annotations for integrative interpretation. Optionally, if metagenome-assembled - genomes (MAGs) are available for the metagenome, these MAGs can be used to create - population-level ePGDBs. MetaCyc pathways are exported in a tabular format for - downstream analysis. - -Bibliography -~~~~~~~~~~~~ - -Please see the following `Zotero Library `_ for a bibliography. - -.. [1] K. M. Konwar, N. W. Hanson, A. P. Pagé, S. J. Hallam, MetaPathways: a modular - pipeline for constructing pathway/genome databases from environmental sequence information. - BMC Bioinformatics 14, 202 (2013) http://www.biomedcentral.com/1471-2105/14/202 - -.. [2] D. Hyatt, P. F. LoCascio, L. J. Hauser, and E. C. Uberbacher. Gene and - translation initiation site prediction in metagenomic sequences. Bioinformatics, 28(17):2223–2230, 2012. - -.. [3] S. Jaenicke. PProdigal: Parallelized gene prediction based on Prodigal. - https://github.com/sjaenick/pprodigal, Dec. 2023. original-date: 2019-08-10T10:51:03Z. - -.. [4] P. P. Chan and T. M. Lowe. tRNAscan-SE: Searching for tRNA genes - in genomic sequences. Methods in molecular biology (Clifton, N.J.), 1962:1–14, 2019. - -.. [5] T. Seemann. Barrnap: BAsic Rapid Ribosomal RNA Predictor. - https://github.com/tseemann/barrnap, Dec. 2023. original-date: 2013-08-03T08:25:45Z. - -.. [6] Z. Zhang, W. Miller, and D. J. Lipman. Gapped BLAST and PSIBLAST: a new - generation of protein database search programs. Nucleic Acids Research, 25(0):17, 1997. - -.. [7] D. Kim, A. S. Hahn, N. W. Hanson, K. M. Konwar, and S. J. Hallam. - FAST: Fast annotation with synchronized threads. In 2016 IEEE Conference on Computational - Intelligence in Bioinformatics and Computational Biology (CIBCB), pages 1–8, 2016. - -.. [8] Boutet, Emmanuel, Damien Lieberherr, Michael Tognolli, Michel Schneider, - and Amos Bairoch. "UniProtKB/Swiss-Prot: the manually annotated section of the UniProt - KnowledgeBase." In Plant bioinformatics: methods and protocols, pp. 89-112. Totowa, NJ: - Humana Press, 2007. - -.. [9] Suzek, B.E., Wang, Y., Huang, H., McGarvey, P.B., Wu, C.H. and UniProt Consortium, - 2015. UniRef clusters: a comprehensive and scalable alternative for improving sequence - similarity searches. Bioinformatics, 31(6), pp.926-932. - -.. [10] Caspi, R., Altman, T., Billington, R., Dreher, K., Foerster, H., Fulcher, C.A., - Holland, T.A., Keseler, I.M., Kothari, A., Kubo, A. and Krummenacker, M., 2014. The MetaCyc - database of metabolic pathways and enzymes and the BioCyc collection of Pathway/Genome - Databases. Nucleic acids research, 42(D1), pp.D459-D471. - -.. [11] Garron, M.L. and Henrissat, B., 2019. The continuing expansion of CAZymes and their - families. Current opinion in chemical biology, 53, pp.82-87. - -.. [12] Pruesse, E., Quast, C., Knittel, K., Fuchs, B.M., Ludwig, W., Peplies, J. and Glöckner, - F.O., 2007. SILVA: a comprehensive online resource for quality checked and aligned ribosomal - RNA sequence data compatible with ARB. Nucleic acids research, 35(21), pp.7188-7196. - -.. [13] P. D. Karp, M. Latendresse, R. Caspi, The pathway tools pathway prediction algorithm. - Stand Genomic Sci 5, 424–429 (2011). - diff --git a/docs/src/quick_start.rst b/docs/src/quick_start.rst deleted file mode 100644 index 828234b..0000000 --- a/docs/src/quick_start.rst +++ /dev/null @@ -1,58 +0,0 @@ -Quick Start (TLDR;) -******************* -Are you a reviewer or just need to test this out quickly? -The commands below will allow you to install and test MP-3.5 quickly and efficiently. -Sorry you still need Conda/Mamba. - -.. note:: - - If you run the last two commands, make sure to replace the ${variables} with real values. - -Try it with Mamba (recommended) -=============================== - -.. code-block:: bash - - # install metapathways with Mamba - mamba create -n metapathways_env -c hallamlab -c bioconda -c conda-forge metapathways - mamba activate metapathways_env - - # test the install - metapathways build_db --test - metapathways run --test # results in `${CWD}/test` - - ### Try your own data! ### - - # build the minimal reference DB - metapathways build_db -d ${path/to/MPDB} --func swissprot -a fast - - # run minimal DB on your personal data - metapathways run -i ${your_metagenome.fa} -o ${path/to/output_dir} -d ${path/to/MPDB} - -.. note:: - - The build_db test does not create a DB that is to be used for downstream DB builds. - For the full reference build the use is to provide a location with acceptable storage - capacity and accessibility. - -Try it with Docker -===================== - -.. code-block:: bash - - # pull down with docker - docker pull quay.io/hallamlab/metapathways - docker run -it quay.io/hallamlab/metapathways /bin/bash - - # test the install (within docker) - metapathways build_db --test - metapathways run --test # results in `${CWD}/test` - - ### Try your own data! ### - - # build the minimal reference DB - metapathways build_db -d ${path/to/MPDB} --func swissprot -a fast - - # run minimal DB on your personal data - metapathways run -i ${your_metagenome.fa} -o ${path/to/output_dir} -d ${path/to/MPDB} - diff --git a/docs/src/reproducibility.rst b/docs/src/reproducibility.rst deleted file mode 100644 index ebf257d..0000000 --- a/docs/src/reproducibility.rst +++ /dev/null @@ -1,133 +0,0 @@ -Reproducibility -*************** - -Source and license -================== - -The canonical v3.5 source is `hallamlab/MetaPathways -`_. MetaPathways is distributed under -the MIT license; bundled third-party source retains its own license notices. -The historical GitHub codebase is separate from v3.5. - -The Conda package and Quay container are distributed separately. A GitHub tag -alone does not update either distribution. Record the package version or -container digest used for an analysis, alongside the Git commit when installing -from source. - -Installation test -================= - -In a writable environment with the dependencies from ``docker/conda_base.yml`` -and this checkout installed, run: - -.. code-block:: bash - - metapathways build_db --test - metapathways run --test - -The package includes the compressed K12 nucleotide FASTA, paired FASTQ reads, -and small SwissProt/SILVA sequence fixtures in ``metapathways/regtests``. -The first command formats these fixtures and downloads the current ExPASy enzyme -records and NCBI taxonomy. It needs internet access and writes into the installed -package's test database directory. It is not an offline or fully pinned benchmark. -The second command writes results to ``./test/``. Inspect command output, -``metapathways_steps_log.txt`` and ``errors_warnings_log.txt`` in the sample output -directory; do not rely on the process exit code alone. - -Keep generated databases and outputs outside version control. Preserve their -checksums and download dates when recording an analysis. The raw sequence fixtures -are versioned with the source; they must not be removed as generated data. - -Supported inputs and outputs -============================ - -The current ``run`` CLI accepts nucleotide FASTA (``--input_format fasta``) or -amino-acid FASTA (``--input_format fasta-amino``). Compressed FASTA is used by the -included test. FASTQ reads supplied with ``-1``/``-2`` or ``--interleaved`` enable -abundance calculations. GFF and GenBank are outputs, not supported primary input -formats of this CLI. - -Outputs under each sample directory include: - -* ``preprocessed/`` and ``orf_prediction/``: processed nucleotide/protein sequences - and predicted features. -* ``genbank/``: annotated GFF and GenBank files when their stages are enabled. -* ``results/annotation_table/``: tab-separated functional/taxonomic annotations. -* ``results/rpkm/``: abundance/count tables when read mapping is enabled. -* ``results/rRNA/`` and ``results/tRNA/``: RNA annotations and statistics. -* ``run_statistics/`` and the sample log files: processing counts and step status. -* ``ptools/``: Pathway Tools input files. Creating an ePGDB is a separate - ``metapathways ptools`` command requiring a licensed Pathway Tools installation - and the optional PGDB dependencies. - -Dependency versions -=================== - -``docker/conda_base.yml`` specifies the runtime environment; optional development, -MAG splitting and PGDB dependencies are listed in ``docker/conda_dev.yml`` and -``conda_recipe/meta_template.yaml``. These environment specifications are not -complete version locks. The migration validation environment on Linux x86-64 used: - -.. list-table:: Validation environment - :header-rows: 1 - - * - Dependency - - Version - * - Python - - 3.10.19 - * - Snakemake - - 7.32.4 - * - Prodigal / PProdigal - - 2.6.3 / 1.0.1 - * - BLAST+ - - 2.17.0 - * - BWA / SAMtools - - 0.7.19 / 1.22.1 - * - CoverM - - 0.7.0 - * - tRNAscan-SE / Barrnap - - 2.0.12 / 0.9 - * - pandas / pybedtools / pysam / pyfastx - - 2.3.3 / 0.12.0 / 0.23.3 / 2.2.0 - -FAST and metacount executables are included in the package. Their native source, -Makefiles, and relevant third-party tests are retained under ``extensions/``. -The bundled executables target Linux; native Windows is not supported. Record an -explicit environment export and checksums of reference files for reproducible -analyses. - -Manuscript benchmark provenance -=============================== - -The `v1 preprint `_, section 3.1, -reports 15 CAMI2 Human Microbiome metagenomes and 622 MAGs across five body sites. -It specifies SwissProt release 2023_05, MetaCyc 27.1, MetaBAT2 2.15, and Pathway -Tools 27.0, with 16 cores on Ubuntu 20.04.6. These historical reference/software -versions differ from the moving downloads used by the installation test. - -The `CAMI2 study `_ identifies the -human dataset at `PUBLISSO, DOI 10.4126/FRL01-006425518 -`_. This identifies the source collection; -it does not establish the exact input files used for the MetaPathways results. - -Before claiming reproduction of the manuscript benchmark, the authors still need -to provide the exact 15-sample input/assembly manifest with checksums, commands -for preparing the 622 MAGs and contig mappings, reference acquisition/version -records, the pipeline revision and environment, benchmark commands, and the -result tables underlying the reported metrics. These artifacts were not found in -the current source checkout. The bundled K12 example does not substitute for -this benchmark archive. - -No Zenodo DOI has been assigned by this migration. After the validated GitHub -release is published and archived through Zenodo, cite the actual version DOI -in the manuscript and data-availability statement. - -Automated checks -================ - -The GitHub Actions smoke workflow builds the source/wheel distributions, checks -installed CLI startup on Python 3.10, and builds documentation. The full -bioinformatics example above is a separate integration check requiring the -Conda dependencies and public database downloads. It is not run by the lightweight -smoke workflow. The old Python 3.6--3.8 CircleCI/tox configuration was retired -because it referenced a removed test suite and requirements file. diff --git a/docs/src/static/glyph.png b/docs/src/static/glyph.png deleted file mode 100644 index a8aab08..0000000 Binary files a/docs/src/static/glyph.png and /dev/null differ diff --git a/docs/src/usage.rst b/docs/src/usage.rst deleted file mode 100644 index 6146477..0000000 --- a/docs/src/usage.rst +++ /dev/null @@ -1,316 +0,0 @@ -Full Usage -********** - -Build Reference Databases -------------------------- - -Metapathways requires reference databases to perform functional/taxonomic annotation. -Below provides the commands for building currently supported database. - -.. note:: - - `uniref90` and `uniref50` are the largest databases at ~270 GB and ~30 GB respectively after set up. - Others are less than ~10GB each. - - `SILVA` is the only supported taxonomic reference database at this time - and therefore is built automatically along with functional references. - -Minimal Usage: -~~~~~~~~~~~~~~ - -.. code-block:: bash - - metapathways build_db \ - -t ${threads} \ - -d ${path/to/save/reference_databases} \ - --func swissprot \ - -a fast - -`build-db` parameters: -~~~~~~~~~~~~~~~~~~~~~~ - -.. code-block:: bash - - usage: metapathways [-h] [-d PATH] [--func [CATEGORICAL ...]] [-a ALIGNER] - [-t INT] [--dryrun] [--snakemake [SNAKEMAKE ...]] [--test] - - automated database install - - options: - -h, --help show this help message and exit - -t INT, --threads INT - max number of cores to use in multithreaded steps [1] - --dryrun dry run snakemake - --snakemake [SNAKEMAKE ...] - additional snakemake cli args in the form of KEY="VALUE" - or KEY (no leading dashes) - --test use test values for all arguments - - database arguments: - -d PATH, --refdb_dir PATH - path to save the reference DB, [DEFAULT "./"] - --func [CATEGORICAL ...] - functional references, space-delimited list from - ['metacyc', 'swissprot', 'cazy', 'eggnog', 'uniref50', 'uniref90'], - [DEFAULT ['metacyc', 'swissprot']] - -a ALIGNER, --aligner ALIGNER - local aligner to index for, select one of ['fast', 'blast'], - [DEFAULT fast] - -Supported Functional Databases: -=============================== - -+-----------+---------------------------------------------------+---------------------+ -| Database | Description | Size (after setup) | -+===========+===================================================+=====================+ -| uniref90 | UniRef90 functional annotation database | ~270 GB | -+-----------+---------------------------------------------------+---------------------+ -| uniref50 | UniRef50 functional annotation database | ~30 GB | -+-----------+---------------------------------------------------+---------------------+ -| swissprot | SwissProt functional annotation database | <10 GB | -+-----------+---------------------------------------------------+---------------------+ -| metacyc | MetaCyc functional annotation database | <10 GB | -+-----------+---------------------------------------------------+---------------------+ -| cazy | CAZymes functional annotation database | <10 GB | -+-----------+---------------------------------------------------+---------------------+ - -.. note:: - - **METACYC**: When building MetaCyc you require a username and password due the their license. - During the build process there will be a prompt to enter your username, followed by - a prompt to enter your password. This will allow MetaPathways to access the download - for MetaCyc and build the database. - -Annotate a Metagenome ---------------------- - -MetaPathways has many parameters and flags to allow for explicit control of many aspects of proccessing. -However, the minimal (default) analysis requires very few user inputs. - -Minimal Usage: -~~~~~~~~~~~~~~ - -.. code-block:: bash - - metapathways run \ - -i $[input_metagenome.fa] \ - -d ${path/to/save/reference_databases} \ - -o ${path/to/output} \ - -t ${threads} - -`run` parameters: -~~~~~~~~~~~~~~~~~ - -.. code-block:: bash - - usage: Metapathways run [options] - - Minimum REQUIRED Command: - MetaPathways run -i INPUT_FILE -o OUTPUT_DIR -d REFDB_DIR - - options: - -h, --help show this help message and exit - - Minimum Required Arguments: - -i INPUT_FILE, --input_file INPUT_FILE - path to the input fasta file/input dir [REQUIRED] - -o OUTPUT_DIR, --output_dir OUTPUT_DIR - path to the output directory [REQUIRED] - -d REFDB_DIR, --refdb_dir REFDB_DIR - path to the reference DB [REQUIRED] - - Quality Controls Arguments: - --input_format {fasta,fasta-amino} - Input format, FASTA support only [fasta] - --qc_min_length QC_MIN_LENGTH - Minimum length for quality control [180] - --qc_delete_replicates {yes,no} - Delete replicates in quality control [yes] - - ORF Prediction Arguments: - --orf_strand {pos,neg,both} - Strand for ORF prediction [both] - --orf_algorithm ORF_ALGORITHM - Algorithm for ORF prediction, Prodigal support only [prodigal] - --orf_min_length ORF_MIN_LENGTH - Minimum ORF length [60] - --orf_translation_table ORF_TRANSLATION_TABLE - Translation table for ORF prediction, see Prodigal for translation tables [11] - --orf_mode {single,meta} - Mode for ORF prediction [meta] - - Functional Annotation Arguments: - --annotation_algorithm {FAST,BLAST} - Algorithm for ORF annotation [FAST] - --annotation_dbs ANNOTATION_DBS [ANNOTATION_DBS ...] - Database(s) for annotation, space-separated list [swissprot] - --annotation_min_bsr ANNOTATION_MIN_BSR - Minimum BSR for annotation [0.4] - --annotation_max_evalue ANNOTATION_MAX_EVALUE - Maximum e-value for annotation [0.000001] - --annotation_min_score ANNOTATION_MIN_SCORE - Minimum score for annotation [20] - --annotation_min_length ANNOTATION_MIN_LENGTH - Minimum length for annotation [45] - --annotation_max_hits ANNOTATION_MAX_HITS - Maximum hits for annotation [5] - --annotation_run_mode {default,pervol} - Run mode for annotation, FAST only [pervol] - - rRNA Annotation Arguments: - --rRNA_refdbs RRNA_REFDBS [RRNA_REFDBS ...] - Reference databases for rRNA annotation, space-separated list - [SILVA_138.1_LSURef_NR99_tax_silva_trunc SILVA_138.1_SSURef_NR99_tax_silva_trunc] - --rRNA_max_evalue RRNA_MAX_EVALUE - Maximum e-value for rRNA annotation [0.000001] - --rRNA_min_identity RRNA_MIN_IDENTITY - Minimum identity for rRNA annotation [20] - --rRNA_min_bitscore RRNA_MIN_BITSCORE - Minimum bitscore for rRNA annotation [50] - - Read Mapping Arguments (single sample support only): - -1 FWD_FASTQ, --fastq FWD_FASTQ - location of the raw fastq file, either forward or interleaved - -2 REV_FASTQ, --rev_fastq REV_FASTQ - location of the raw reverse fastq file, if separate paired-end - --interleaved if paired-end is interleaved [False] - - Pipeline Step Arguments: - --PREPROCESS_INPUT {yes,skip,redo} - Step: PREPROCESS_INPUT [yes] - --ORF_PREDICTION {yes,skip,redo} - Step: ORF_PREDICTION [yes] - --FILTER_AMINOS {yes,skip,redo} - Step: FILTER_AMINOS [yes] - --SCAN_rRNA {yes,skip,redo} - Step: SCAN_rRNA [yes] - --SCAN_tRNA {yes,skip,redo} - Step: SCAN_tRNA [yes] - --FUNC_SEARCH {yes,skip,redo} - Step: FUNC_SEARCH [yes] - --PARSE_FUNC_SEARCH {yes,skip,redo} - Step: PARSE_FUNC_SEARCH [yes] - --ANNOTATE_ORFS {yes,skip,redo} - Step: ANNOTATE_ORFS [yes] - --GENBANK_FILE {yes,skip,redo} - Step: GENBANK_FILE [yes] - --CREATE_ANNOT_REPORTS {yes,skip,redo} - Step: CREATE_ANNOT_REPORTS [yes] - --PATHOLOGIC_INPUT {yes,skip,redo} - Step: PATHOLOGIC_INPUT [yes] - --COMPUTE_TPM {yes,skip,redo} - Step: COMPUTE_TPM [yes] - --force_redo Redo all steps [False] - - Miscellaneous Arguments: - -s SAMPLES [SAMPLES ...], --samples SAMPLES [SAMPLES ...] - process only specific samples, space-separated list - -t THREADS, --threads THREADS - max number of cores to use in multithreaded steps [1] - -v, --verbose print more information on the stdout - --test use test values for all arguments - -Add MAGs to MP output ---------------------- - -MetaPathways allows you to add MAG contig mappings to the annotated metagenome output. -Simple using the mag splitting command with a mapping file that maps the original contig -IDs (column 1) to their respective MAGs (column 2). - -.. note:: - - Be sure to annotated the metagenome first, then use the output dirctory as the (-o) of - the `mag_split` command. - -Minimal Usage: -~~~~~~~~~~~~~~ - -.. code-block:: bash - - metapathways mag_split \ - -o ${path/to/output}/{metagenome_id} \ - -m ${contig2mag_mapping} - -`mag_split` parameters: -~~~~~~~~~~~~~~~~~~~~~~~ - -.. code-block:: bash - - usage: Metapathways mag_split [options] - - Minimum REQUIRED Command: - Metapathways mag_split -o output_dir -m contig_mag_map - - options: - -h, --help show this help message and exit - -o OUTPUT_DIR, --output_dir OUTPUT_DIR - path where MP output was saved [REQUIRED] - -m MAG_MAP, --contig_mag_map MAG_MAP - TSV file that contains contig-to-MAG mapping [REQUIRED] - -Contig mapping format: -====================== - -+---------+------+ -| contig1 | MAG1 | -+---------+------+ -| config2 | MAG2 | -+---------+------+ -| contig3 | MAG2 | -+---------+------+ -| contig4 | MAG3 | -+---------+------+ -| contig5 | MAG1 | -+---------+------+ - -Environmental Pathway Genome Databases (ePGDBs) ------------------------------------------------ - -MetaPathways supports the creation of ePGDBs by interfacing with the -`Pathway Tools software `_. -In order to use this feature, you must first install Pathway Tools which requires a license. -Pathway Tools is freely available for research purposes to academic, non-profit, and government institutions, -and is available for a fee to commercial institutions. For academic use, a license and the download can be -found here: `Download `_. - -.. note:: - - Although MetaPathways is installed in a conda/mamba environment, Pathway Tools is installed in the - $HOME directory by default and is globally available in the commandline. MetaPathways expects that - the Pathway Tools software has been installed using defaults and may not work as expected if installed - in a custom way. - -Minimal Usage: -~~~~~~~~~~~~~~ - -.. code-block:: bash - - metapathways ptools \ - -o ${path/to/output}/{metagenome_id} - -`ptools` parameters: -~~~~~~~~~~~~~~~~~~~~~~~ - -.. code-block:: bash - - usage: Metapathways ptools [options] - - Minimum REQUIRED Command: - Metapathways ptools -o output_dir - - options: - -h, --help show this help message and exit - -o OUTPUT_DIR, --output_dir OUTPUT_DIR - path where MP output was saved [REQUIRED] - --tag TAG Custom name for ePGDB [optional] - --container Flag only used in containerized env [special flag] - --taxprune Set taxonomic pruning in pathway tools to True - -.. note:: - The container option only functions if both MetaPathways and Pathway Tools are installed - in the same container. But due to licensing constraints this cannot be distributed and therefore - is only experimental at this point. - - *There will be more work on this in the near future* - - diff --git a/docs/templates/redirect.html b/docs/templates/redirect.html new file mode 100644 index 0000000..57526f8 --- /dev/null +++ b/docs/templates/redirect.html @@ -0,0 +1,10 @@ + + + + + + + MetaPathways documentation + +

This guide has moved. Continue to the documentation.

+ diff --git a/docs/test-bundle.md b/docs/test-bundle.md new file mode 100644 index 0000000..b00cc0b --- /dev/null +++ b/docs/test-bundle.md @@ -0,0 +1,9 @@ +# Tiny CAMI II test inputs + +[Test walkthrough](test.md) · [Citation downloads](cami-references.md) + +This page includes the bundle README from the same repository revision as this documentation build. + +```{include} ../metapathways/regtests/cami_test/README.md +:start-after: +``` diff --git a/docs/test.md b/docs/test.md new file mode 100644 index 0000000..000762e --- /dev/null +++ b/docs/test.md @@ -0,0 +1,135 @@ +# Test walkthrough: three tiny CAMI samples + +Use this small test to check your installation and try the workflow before running a larger dataset. It is intended for any user. + +[Home](index.md) · Previous: [getting started](getting-started.md) · Next: [your inputs](inputs.md) + +This walkthrough uses ordinary MP commands and three distinct CAMI II samples ([Meyer et al., 2022](#cami-references)) included in the repository and installed package. No Pathway Tools license, installer or container is needed for the default commands. They run annotation, paired-read mapping and abundance, genome splitting, and the report/explorer. Pathway inference is explicitly skipped; licensed users can enable it in the optional section below. + +The input bundle is about **2.4 MiB**, separate from software and reference-database downloads. Each sample contains three 50 kb assembly regions assigned to three CAMI source genomes: + +| Sample | Assembly bases | Read pairs | Genome bins | Manifest | +| --- | ---: | ---: | ---: | --- | +| Urogenital_22 | 150,000 | 3,000 | 3 | `single.tsv` | +| Gastrointestinal_5 | 150,000 | 2,326 | 3 | `pair.tsv` | +| Skin_28 | 150,000 | 3,000 | 3 | `pair.tsv` | + +These are real subsets of simulated CAMI reads and gold-standard assemblies, not duplicated samples. Bins contain small genome fragments, not complete genomes or recovered MAGs. We selected abundant contigs and matching reads to exercise the interfaces. Do not use these coverage-biased subsets to assess biological accuracy, abundance, genome completeness, pathway recovery or publication performance. + +Protein taxonomy supports SwissProt (including the small test database), UniRef and eggNOG. Each annotation row reports the taxon of its own reference hit in `taxonomy`; `lca_taxonomy` is computed from score-qualified hits in that same database, with support counted independently for each database. The primary ORF row follows its selected annotation's `reference_db` and target. No database priority or cross-database fallback is used. These describe reference-hit evidence, not a definitive organism assignment for the query. Unsupported databases report `Not computed`; missing or unknown taxon IDs in supported databases report `Unclassified`. An actual LCA of `root` remains `root`. SILVA rRNA taxonomy is separate. Existing outputs need annotation-table regeneration to gain these fields. + +## 1. Install MP and prepare the small reference database + +Install with [Conda/Mamba or GitHub source](installation.md), then activate the MP environment. Container users can follow the same test through [Docker or Apptainer](containers.md). + +```bash +metapathways prepare_test -o ~/mp-test +cd ~/mp-test +metapathways build_db --test -d MPDB +``` + +`prepare_test` copies the included assemblies, paired reads, genome maps, manifests and reference FASTAs into your workspace. It leaves matching files in place and refuses to overwrite changed files. Use a new workspace for a fresh test. `build_db` indexes the small SwissProt/SILVA references and downloads enzyme/taxonomy support records into `MPDB`. Preparation requires internet access and writable workspace storage; it does not modify your installation. + +The bundled `swissprot_test` contains 227 genuine SwissProt proteins: the 39 original K12 test references plus 188 reference targets matched by the three CAMI samples in a full-SwissProt run. The protein FASTA is about 109 KB. This gives the test useful annotation, taxonomy and genome-splitting examples without a full protein database download. Selection and reference provenance are recorded in {download}`test_reference.json <../metapathways/regtests/test_db/test_reference.json>`. The deliberately selected reference set is for workflow testing, not annotation accuracy or biological benchmarking. SILVA fixtures remain small; empty RNA tables can be legitimate. + +When updating an existing installation, rerun `metapathways build_db --test -d MPDB` after reinstalling MP to rebuild the reference indexes and mappings, and use a fresh output directory for the test acceptance run. + +## Run all three samples first + +The fixture is included in both the repository and installed package; no separate input download is needed. Keep the prepared folder structure intact so manifest paths resolve correctly. + +```bash +cd ~/mp-test +metapathways analysis_wf \ + --manifest cami-test/all.tsv \ + -o all -d MPDB \ + --annotation_dbs swissprot_test \ + --rRNA_refdbs SILVA_SSU_test SILVA_LSU_test \ + --skip_ptools \ + --threads 4 --memory '4 GB' --max_tasks 2 + +metapathways report -o all --serve --no-browser --port 8765 +``` + +The analysis uses the references in your workspace’s `MPDB` directory. It requires neither a full SwissProt/UniRef download nor Pathway Tools. + +The explorer starts at **Samples**. On a remote machine, use the [SSH tunnel instructions](reports-tutorial.md#view-a-remote-report-through-ssh) and open the token-bearing URL printed by the report command. The single- and two-sample commands below are additional checks; they are not prerequisites for the three-sample run. + +## 2. Run the single-sample workflow + +```bash +cd ~/mp-test +metapathways analysis_wf \ + --manifest cami-test/single.tsv \ + -o single -d MPDB \ + --annotation_dbs swissprot_test \ + --rRNA_refdbs SILVA_SSU_test SILVA_LSU_test \ + --skip_ptools \ + --threads 4 --max_cpus 8 --memory '4 GB' --max_memory '16 GB' +``` + +This uses `Urogenital_22`. These explicit resource settings suit a small interface test; use resources actually available on your machine. They are not recommended memory estimates for full metagenomes. Append `--dryrun` first if you want to inspect the plan, then repeat without that flag to execute it. Do not add `--test` to `analysis_wf`: that mode belongs to the separate bundled installation check and selects its own inputs. + +## 3. Run two distinct samples together + +```bash +metapathways analysis_wf \ + --manifest cami-test/pair.tsv \ + -o pair -d MPDB \ + --annotation_dbs swissprot_test \ + --rRNA_refdbs SILVA_SSU_test SILVA_LSU_test \ + --skip_ptools \ + --threads 4 --max_cpus 8 --memory '4 GB' --max_memory '16 GB' +``` + +This uses `Gastrointestinal_5` and `Skin_28`. Ready tasks can overlap within the CPU and memory budgets; dependencies still determine when each task may start. Repeat the same command once to check reuse: successful unchanged tasks should report `ALREADY_COMPUTED`. That repeat is not a fresh runtime benchmark. + +`all.tsv` selects all three samples. The manifests use paths relative to the manifest, so you can move the entire `cami-test` folder without editing them. The `sample_id` column controls output names independently of assembly filenames. + +To exercise automatic discovery on all three samples, replace `--manifest cami-test/pair.tsv` with `-i cami-test/inputs`, and choose a new output such as `-o all-auto`. The discovery root contains only `assemblies/`, `reads/`, and `mag_maps/`. Manifests and provenance live outside it so strict discovery does not reject extra files. + +## 4. Check outputs and explore tables + +| Check | Single-sample location | +| --- | --- | +| Resolved sample ID and paired read paths | `single/inputs.resolved.tsv` | +| Functional/taxonomic annotation tables | `single/Urogenital_22/results/annotation_table/` | +| Contig and ORF abundance | `single/Urogenital_22/results/rpkm/` | +| Genome assignments and split inputs | `single/Urogenital_22/magsplitter/` | +| Task outcomes and resource records | `single/logs/analysis_wf/RUN_ID/` | +| Report and explorer | `single/reports/` | + +`RUN_ID` means the generated invocation directory, not a literal directory name. Inspect `summary.json` and the linked logs. Required tasks must succeed; a report existing alone does not prove success. Both paired-read paths must be distinct, all three genome bins should have split inputs, and the pair report should contain both sample IDs. Pathway tables are absent by design with `--skip_ptools`. + +```bash +metapathways report -o single --serve --no-rebuild +``` + +In `EDA_portal.html`, select a sample, search an annotation, follow an ORF to its related records, and export a filtered CSV. Check that sample/entity identifiers remain in the export. Stop the server with Ctrl-C and use `-o pair` to inspect the two-sample output. Follow [the report tutorial](reports-tutorial.md) for remote SSH access and table relationships. + +## 5. Optional: include licensed Pathway Tools + +Follow [the license and installer guide](pathway-tools.md#get-the-installer). Build and register your own image: + +```bash +metapathways build_pt \ + -i ~/Downloads/pathway-tools-29.5-linux-64-tier1-install \ + -o ~/mp-test/containers +``` + +Repeat either workflow command with a new output directory, remove `--skip_ptools`, and add `--taxprune --taxonomic_scope all`. MP uses the registered image. This adds community and per-bin PGDB inference/export. It does not require MetaCyc as a sequence annotation database: inference uses the licensed MetaCyc inside the image. See the Pathway Tools guide if you also want to build MetaCyc annotation references. + +Tiny genome fragments and sparse annotation references may yield few or no pathways. Check task outcomes and exports rather than requiring a universal pathway count. A full-reference test is more informative biologically. + +## Provenance and validation + +The bundle's [README](test-bundle.md), {download}`provenance.json <../metapathways/regtests/cami_test/provenance.json>`, and {download}`validation.json <../metapathways/regtests/cami_test/validation.json>` describe selection, original sample paths, retained contig regions, matching read counts, hashes, and checks performed. The preparation script is {download}`scripts/prepare_cami_test.py <../scripts/prepare_cami_test.py>`. No Pathway Tools software or MetaCyc sequence/database material is included. + +Input validation covers all manifests, automatic discovery, intact mate pairing, complete contig-to-genome maps and gene prediction on every retained contig. It does not substitute for the test executing the full workflow above. See [benchmarking](benchmarking.md) before collecting publication statistics from the complete CAMI samples. + +## Recorded three-sample validation + +The 2026-10-02 local SwissProt test run completed all 51 tasks successfully. The [audit record](validation/reviewer-2026-10-02.json) confirms all 393 CDS records were retained, paired read identifiers matched, all 421 gene/RNA abundance records and 9 contig records matched their source measurements, and taxonomy remained tied to each reference database and target. Each sample produced three CAMI genome bins. The 28 report placeholders were RNA features, not missing CDS annotations. Skin_28 had no rRNA queries, explaining its two BLAST empty-query warnings. This run used `--skip_ptools`; it does not establish successful PGDB construction or validate full-scale/HPC performance. + +```{include} includes/cami-references.md +``` diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md new file mode 100644 index 0000000..b4fa74a --- /dev/null +++ b/docs/troubleshooting.md @@ -0,0 +1,10 @@ +# Troubleshooting and reproducibility + +- **A command fails before execution:** check the CLI/error logs, activate the environment and use `command -v nextflow coverm fastal` to check dependencies. +- **A task exceeds the memory budget:** lower concurrency or adjust `--memory`/`--max_memory` to measured requirements; Slurm can enforce its allocation. +- **Some MAG pathways are missing:** inspect the entity's last task status and logs. Expected MAG inference failure is distinct from missing annotation inputs. +- **The portal says local explorer required:** launch `metapathways report -o OUTPUT --serve --no-rebuild` and open the URL it prints. +- **Results were changed after report generation:** rebuild the report. It is an indexed snapshot, not a live view of files. +- **A report rebuild fails:** inspect the named source and expected schema. Fix/correct the source or report importer; do not invent missing identifiers or replace absent values with zero. + +Record commands, Git commit, software versions, reference versions/checksums and input checksums for reproducible work. See [reproducibility notes](reproducibility.md) for the historical benchmark and the distinction between installation tests, synthetic software tests and full biological validation. diff --git a/docs/validation/reviewer-2026-10-02.json b/docs/validation/reviewer-2026-10-02.json new file mode 100644 index 0000000..2bf2a3c --- /dev/null +++ b/docs/validation/reviewer-2026-10-02.json @@ -0,0 +1,87 @@ +{ + "date": "2026-10-02", + "version": "3.5.2", + "dataset": "bundled three-sample CAMI reviewer inputs", + "output_directory": "/data/users/ryan/MetaPathways_AppNote/mp352-conda-reviewer/output-taxonomy", + "workflow_tasks": 51, + "all_tasks_successful": true, + "sqlite_integrity": "ok", + "foreign_key_violations": 0, + "report_schema": 3, + "pathway_tools_executed": false, + "samples": [ + { + "sample": "Gastrointestinal_5", + "assembly_contigs": 3, + "read_pairs": 2326, + "CDS": 140, + "abundance_rows": 155, + "genome_bins": 3, + "reference_hit_taxonomy_rows": 311, + "primary_rows_with_hit_taxonomy": 68, + "orf_tpm_sum": 1000000.0000000001, + "contig_tpm_sum": 1000000.0399999999, + "unannotated_placeholders_are_only_RNA": true, + "feature_counts": { + "CDS": 140, + "tRNA": 12, + "rRNA": 3 + } + }, + { + "sample": "Skin_28", + "assembly_contigs": 3, + "read_pairs": 3000, + "CDS": 117, + "abundance_rows": 121, + "genome_bins": 3, + "reference_hit_taxonomy_rows": 221, + "primary_rows_with_hit_taxonomy": 63, + "orf_tpm_sum": 1000000.0000000006, + "contig_tpm_sum": 1000000.06, + "unannotated_placeholders_are_only_RNA": true, + "feature_counts": { + "CDS": 117, + "tRNA": 4 + } + }, + { + "sample": "Urogenital_22", + "assembly_contigs": 3, + "read_pairs": 3000, + "CDS": 136, + "abundance_rows": 145, + "genome_bins": 3, + "reference_hit_taxonomy_rows": 244, + "primary_rows_with_hit_taxonomy": 68, + "orf_tpm_sum": 999999.9999999998, + "contig_tpm_sum": 999999.93, + "unannotated_placeholders_are_only_RNA": true, + "feature_counts": { + "CDS": 136, + "tRNA": 6, + "rRNA": 3 + } + } + ], + "checks_passed": [ + "paired FASTQ identifiers and sequence/quality lengths", + "all CDS represented in primary annotations", + "every source abundance value matches the wide explorer", + "no unavailable ORF coverage converted to zero", + "functional taxonomy joined by sample/ORF/database/target", + "primary taxonomy agrees with selected reference hit", + "all primary annotation placeholders explained by RNA features", + "three contigs assigned to three genome bins per sample", + "TPM sums approximately one million separately for ORF and contig features", + "51 Nextflow trace rows completed with exit zero" + ], + "expected_messages": [ + "Two Skin_28 BLAST empty-query warnings: no rRNA queries in this subset", + "28 RNA features have abundance but no primary protein annotation", + "PGDBs unavailable because --skip_ptools was used", + "HTTP 404s from report browsing, not workflow failures" + ], + "report_notes_after_cleanup": 0, + "report_cleanup": "Removed generic abundance notices and excluded explicitly identified tRNA/rRNA features from missing protein annotation warnings; all 430 abundance rows retained." +} diff --git a/docs/workflow.md b/docs/workflow.md new file mode 100644 index 0000000..f457826 --- /dev/null +++ b/docs/workflow.md @@ -0,0 +1,206 @@ +# Workflow and output guide + +For the tool-by-tool diagrams and citations, see the [detailed workflow](detailed-workflow.md). + +[Start with the README](index.md) · [Complete CLI flags](cli-reference.md) + +For guided usage, start with the [command cookbook](commands.md) or [Pathway Tools chapter](pathway-tools.md). This page describes the stage and output contracts. + +**Before running the complete workflow with PGDBs, follow the [Pathway Tools installation guide](pathway-tools.md)** to build and register your licensed SIF. Use `analysis_wf --skip_ptools` when you do not need pathway inference. + +## Command boundaries + +| Command | Work performed | Requires | +| --- | --- | --- | +| `analysis_wf` | One DAG for multiple assemblies, per-sample reads/MAG maps, PGDBs and combined reports | [Layout or manifest](analysis.md#analysis-wf-input-layout), MPDB, registered SIF | +| `build_db` | Download references, build FAST/BLAST indexes and supporting maps through Nextflow | Network, writable database directory, indexing tools | +| `run` | Assembly QC, prediction, search, annotation, optional mapping, Pathway Tools input preparation | Assembly and compatible MPDB; reads optional | +| `mag_split` | Reuse existing annotations to create MAG Pathway Tools inputs; preserve the original contig map | Completed community outputs, MAGSplitter, headerless two-column contig map | +| `build_pt` | Build, validate and register a private Pathway Tools SIF | Licensed installer, Apptainer, Nextflow, build prerequisites | +| `ptools` | Build/extract community and optional MAG PGDBs | Pathway Tools inputs, SIF/native installation, Camelot extraction dependency | +| `report` | Index existing outputs; optionally open a local search/export portal | Existing MP output; no biological tools needed for indexing | + +Nextflow schedules tasks; existing MP scripts still perform the biological calculations and write the established sample paths. Public database construction uses the new Nextflow planner. The historical Snakefiles and old `bin/metapathways-data-install.sh` helper remain legacy artifacts; use `metapathways build_db` for supported construction. `--snakemake` on that command accepts only explicitly supported compatibility options (`cores`, `jobs`, `dryrun`, `dry-run`, `forceall`, `keep-going`, `rerun-incomplete`, `printshellcmds`, `latency-wait`); unsupported options fail rather than being interpreted as arbitrary Nextflow settings. Prefer the named resource flags. + +## Annotation stages + +The pipeline follows sample dependencies while permitting independent database searches/parses to run concurrently. Each sample maintains its own context and final output paths. + +| Stage | Purpose | Representative output | +| --- | --- | --- | +| `PREPROCESS_INPUT` | Filter sequences; establish MP/original contig identifiers | `preprocessed/` | +| `ORF_PREDICTION` / `ORF_TO_AMINO` | Predict ORFs and derive sequences | `orf_prediction/` | +| `FILTER_AMINOS` | Apply amino-acid sequence filters | QCed protein sequences | +| `FUNC_SEARCH` | Search each selected functional database | `blast_results/` | +| `COMPUTE_REFSCORES` | Compute reference scores used by parsing | Refscore file | +| `PARSE_FUNC_SEARCH` | Apply search thresholds and parse database results | Parsed search tables | +| `SCAN_rRNA` | Barrnap prediction followed by reference search/summary | `results/rRNA/` | +| `SCAN_tRNA` | Predict tRNAs | `results/tRNA/` | +| `ANNOTATE_ORFS` | Assign functional annotations | Annotated GFF and annotation products | +| `CREATE_ANNOT_REPORTS` | Write primary functional/taxonomic and reference tables | `results/annotation_table/` | +| `GENBANK_FILE` | Export annotated sequence record | `genbank/*.gbk` | +| `PATHOLOGIC_INPUT` | Prepare selected Pathway Tools inputs and EC/reaction mappings | `ptools/`, `*.EC_RXN_map.tsv`, `*.ptinput.tsv` | +| `COMPUTE_TPM` | Map supplied reads and derive abundance/count tables | `results/rpkm/` | + +FASTA-amino inputs take the compatible protein-input path. Stage choices and thresholds are in [CLI reference](cli-reference.md); internal derived stages do not necessarily have independent public flags. MP preserves the existing `yes`, `skip`, `redo` controls. Skipping a producer does not manufacture the inputs needed by its consumers. + +QC defaults, ORF prediction parameters, alignment algorithm/mode, functional score/e-value/identity thresholds and rRNA database thresholds remain configurable through the CLI. Search flags must match the configured database indexes. The portal reports existing assignments; it does not change thresholding or infer taxonomic ranks from free-text lineage strings. + +## Output layout + +```text +OUTPUT/ +├── SAMPLE/ +│ ├── preprocessed/ # Contigs and original-name map +│ ├── orf_prediction/ # Predicted and filtered sequences +│ ├── blast_results/ # Search and parsed hits +│ ├── genbank/ # GFF/GenBank annotation +│ ├── ptools/ # Community Pathway Tools inputs +│ ├── magsplitter/ +│ │ ├── contig_to_mag.tsv # Preserved original two-column map +│ │ └── results/MAG_ID/ # MAG-specific Pathway Tools inputs +│ ├── results/ +│ │ ├── annotation_table/ +│ │ ├── rpkm/ +│ │ ├── rRNA/ +│ │ ├── tRNA/ +│ │ └── pgdb/ +│ │ ├── community/ +│ │ └── MAGs/MAG_ID/ +│ ├── run_statistics/ +│ ├── metapathways_steps_log.txt +│ └── errors_warnings_log.txt +├── reports/ # Run report, EDA portal and SQLite +├── logs/ # Durable CLI/Nextflow records +└── .metapathways/ # Receipts, locks and temporary work +``` + +Commands operating on one existing sample (`mag_split`, `ptools`) also store their logs below that sample directory. A report over the parent includes these execution summaries. Running the report directly on a sample creates `SAMPLE/reports/`; running it on the parent creates `OUTPUT/reports/`. + +## Dependency and cache behavior + +Resource requests belong to each task. pProdigal, barrnap, ptRNAscan, FAST/BLAST, and read mapping/counting use the requested thread count capped by the total CPU budget. Serial stages, including Pathway Tools and Python parsers, reserve one CPU. Nextflow can mix threaded and serial tasks within the CPU and memory budgets. Independent database contexts share a preceding-stage dependency, and later stages wait for all contexts in that group. A failing required task stops dependent work. Optional MAG PGDB failures remain recorded while other entities proceed. + +MP retains durable task receipts outside Nextflow work directories. Reuse compares command signatures and tracked file state (size and modification time, including tracked reference/index files). These are not full content hashes of all biological inputs. The report source manifest independently records SHA-256 for files it parses. Use external input/reference checksums for stronger provenance, and `redo` when changing tool implementations or environments that are not represented by command/file fingerprints. + +Some legacy stages rewrite earlier outputs, such as the final GenBank record. The final producer owns that file's cached fingerprint. Old abundance output is not adopted without rerunning the corrected read mapping. Database index receipts track all index shards, so a missing shard invalidates reuse. + +Temporary cleanup archives diagnostic files before removal. Explicit work/cache paths and interruptions are retained. Final results and logs are never considered disposable caches. Multiple controllers targeting the same command/output are blocked by a file lock; per-invocation resource budgets do not govern unrelated MP commands. + +## Report limitations and error recovery + +Report generation runs after successful workflow completion. It is disk-backed, single-controller postprocessing rather than a scheduled biological stage. Large multi-sample report builds still use headnode I/O/CPU; run standalone `report` where site policy permits metadata indexing. It does not alter biological outputs. For a failed workflow, generate a report explicitly over the partial output to inspect what exists. + +A report requires recognized sample directories. Headers and identifiers matter: the importer rejects ambiguous duplicate source tables and malformed records. Do not merge different samples by concatenating tables containing unscoped `C1-G1` identifiers. Report schema versioning and source records make supported mappings explicit. + +Missing source tables, unannotated placeholders and expected optional MAG failures are visible instead of silently becoming zeros. See [results schema](results-schema.md) for exact semantics and [reproducibility](reproducibility.md) for benchmark limitations. + +For complete multi-sample execution, see [automatic input matching](analysis.md#analysis-wf-input-layout) and the [custom manifest](analysis.md#custom-analysis-manifest). `analysis_wf` assigns independent per-sample dependencies, saves the resolved input manifest, and builds the combined report after Nextflow completes so final statuses are included. + +### RNA identifiers and abundance integrity + +RNA annotations are emitted once per locus, including contigs with no CDS predictions. rRNA IDs include contig, subtype, coordinates and strand; distinct copies of the same rRNA must not share a gene ID. GFF-to-GTF conversion rejects duplicate IDs before mapping. ORF abundance uses featureCounts' `Length` column, requires a one-to-one gene-ID join with the GTF, and reports zero RPKM/TPM when all counts are zero. Tool output and errors are streamed into the task log, and failed sorting/conversion/counting/calculation commands stop the task immediately. + +### Concurrent FAST searches + +MP supplies FAST's `-X` option with a unique temporary directory for each search and reference-score invocation. FAST's default temporary names use a time-based seed and can collide when independent searches start in the same second. Older wrappers could therefore mix hits from different databases even when both processes returned success. A parser error showing Swiss-Prot `sp|...` targets in a MetaCyc result is one symptom; the databases themselves may be intact. + +The corrected planner invalidates previous functional-search and reference-score receipts and does not adopt untracked outputs for these stages. Repeat the original run command to regenerate the searches and affected downstream results; unrelated unchanged preprocessing can be reused. Search volumes stop on the first tool error, and merged results are published only after all volumes succeed. This change does not require rebuilding MPDB or the Pathway Tools SIF. + +### Pathway Tools failure diagnostics + +Before each community or MAG PGDB invocation, MP expands its compact `0.pf` annotations into per-contig PathoLogic inputs in private staging. Each genetic element has an annotation file and a `SEQ-FILE` containing its real preprocessed contig sequence. Feature IDs are preserved. The sample's `ptinput.tsv` supplies each feature's original contig, coordinates and strand; these restore MAG member coordinates that the splitter copied from a representative annotation. Missing sequences, unmapped/duplicate features or invalid source coordinates stop preparation before Pathway Tools starts. Unannotated contigs without PGDB features are omitted. The sequence counts and coordinate-restoration counts are retained in `diagnostics//input/sequence-input.json`. In compact mode this diagnostic is inside the PGDB diagnostics archive; see [scratch and archive behavior](benchmarking.md#compact-results-on-limited-storage). This enables Pathway Tools to derive protein sequences for its PGDB BLAST databases. The source annotation files and MAG split outputs are not modified. + +Each container PGDB attempt saves its internal `pathologic.log`, other available logs, reports and `execution.json` under the entity output's `diagnostics/ATTEMPT/` directory. The execution record identifies the image, exit code and phase (`build`, `export`, or `archive`). On failure MP also prints the last 80 lines of the internal Pathologic log. These diagnostics survive temporary container-state cleanup. An optional MAG task failing does not by itself establish that the failure is biological or expected; inspect its diagnostics. Community PGDB failures remain fatal. + +Older container wrappers deleted the private Pathologic log along with temporary state, so the root cause of a historical exit 255 may be unavailable. Updating MP and repeating the same `ptools` invocation retains successful task receipts and retries unsuccessful entities, saving the internal error if it recurs. This logging fix works with an existing SIF. Newly built images also include NCBI BLAST+; the image recipe hash changes, so `build_pt` creates a separate image when rebuilt. + +`build_pt` also snapshots official SRI patches for the selected release and validates a synthetic BLAST database/search before registration. Analysis tasks keep patch downloads disabled. Rebuilding creates a distinct image; changing the selected image can invalidate prior PGDB receipts. A patched image is not evidence that a particular vendor inference bug is resolved. Add `-d /path/to/MPDB -a fast` to prepare matching MetaCyc sequences, indexes and tables after the image build; see [MetaCyc from Pathway Tools](pgdb-workflow.md#metacyc-from-pathway-tools). This optional preparation operates on the reference, not on sample annotations or sample PGDBs. + +Pathway Tools EC inputs are written as one `EC` line per identifier, including +when source annotations contain comma-, semicolon-, or pipe-separated lists. +PGDB staging applies the same normalization to existing PF inputs without +modifying the originals or rerunning annotation searches. The staging diagnostic +`sequence-input.json` records how many features had their EC entries normalized. +Provisional identifiers (for example `3.6.5.n1`) are preserved; Pathway Tools may +reject these, and its warnings remain in the saved logs. A successful Nextflow +wrapper for an optional MAG is not proof of a successful PGDB: consult MP's task +receipts and the per-entity `execution.json` status. + +PGDB staging separates the amino-acid label and anticodon in recognized MP tRNA +names (for example, `C1094.tRNA2-GluTTC` becomes `C1094.tRNA2-Glu-TTC` in the +`NAME` field). This prevents the label's last letter from merging with the +three-base anticodon during name parsing. Feature `ID` fields, sequences, +coordinates and source files remain unchanged, so annotation and abundance +joins retain their original identifiers. The original and staged names are +recorded in `sequence-input.json` under `normalized_trna_names`. Unknown +anticodons such as `NNN` and unrecognized name formats are left unchanged; +MP does not guess their biological assignments. Updating this preparation +invalidates existing PGDB receipts, but does not require rebuilding the SIF +or rerunning annotations. + +When the original `orf_prediction/.cds.gff` is available, PGDB staging +also preserves Prodigal's per-contig translation table through PathoLogic's +`CODON-TABLE` field. This avoids treating table-4 TGA codons as premature stops. +Older outputs without that metadata retain Pathway Tools' default genetic code; +no alternative code is guessed. The chosen codes are recorded in +`sequence-input.json`. This uses the input format documented in Pathway Tools' +installed `sample-genetic-elements.dat`; it does not patch Pathway Tools. + +Concurrent SIF PGDB tasks use Xvfb with a private filesystem display socket. +Apptainer's `--containall` isolates `/tmp` but does not isolate Linux abstract +X sockets. MP disables Xvfb's abstract listener (`-nolisten local`) and creates +the private `/tmp/.X11-unix` directory. This avoids exhausting `xvfb-run`'s ten +display attempts when many containers start together, which can otherwise +return exit 1 during cleanup even after Pathway Tools saves its PGDB and prints +`Done`. Build and export X-server logs are retained as `build-xvfb.log` and +`export-xvfb.log` in the invocation diagnostics. This wrapper change requires +no SIF rebuild or vendor patch; nonzero build/export exits still fail the task. + +For a controlled transport-inference comparison with an installed SIF, use +`metapathways ptools -o OUTPUT/SAMPLE --entity community --no_transport_inference`. +Omit `--entity` to process the community and MAGs. `analysis_wf` also accepts +`--no_transport_inference`. TIP remains enabled by default; changing the setting +invalidates the corresponding PGDB task cache. Failed SIF builds retain their +on-disk databases under `diagnostics//failed-pgdbs/`, together with the +staged inputs and original failure status. This preserves recovery material; it +does not automatically export or label partial databases as successful. + +For an explicit PGDB taxon override, `ptools` and `analysis_wf` accept +`--taxon_id NCBI_ID`. This changes only private staged `organism-params.dat` +inputs, which are retained in diagnostics. It applies to every selected entity; +use `--entity community` with `ptools` for a community-only experiment. +`--taxprune --taxon_id 131567` uses cellular organisms with taxonomic pruning +enabled, avoiding the separate unpruned rescoring pass documented for +`-no-taxonomic-pruning` in the Pathway Tools User Guide (printed pp. 16–17). +This broad taxon includes multicellular eukaryotes too and is not a microbial-only +filter. Pruned and unpruned inference are different analysis settings. +Neither flag changes MP's gene taxonomic annotations. + +`--taxonomic_scope all|bacteria|archaea|eukaryotes` provides readable aliases +for taxon IDs 131567, 2, 2157, and 2759, respectively; `euks` is accepted as an +alias for `eukaryotes`. It is mutually exclusive with `--taxon_id`. For example, +`--taxprune --taxonomic_scope all` is equivalent to +`--taxprune --taxon_id 131567`. Omitting both defaults to taxonomic pruning with scope `all` (cellular life). +Use `--no_taxprune` to disable pruning. A `prokaryotes` scope is not +yet supported because it needs a union of Bacteria and Archaea; MP rejects it +explicitly rather than silently substituting cellular life (which includes +eukaryotes). Scope changes guide PGDB inference and do not filter input contigs. + +### If only `--threads 4` is specified + +For the default local executor, capable tools receive up to four CPUs each (capped by the available CPU budget). MP detects available CPUs and memory for the total scheduling budget; four threads is not a four-CPU limit for the whole workflow. General tasks reserve the default 16 GB each. In `analysis_wf`, PGDB tasks inherit the same memory request and always reserve one CPU; `--ptools_memory` is an optional override. Nextflow schedules ready tasks together while their reservations fit. For example, a detected budget of 32 CPUs and 64 GB permits at most four simultaneous tasks that each request four CPUs and 16 GB, even though CPUs remain free. Dependencies can reduce concurrency further. + +Detection is a startup snapshot, not exclusive ownership of the server. Set `--max_cpus` and `--max_memory` on shared machines. Local memory reservations govern scheduling, not hard memory enforcement. If the available memory budget is smaller than a task's reservation, planning fails; choose a suitable explicit `--memory` for a tiny test or provide more capacity. Slurm defaults to four submitted jobs and six submissions per minute; aggregate maxima apply only when explicitly supplied. + +### Abundance checkpoints + +`COMPUTE_TPM` checks the assembly, annotation GFF, supplied read files (including R2 for paired inputs), commands, and final abundance outputs when deciding whether to reuse results. Its `bwa/` directory contains generated intermediate files and is not an input dependency. Changes to those intermediates alone do not require recomputing valid abundance tables. + +Checkpoints written before this correction included `bwa/` as an input. Their first invocation with the corrected code recomputes abundance once to replace that old checkpoint; subsequent unchanged runs reuse it. Other tasks retain their existing checkpoint rules. + +### Large multi-sample workflows + +MP writes `main.nf` plus small files under `modules/` in each invocation log directory. This keeps thousands of sample/bin tasks below the JVM compiled-method size limit. Modules expose individual task completion channels: they do not introduce a wait for the whole module, a separate resource budget, or serial sample execution. The shared executor limits and durable MP task checkpoints still apply. Keep the module files with `main.nf` when archiving a generated workflow. + +If an older version failed at startup with `Method too large: Main.runScript`, update MP and repeat the same command with the same output directory. Compilation fails before any tasks are submitted; deleting the output is unnecessary. The next invocation generates a new modular workflow. diff --git a/metapathways/LCAComputation.py b/metapathways/LCAComputation.py index 771d687..e81d69b 100644 --- a/metapathways/LCAComputation.py +++ b/metapathways/LCAComputation.py @@ -232,24 +232,8 @@ def getTaxonomy(self, name_groups, taxid=False, return_id=False): # extracts taxon names for a refseq annotation def get_species(self, hit, dbname): - accession_PATT = re.compile(r"ref\|(.*)\|") - if not "comment" in hit and not "target" in hit: - return None - species = "" - try: - if 'eggnog' in dbname.lower(): - m = hit['target'].split('.', 1)[0] - species = str(m) - elif 'uniref' in dbname.lower(): - m = hit['product'].split('TaxID ', 1)[1].split(' ')[0] - species = str(m) - except: - return None - - if species and species != "": - return species - else: - return None + from metapathways.protein_taxonomy import hit_taxid + return hit_taxid(hit, dbname) # used for optimization def set_results_dictionary(self, results_dictionary): diff --git a/metapathways/MetaPathways_annotate_fast.py b/metapathways/MetaPathways_annotate_fast.py index ee3bc0a..5a699e9 100644 --- a/metapathways/MetaPathways_annotate_fast.py +++ b/metapathways/MetaPathways_annotate_fast.py @@ -362,7 +362,7 @@ def insert_orf_into_dict(line, contig_dict, shortenorfid=False): seqname = attributes["orf_id"] attributes["seqname"] = seqname if feature_type == "rRNA": - seqname = seqname + "." + attributes["name"] + "_" + str(fields[3]) + seqname = seqname + "." + attributes["name"] + "_" + "_".join((fields[3], fields[4], fields[6])) attributes["seqname"] = seqname if not seqname in contig_dict: contig_dict[seqname] = [] @@ -566,8 +566,13 @@ def write_16S_tRNA_gene_info(contig_id, f_rec, outputgff_file, tag): output_line += "\t" + str(f_rec["score"]) output_line += "\t" + str(f_rec["strand"]) output_line += "\t" + str(f_rec["frame"]) - attributes = "ID=" + str(f_rec["seqname"]).rsplit('-', 1)[1].rsplit('_')[0] - attributes += ";" + "locus_tag=" + str(f_rec["name"]) + # Distinct copies of the same rRNA on a contig are distinct loci. + short_contig = contig_id.rsplit('-', 1)[-1] + subtype = str(f_rec['name']).removesuffix('_rRNA') + strand = 'plus' if f_rec['strand'] == '+' else 'minus' + identifier = f"{short_contig}.{subtype}_{f_rec['start']}_{f_rec['end']}_{strand}" + attributes = "ID=" + identifier + attributes += ";locus_tag=" + identifier attributes += ";" + "product=" + f_rec["product"] output_line += "\t" + attributes @@ -1013,15 +1018,14 @@ def create_annotation( compact_output=compact_output, ) count += 1 # move to the next orf - # Add rRNA and tRNA records to output gff - if rRNA_yes == True: - if contig in rRNA_dictionary: - for rec in rRNA_dictionary[contig]: - write_16S_tRNA_gene_info(contig, rec, outputgff_file, "_rRNA") - if tRNA_yes == True: - if contig in tRNA_dictionary: - for rec in tRNA_dictionary[contig]: - write_16S_tRNA_gene_info(contig, rec, outputgff_file, "_tRNA") + # RNA features belong to contigs, not CDS parser buffers. Emit them once, + # including contigs with RNA genes but no protein-coding predictions. + for contig, records in rRNA_dictionary.items(): + for rec in records: + write_16S_tRNA_gene_info(contig, rec, outputgff_file, "_rRNA") + for contig, records in tRNA_dictionary.items(): + for rec in records: + write_16S_tRNA_gene_info(contig, rec, outputgff_file, "_tRNA") output_comp_annot_file1.close() output_comp_annot_file2.close() diff --git a/metapathways/MetaPathways_create_genbank_ptinput.py b/metapathways/MetaPathways_create_genbank_ptinput.py index 8cbf502..f56d80a 100644 --- a/metapathways/MetaPathways_create_genbank_ptinput.py +++ b/metapathways/MetaPathways_create_genbank_ptinput.py @@ -22,6 +22,7 @@ from os import makedirs, path, listdir, remove, rename + from metapathways.pt_ec import normalize_ecs from metapathways import errorcodes as errormod from metapathways import general_utils as gutils from metapathways import metapathways_utils as mputils @@ -229,6 +230,16 @@ def process_gff_file(gff_file_name, output_filenames, nucleotide_seq_dict, \ ) # this function creates the pathway tools input files +def ptinput_dataframe(features, nucleotide_sequences): + """Keep the coordinate schema valid for protein, RNA-only and empty inputs.""" + frame = pd.DataFrame.from_dict(features, orient='index') + for column in ('id', 'seqname', 'start', 'end', 'strand'): + if column not in frame: + frame[column] = pd.Series(index=frame.index, dtype=object) + frame['contig_length'] = [len(nucleotide_sequences[name]) for name in frame['seqname']] + return frame + + def write_ptinput_files(outfiles, contig_dict, sample_name, nucleotide_seq_dict, \ protein_seq_dict, compact_output, orf_to_taxonid={}): @@ -427,6 +438,7 @@ def write_ptinput_files(outfiles, contig_dict, sample_name, nucleotide_seq_dict, else: attrib['ec'] = [attrib['ec']] + attrib['ec'] = normalize_ecs(attrib['ec']) # keep all ORFs used for Ptools pt_attrib_dict[shortid] = attrib @@ -491,7 +503,7 @@ def write_ptinput_files(outfiles, contig_dict, sample_name, nucleotide_seq_dict, write_input_sequence_file(output_dir_name, shortid, fastaStr) ''' #endif - pt_attrib_df = pd.DataFrame.from_dict(pt_attrib_dict, orient='index') + pt_attrib_df = ptinput_dataframe(pt_attrib_dict, nucleotide_seq_dict) if 'ec' in pt_attrib_df.columns: pt_attrib_df['ec'] = ['|'.join(x) if isinstance(x, list) else '' for x in pt_attrib_df['ec']] if 'rxn' in pt_attrib_df.columns: @@ -556,11 +568,7 @@ def write_to_pf_file(output_dir_name, shortid, attrib, pfFile, compact_output): gutils.fprintf(pfFile, "METACYC\t%s\n", rxn_val) if 'ec' in attrib: - ec_val = attrib['ec'] - #if ec_val: - # gutils.fprintf(pfFile, "EC\t%s\n", ec_val) - ec_list = list(set(attrib['ec'])) - for ec_val in ec_list: + for ec_val in normalize_ecs(attrib['ec']): gutils.fprintf(pfFile, "EC\t%s\n", ec_val) if 'taxon' in attrib: @@ -968,4 +976,3 @@ def MetaPathways_create_genbank_ptinput(argv, errorlogger = None, runstatslogger if len(sys.argv) > 1: main(sys.argv[1:]) - diff --git a/metapathways/MetaPathways_create_reports_fast.py b/metapathways/MetaPathways_create_reports_fast.py index 949ad8d..18d2935 100644 --- a/metapathways/MetaPathways_create_reports_fast.py +++ b/metapathways/MetaPathways_create_reports_fast.py @@ -8,6 +8,9 @@ try: import traceback + import csv + from itertools import groupby + from metapathways.protein_taxonomy import (supports_taxonomy, hit_taxonomy, raw_lca, taxon_label, NOT_COMPUTED, UNCLASSIFIED) import re import gc import resource @@ -395,18 +398,13 @@ def create_annotation( orfToContig[shortORFId] = contig - taxonomy = None - if shortORFId in Taxons: - taxonomy1 = Taxons[shortORFId] - taxonomy_id = lca.get_supported_taxon(taxonomy1, return_id=True) - preferred_taxonomy = lca.get_preferred_taxonomy(taxonomy_id) - - if preferred_taxonomy: - taxonomy = preferred_taxonomy - else: - taxonomy = Taxons[shortORFId] - else: - taxonomy = "root" + # The GFF records the database and target that supplied this annotation. + source_db = orf.get("sourcedb", "") + candidates = results_dictionary.get(source_db, {}).get(shortORFId, []) + hit = next((h for h in candidates if h.get("target") == orf["target"]), {}) + _, taxonomy = hit_taxonomy(hit, source_db, lca) + lca_taxonomy = Taxons.get(source_db, {}).get(shortORFId, + UNCLASSIFIED if supports_taxonomy(source_db) else NOT_COMPUTED) product = orf["product"] orf_id = orf["id"] seqname = orf["seqname"] @@ -423,7 +421,7 @@ def create_annotation( gutils.fprintf(output_table_file, "\t%s", orf["strand"]) gutils.fprintf(output_table_file, "\t%s", orf["target"]) gutils.fprintf(output_table_file, "\t%s", product) - gutils.fprintf(output_table_file, "\t%s\n", taxonomy) + gutils.fprintf(output_table_file, "\t%s\t%s\t%s\n", taxonomy, source_db, lca_taxonomy) output_table_file.close() @@ -1290,7 +1288,7 @@ def main(argv, errorlogger=None, runstatslogger=None): output_table_file, '\t'.join(["ORF_ID", "ORF_length", "start", "end", "Contig_Name", "Contig_length", - "strand", "target", "product", "taxonomy"])+"\n" + "strand", "target", "product", "taxonomy", "reference_db", "lca_taxonomy"])+"\n" ) @@ -1307,7 +1305,7 @@ def main(argv, errorlogger=None, runstatslogger=None): else: database_names = opts.database_name input_blastouts = opts.input_blastout - weight_dbs = opts.weight_db + weight_dbs = [1] * len(database_names) ##### uncomment the following lines for dbname, blastoutput in zip(database_names, input_blastouts): @@ -1334,56 +1332,40 @@ def main(argv, errorlogger=None, runstatslogger=None): blastParsers[dbname].setMaxErrorsLimit(5) blastParsers[dbname].setErrorAndWarningLogger(errorlogger) - # this part of the code computes the occurence of each of the taxons - # which is use in the later stage is used to evaluate the min support - # as used in the MEGAN software - - start = 0 + # Train each database independently. Neither LCA support nor assignments + # may be inherited from a different reference database. Length = len(listOfOrfs) _stride = 5000000 Taxons = {} - while start < Length: - pickorfs = {} - last = min(Length, start + _stride) - for i in range(start, last): - pickorfs[listOfOrfs[i]] = "root" - start = last - # print 'Num of Min support orfs ' + str(start) - results_dictionary = {} - for dbname, blastoutput in zip(database_names, input_blastouts): - if "eggnog" in dbname: - results = "eggnog" - elif "uniref" in dbname: - results = "uniref" + for dbname in database_names: + if not supports_taxonomy(dbname): + continue + for node in lca.taxid_to_ptaxid.values(): + node[2] = 0 + raw_taxons = {} + known_orfs = set(listOfOrfs) + for orf_id, group in groupby(blastParsers[dbname], key=lambda hit: hit['query']): + records = [hit for hit in group if isWithinCutoffs(hit, opts)] + if orf_id not in known_orfs or not records: + continue + taxid = raw_lca(records, dbname, lca) + raw_taxons[orf_id] = taxid + if taxid is not None: + lca.update_taxon_support_count(lca.id_to_name[taxid]) + Taxons[dbname] = {} + for orf_id, taxid in raw_taxons.items(): + if taxid is None: + label = UNCLASSIFIED else: - results = False - if results: - try: - results_dictionary[dbname] = {} - gutils.eprintf("\nScanning database : %s...", dbname) - process_parsed_blastoutput( - dbname, - blastParsers[dbname], - opts, - results_dictionary[dbname], - pickorfs, - callnum=1, - ) - lca.set_results_dictionary(results_dictionary) - lca.compute_min_support_tree( - opts.input_annotated_gff, pickorfs, dbname=dbname - ) - for key, taxon in pickorfs.items(): - Taxons[key] = taxon - except: - gutils.eprintf("ERROR: while training for min support tree %s\n", dbname) - errormod.insert_error(errorcode) - traceback.print_exc() - for dbname in results_dictionary.keys(): - gutils.eprintf( - "\n\tINFO:\tNumber of collected in block hits in {}: {}".format( - dbname, len(results_dictionary[dbname].keys()) - )) + supported = lca.get_supported_taxon(lca.id_to_name[taxid], return_id=True) + label = taxon_label(lca, supported) + Taxons[dbname][orf_id] = label + gutils.eprintf("\nTaxonomy computed independently for %s: %d ORFs\n", dbname, len(raw_taxons)) + + taxonomy_path = path.join(opts.output_dir, opts.sample_name + '.annotation_taxonomy.tsv') + with open(taxonomy_path, 'w') as stream: + csv.writer(stream, delimiter='\t').writerow( + ['orf_id', 'reference_db', 'target', 'taxid', 'taxonomy', 'lca_taxonomy']) # this loop determines the actual/final taxonomy of each of the ORFs # taking into consideration the min support @@ -1433,6 +1415,22 @@ def main(argv, errorlogger=None, runstatslogger=None): gutils.eprintf("\n\tINFO:\tNumber of ORFs processed : %s\n", str(start)) + # Keep hit taxonomy distinct from the within-database ORF LCA. + with open(taxonomy_path, 'a') as stream: + writer = csv.writer(stream, delimiter='\t') + for dbname, orfs in results_dictionary.items(): + for orf_id, hits in orfs.items(): + seen = set() + for hit in hits: + if hit['target'] in seen: + continue + seen.add(hit['target']) + taxid, label = hit_taxonomy(hit, dbname, lca) + consensus = Taxons.get(dbname, {}).get(orf_id, + UNCLASSIFIED if supports_taxonomy(dbname) else NOT_COMPUTED) + output_id = mputils.ShortenORFId(orf_id) if opts.compact_output else orf_id + writer.writerow([output_id, dbname, hit['target'], taxid, label, consensus]) + # create the annotations now orfToContig = {} @@ -1498,11 +1496,12 @@ def process_subsys2peg_file(subsystems2peg, subsystems2peg_file): def print_orf_table(results, orfToContig, output_dir, outputfile, compact_output=False): - addHeader = True if not path.exists(output_dir): makedirs(output_dir) - orf_dict = {} + # This is also MAGSplitter's structural ORF-to-contig map. Preserve + # predicted CDS even when the selected references have no qualifying hits. + orf_dict = {orf: {"contig": contig} for orf, contig in orfToContig.items()} for dbname in results.keys(): gutils.eprintf("\n\tINFO:\tnumber of hits in {}: {}".format(dbname, len(results[dbname].keys()))) for orfname in results[dbname]: @@ -1546,6 +1545,9 @@ def print_orf_table(results, orfToContig, output_dir, outputfile, compact_output dbnames.append(dbname) headers.append(std_dbname) + if outputfile.tell() == 0: + gutils.fprintf(outputfile, "# %s\n", "\t".join(headers)) + sampleName = None for orfn in orf_dict: @@ -1565,10 +1567,6 @@ def print_orf_table(results, orfToContig, output_dir, outputfile, compact_output else: row.append("") - if addHeader: - gutils.fprintf(outputfile, "# %s\n", "\t".join(headers)) - addHeader = False - gutils.fprintf(outputfile, "%s\n", "\t".join(row)) diff --git a/metapathways/MetaPathways_func_search.py b/metapathways/MetaPathways_func_search.py index c1c805c..da7a088 100644 --- a/metapathways/MetaPathways_func_search.py +++ b/metapathways/MetaPathways_func_search.py @@ -13,6 +13,8 @@ import glob import os import pandas as pd + import tempfile + import shlex from os import path, _exit, rename from optparse import OptionParser, OptionGroup @@ -209,72 +211,52 @@ def main(argv, errorlogger=None, runcommand=None, runstatslogger=None): def _execute_FAST(options, logger=None): - - volumes = 0 - if options.run_mode == 'pervol': - # open *.prj file to check if there are multiple volumes - # if there are then use the per-volume function - with open(options.last_db + '.prj', 'r') as prj_in: - dat_rl = prj_in.readlines() - for line in dat_rl: - if 'volumes=' in line: - volumes = int(line.split('=')[1].strip('\n')) - - # create argument list(s), depending on the number of volumes - args_list = [] - if volumes > 0: - for v in list(range(volumes)): - args = [] - args.append(options.last_executable) - args += ["-f", options.last_f] - args += ["-o", options.last_o + str(v) + ".tmp"] - args += ["-P", options.num_threads] - args += [" -K", options.num_hits] - args += [options.last_db + str(v)] - args += [options.last_query] - args_list.append(args) - else: # if only one volume OR running in default mode - args = [] - args.append(options.last_executable) - args += ["-f", options.last_f] - args += ["-o", options.last_o + ".tmp"] - args += ["-P", options.num_threads] - args += [" -K", options.num_hits] - args += [options.last_db] - args += [options.last_query] - args_list.append(args) - - result = None + """FAST's time-seeded sort filenames must live in an invocation-private root.""" + output = os.path.abspath(options.last_o) try: - if len(args_list) == 1: - a = args_list[0] - result = sysutils.getstatusoutput(" ".join(a)) - rename(a[4], a[4].rsplit('.', 1)[0]) - else: - for a in args_list: - result = sysutils.getstatusoutput(" ".join(a)) - rename(a[4], a[4].rsplit('.', 1)[0]) - out_list = glob.glob(options.last_o + '*') - with open(options.last_o, 'w') as outfile: - for fname in out_list: - with open(fname) as infile: - for line in infile: - outfile.write(line) - os.remove(fname) - # sort the final table on ORF and Bitscore - last_df = pd.read_csv(options.last_o, sep='\t', header=None) - last_df.sort_values(by = [0, 11], ascending = [True, False], inplace=True) - last_df.to_csv(options.last_o, sep='\t', header=False, index=False) - except: - message = "Could not run FAST correctly" - if result and len(result) > 1: - message = result[1] + volumes = 0 + if options.run_mode == 'pervol': + with open(options.last_db + '.prj') as stream: + for line in stream: + if line.startswith('volumes='): + volumes = int(line.split('=', 1)[1]) + with tempfile.TemporaryDirectory(prefix='.fast-', dir=os.environ.get('METAPATHWAYS_COMPACT_SCRATCH') or os.path.dirname(output)) as work: + parts = [] + for v in range(volumes or 1): + part = os.path.join(work, f'hits-{v}.tsv') + db = options.last_db + str(v) if volumes else options.last_db + args = [options.last_executable, '-f', str(options.last_f), '-o', part, + '-P', str(options.num_threads), '-K', str(options.num_hits), + '-X', work, db, options.last_query] + result = sysutils.getstatusoutput(shlex.join(args)) + if result[0]: + return result + if not os.path.isfile(part): + raise RuntimeError('FAST returned success without producing its output') + parts.append(part) + if len(parts) == 1: + from metapathways.compact_storage import publish_file + publish_file(parts[0], output) + else: + merged = os.path.join(work, 'merged.tsv') + with open(merged, 'w') as destination: + for part in parts: + with open(part) as source: + for line in source: + destination.write(line) + if os.path.getsize(merged): + table = pd.read_csv(merged, sep='\t', header=None) + table.sort_values(by=[0, 11], ascending=[True, False], inplace=True) + table.to_csv(merged, sep='\t', header=False, index=False) + from metapathways.compact_storage import publish_file + publish_file(merged, output) + return (0, '') + except Exception as exc: + message = 'Could not run FAST correctly: ' + str(exc) if logger: - logger.printf("ERROR\t%s\n", message) + logger.printf('ERROR\t%s\n', message) return (1, message) - return (result[0], result[1]) - def _execute_BLAST(options, logger=None): args = [] diff --git a/metapathways/MetaPathways_parse_blast.py b/metapathways/MetaPathways_parse_blast.py index d830e38..7ec0ac2 100644 --- a/metapathways/MetaPathways_parse_blast.py +++ b/metapathways/MetaPathways_parse_blast.py @@ -770,20 +770,17 @@ def process_blastoutput( gutils.fprintf(outputfile, "\t%s", field) gutils.fprintf(outputfile, "\n") - pattern = re.compile(r"" + "(\d+_\d+)$") - count = 0 - uniques = {} + uniques = set() for data in blastparser: if not data: continue try: gutils.fprintf(outputfile, "%s", data["query"]) - result = pattern.search(data["query"]) - if result: - name = result.group(1) - uniques[name] = True + # Query identifiers are opaque: current C1-G1 identifiers and + # historical/custom identifiers must be counted without rewriting. + uniques.add(data["query"]) except: print("data is : ", data, "\n") return count, len(uniques) diff --git a/metapathways/MetaPathways_rRNA_stats_calculator.py b/metapathways/MetaPathways_rRNA_stats_calculator.py index 786931f..f5cdd73 100644 --- a/metapathways/MetaPathways_rRNA_stats_calculator.py +++ b/metapathways/MetaPathways_rRNA_stats_calculator.py @@ -131,7 +131,7 @@ def append_taxonomic_information(databaseSequences, table, params): for key in table: key = str(key) if ( - int(table[key][5] - table[key][4]) > params["length"] + abs(table[key][5] - table[key][4]) + 1 >= params["length"] and table[key][0] > params["similarity"] and table[key][1] < params["evalue"] and table[key][2] > params["bitscore"] @@ -144,7 +144,25 @@ def append_taxonomic_information(databaseSequences, table, params): table[key].append("-") -def process_blastout_file(blast_file, database, table, subunit, query_fna, errorlogger=None): +def rrna_feature_names(query_gff): + """Map bedtools zero-based coordinate IDs to authoritative GFF features.""" + features = {} + if query_gff: + with open(query_gff) as handle: + for line in handle: + if line.startswith('#') or not line.strip(): + continue + fields = line.rstrip('\n').split('\t') + if len(fields) != 9: + raise ValueError('Malformed rRNA GFF record') + if fields[2] == 'rRNA': + key = '{}:{}-{}({})'.format(fields[0], int(fields[3])-1, + int(fields[4]), fields[6]) + features[key] = fields[8] + return features + + +def process_blastout_file(blast_file, database, table, subunit, query_fna, errorlogger=None, query_gff=None): try: blastfile = open(blast_file, "r") except IOError: @@ -175,6 +193,7 @@ def process_blastout_file(blast_file, database, table, subunit, query_fna, error for x in queryseqs.readlines() if x[0] == '>' } queryseqs.close() + features = rrna_feature_names(query_gff) for line in blastLines: line = line.strip() @@ -190,13 +209,16 @@ def process_blastout_file(blast_file, database, table, subunit, query_fna, error end_pos = int(fields[7].strip()) e_value = float(fields[10].strip()) bitscore = float(fields[11].strip()) - length = end_pos - start_pos + 1 - if subunit in query_id: # check subunit + length = abs(end_pos - start_pos) + 1 + # Recent bedtools headers omit the gene name. Resolve their exact + # coordinates against barrnap's GFF; retain legacy named headers. + feature = features.get(query_key.split('::')[-1], query_id) + if re.search(r'(?= t_bitscore) & (length > t_length)): # only replace if bitscore is better AND alignment is longer table[query_key] = [percent_id, @@ -281,6 +303,8 @@ def createParser(): metavar="NUC_SEQUENCES", help="Query nucleotide sequences", ) + input_group.add_option('--query-gff', dest='query_gff', + help='Barrnap GFF defining query rRNA subunits') parser.add_option_group(input_group) @@ -395,6 +419,7 @@ def main(argv, errorlogger=None, runcommand=None, runstatslogger=None): options.subunit, options.query, errorlogger=errorlogger, + query_gff=options.query_gff, ) priority = 7000 diff --git a/metapathways/MetaPathways_refscore.py b/metapathways/MetaPathways_refscore.py index 30bed7d..b0181e5 100644 --- a/metapathways/MetaPathways_refscore.py +++ b/metapathways/MetaPathways_refscore.py @@ -9,6 +9,8 @@ import traceback import os import re + import tempfile + import shlex from os import makedirs, sys, remove, rename from sys import path @@ -169,13 +171,12 @@ def blast_against_itself(blast_executable, seq_subset_file, blast_table_out): def last_against_itself(last_executable, seq_subset_file, last_table_out): dirname = os.path.dirname(seq_subset_file.name) - cmd = "%s -o %s -f 0 %s %s" % ( - last_executable, - last_table_out, - dirname + PATHDELIM + "subset_db", - seq_subset_file.name, - ) - result = sysutils.getstatusoutput(cmd) + with tempfile.TemporaryDirectory(prefix='.fast-', dir=os.environ.get('METAPATHWAYS_COMPACT_SCRATCH') or os.path.abspath(dirname)) as work: + cmd = shlex.join([last_executable, '-o', last_table_out, '-f', '0', '-X', work, + dirname + PATHDELIM + 'subset_db', seq_subset_file.name]) + result = sysutils.getstatusoutput(cmd) + if result[0]: + raise RuntimeError('FAST reference-score search failed: ' + result[1]) def add_last_refscore_to_file(blast_table_out, refscore_file, allNames): diff --git a/metapathways/MetaPathways_tpm.py b/metapathways/MetaPathways_tpm.py index 9e5f1db..c0817eb 100644 --- a/metapathways/MetaPathways_tpm.py +++ b/metapathways/MetaPathways_tpm.py @@ -15,6 +15,8 @@ import multiprocessing import shutil import subprocess + import shlex + import os from os import path, _exit, rename, system @@ -180,7 +182,8 @@ def runUsingBWA(bwaExec, sample_name, indexFile, readgroup, readFiles, bwaFolder readFiles[0], ) - st_cmd = "samtools sort -O bam -o %s -T %s %s" % ( + st_cmd = "samtools sort -@ %d -O bam -o %s -T %s %s" % ( + max(0, int(num_threads) - 1), stOutputTmp, stInterTmp, bwaOutput, @@ -244,10 +247,32 @@ def getReadFiles(readdir, sample_name): return fastqgroups +def coverm_read_arguments(forward, reverse=None, interleaved=False): + """Validate read layout, including the legacy serialized ``None`` value.""" + forward = None if forward in (None, '', 'None') else forward + reverse = None if reverse in (None, '', 'None') else reverse + if not forward: + raise ValueError('A forward, single-end, or interleaved FASTQ is required.') + if interleaved: + if reverse: + raise ValueError('Interleaved input cannot also specify a reverse FASTQ.') + return ['--interleaved', forward] + if reverse: + if path.realpath(forward) == path.realpath(reverse) or ( + path.exists(forward) and path.exists(reverse) and path.samefile(forward, reverse)): + raise ValueError('Paired FASTQs must be different files; use --interleaved for one interleaved file.') + return ['-1', forward, '-2', reverse] + return ['--single', forward] + + def main(argv, errorlogger=None, runcommand=None, runstatslogger=None): parser = createParser() options, args = parser.parse_args(argv) + try: + read_arguments = coverm_read_arguments(options.fwd_fastq, options.rev_fastq, options.interleaved) + except ValueError as error: + parser.error(str(error)) if not (options.contigs != None and path.exists(options.contigs)): parser.error("ERROR\tThe contigs file is missing") errormod.insert_error(10) @@ -304,34 +329,25 @@ def main(argv, errorlogger=None, runcommand=None, runstatslogger=None): command.append('-m trimmed_mean') command.append('-m rpkm') command.append('-m tpm') - # set up fastqs - if options.fwd_fastq and options.rev_fastq: - command.append('-1') - command.append(options.fwd_fastq) - command.append('-2') - command.append(options.fwd_fastq) - elif options.fwd_fastq and options.interleaved: - command.append('--interleaved') - command.append(options.fwd_fastq) - elif options.fwd_fastq and not options.interleaved: - command.append('--single') - command.append(options.fwd_fastq) - else: - parser.error("ERROR\tFASTQs not specified correctly.") - errormod.insert_error(10) - return 1 + command.extend(shlex.quote(str(arg)) for arg in read_arguments) command.append('-r') command.append(options.contigs) command.append('--output-file') command.append(options.output) command.append('-t') - command.append(options.num_threads) + command.append(str(options.num_threads)) command.append('--bam-file-cache-directory') command.append(options.bwaFolder) command.append('--min-read-percent-identity 97') command.append('--min-read-aligned-percent 97') command.append('--exclude-supplementary') + gff_in = options.orfgff + gtf_out = path.join(path.dirname(gff_in), path.basename(gff_in).rsplit('.', 1)[0] + '.gtf') + # Reject malformed/duplicate annotation IDs before expensive mapping. + with open(options.stats, 'w') as stat_out: + run_logged(['gff2gtf.py', '--feature', 'CDS,rRNA,tRNA,pseudogene', gff_in, gtf_out], stat_out) + rpkmstatus = 0 rpkmtext = '' try: @@ -340,57 +356,41 @@ def main(argv, errorlogger=None, runcommand=None, runstatslogger=None): rpkmstatus = 1 pass - with open(options.stats, 'w') as stat_out: + with open(options.stats, 'a') as stat_out: stat_out.write(' '.join(command)) stat_out.write('\n\n') stat_out.write(rpkmtext) stat_out.write('\n\n') + # A failed mapping must not trigger abundance calculations on cached BAMs. + if rpkmstatus != 0: + gutils.eprintf("ERROR\tRPKM calculation was unsuccessful\n") + errormod.insert_error(10) + return 1 + #Start custom abundance analysis - bam_in = glob.glob(options.bwaFolder + PATHDELIM + '*.bam')[0] + bam_in = path.join(options.bwaFolder, path.basename(options.contigs) + '.' + path.basename(options.fwd_fastq) + '.bam') + if not path.isfile(bam_in): + raise RuntimeError(f'CoverM did not create the expected BAM: {bam_in}') sort_out = options.bwaFolder + PATHDELIM + options.sample_name + '.sorted.bam' tmp_bam = options.bwaFolder + PATHDELIM + options.sample_name + '.tmp' - gff_in = options.orfgff - gtf_out = path.join(path.dirname(options.orfgff), path.basename(options.orfgff).rsplit('.', 1)[0] + '.annot.gtf') + if os.environ.get('METAPATHWAYS_COMPACT_SCRATCH'): + tmp_bam = path.join(os.environ['METAPATHWAYS_COMPACT_SCRATCH'], options.sample_name + '.sort') feat = options.bwaFolder + PATHDELIM + options.sample_name + '.annot.featurecounts.txt' - gen_lens = options.bwaFolder + PATHDELIM + options.sample_name + '.annot.genelengths.txt' abund_calc = path.join(path.dirname(options.output), path.basename(options.output).rsplit('.', 2)[0] + '.orf_counts.tsv') - samcmd = ['samtools', 'sort', '-n', '-T', f'{tmp_bam}', '-o', f'{sort_out}', f'{bam_in}'] - stat_out.write(' '.join(samcmd)) - stat_out.write('\n\n') - sam_result = subprocess.Popen(' '.join(samcmd), stdout=stat_out, stderr=subprocess.PIPE, shell=True) - stat_out.write('\n') - sam_result.wait() - - gtfcmd = ['gff2gtf.py', '--feature', 'CDS,rRNA,tRNA,pseudogene', f'{gff_in}', f'{gtf_out}'] - stat_out.write(' '.join(gtfcmd)) - stat_out.write('\n') - gtf_result = subprocess.Popen(' '.join(gtfcmd), stdout=stat_out, stderr=subprocess.PIPE, shell=True) - stat_out.write('\n') - gtf_result.wait() - - featcmd = ['featureCounts', '-t', 'CDS,rRNA,tRNA,pseudogene', '-p', '-O', '-T', f'{options.num_threads}', '-a', f'{gtf_out}', '-o', f'{feat}', f'{sort_out}'] - stat_out.write(' '.join(featcmd)) - stat_out.write('\n') - feat_result = subprocess.Popen(' '.join(featcmd), stdout=stat_out, stderr=subprocess.PIPE, shell=True) - stat_out.write('\n') - feat_result.wait() - - cutcmd = f'cut -f4,5,9 {gtf_out} | sed \'s/gene_id //g\' | gawk \'{{print $3,$2-$1+1}}\' | tr \' \' \'\t\' > {gen_lens}' - stat_out.write(cutcmd) - stat_out.write('\n') - cut_result = subprocess.Popen(cutcmd, stdout=stat_out, stderr=subprocess.PIPE, shell=True) - stat_out.write('\n') - cut_result.wait() - - abuncmd = ['abund_calc.py', '--counts-file', f'{feat}', '--gene-lengths-file', f'{gen_lens}', '--gtf-file', f'{gtf_out}', '--output', f'{abund_calc}'] - stat_out.write('\n') - stat_out.write(' '.join(abuncmd)) - stat_out.write('\n') - abund_result = subprocess.Popen(' '.join(abuncmd), stdout=stat_out, stderr=subprocess.PIPE, shell=True) - stat_out.write('\n') - abund_result.wait() + # samtools -@ counts additional threads, beyond the main thread. + samcmd = ['samtools', 'sort', '-@', str(max(0, int(options.num_threads) - 1)), '-n', '-T', f'{tmp_bam}', '-o', f'{sort_out}', f'{bam_in}'] + run_logged(samcmd, stat_out) + + featcmd = ['featureCounts', '-t', 'CDS,rRNA,tRNA,pseudogene', '-O', '-T', str(options.num_threads), + '-a', gtf_out, '-o', feat, sort_out] + if read_arguments[0] != '--single': + featcmd.insert(1, '-p') + run_logged(featcmd, stat_out) + + abuncmd = ['abund_calc.py', '--counts-file', feat, '--gtf-file', gtf_out, '--output', abund_calc] + run_logged(abuncmd, stat_out) if rpkmstatus != 0: gutils.eprintf("ERROR\tRPKM calculation was unsuccessful\n") @@ -401,6 +401,23 @@ def main(argv, errorlogger=None, runcommand=None, runstatslogger=None): return rpkmstatus +def run_logged(command, log): + """Stream both output channels and stop immediately on a failed subcommand.""" + display = shlex.join(command) + print('Command: ' + display, flush=True) + log.write(display + '\n') + log.flush() + with subprocess.Popen(command, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, + text=True, errors='replace') as process: + for line in process.stdout: + print(line, end='', flush=True) + log.write(line) + log.flush() + code = process.wait() + if code: + raise subprocess.CalledProcessError(code, command) + + def runRPKMCommand(runcommand=None): if runcommand == None: return False @@ -505,7 +522,7 @@ def MetaPathways_tpm(argv, extra_command=None, errorlogger=None, runstatslogger= if errorlogger != None: errorlogger.write("#STEP\tRPKM_CALCULATION\n") try: - main( + status = main( argv, errorlogger=errorlogger, runcommand=extra_command, @@ -515,7 +532,7 @@ def MetaPathways_tpm(argv, extra_command=None, errorlogger=None, runstatslogger= errormod.insert_error(10) return (1, traceback.print_exc(10)) - return (0, "") + return (status, "") if __name__ == "__main__": diff --git a/metapathways/_version.py b/metapathways/_version.py index c2ed4d8..740e66c 100644 --- a/metapathways/_version.py +++ b/metapathways/_version.py @@ -1,5 +1,5 @@ __author__ = "Ryan J. McLaughlin, Tony X. Liu, Tomer Altman, Aditi N. Nallan, Aria S. Hahn, Julia Anstett, Connor Morgan-Lang, Kishori M. Konwar, Steven J. Hallam" -__version__ = "3.5.1" +__version__ = "4.0.0" __maintainer__ = "Ryan J. McLaughlin" __contact__ = "mcglock@student.ubc.ca" __status__ = "Release" diff --git a/metapathways/analysis_workflow.py b/metapathways/analysis_workflow.py new file mode 100644 index 0000000..7be1a52 --- /dev/null +++ b/metapathways/analysis_workflow.py @@ -0,0 +1,444 @@ +"""Strict input discovery and one shared DAG for complete multi-sample analyses.""" +import argparse +import copy +import csv +import fcntl +import gzip +import io +import json +import os +from pathlib import Path +import re +import shlex +import shutil +import sys + +from metapathways import nextflow + +FIELDS = ('sample_id', 'assembly', 'read_layout', 'reads_1', 'reads_2', 'mag_map') +GUIDE = ('https://hallamlab-metapathways.readthedocs.io/en/latest/analysis.html#analysis-wf-input-layout ' + 'and https://hallamlab-metapathways.readthedocs.io/en/latest/analysis.html#custom-analysis-manifest') +FASTA = re.compile(r'(.+)\.(?:fasta|fna|fa)(?:\.gz)?$') +READS = re.compile(r'(.+)_(R1|R2|interleaved|single)\.(?:fastq|fq)(?:\.gz)?$') +SAFE = re.compile(r'[A-Za-z][A-Za-z0-9_]*\Z') +RESERVED = {'logs', 'reports', 'inputs', 'assemblies', 'reads', 'mag_maps'} + + +def fail(message): + raise ValueError(f'{message}\nSee {GUIDE}') + + +def parser(): + from metapathways.pipeline import runParser + p = runParser('analysis_wf') + sub = next(a for a in p._actions if isinstance(a, argparse._SubParsersAction)).choices['analysis_wf'] + sub.description = 'Annotate multiple metagenomes, split MAGs, build PGDBs and generate a combined report.' + for action in sub._actions: + if action.dest in ('fwd_fastq', 'rev_fastq', 'interleaved', 'samples', 'test'): + action.help = argparse.SUPPRESS + elif action.dest == 'input_file': + action.help = 'dataset root, or assemblies directory with --reads_dir/--mag_maps_dir; alternative: --manifest' + + sub.add_argument('--manifest', help='TSV with sample_id, assembly, read_layout, reads_1, reads_2, mag_map') + sub.add_argument('--reads_dir', help='Flat reads directory; -i then names the assemblies directory') + sub.add_argument('--mag_maps_dir', help='Flat contig-to-MAG map directory; -i then names the assemblies directory') + sub.add_argument('--no_reads', action='store_true', help='Explicitly omit read mapping during automatic discovery') + sub.add_argument('--no_mags', action='store_true', help='Explicitly omit MAG splitting during automatic discovery') + sub.add_argument('--skip_ptools', action='store_true', help='Omit community and MAG PGDB construction') + sub.add_argument('--compact_results', action='store_true', + help='Use task scratch, archive PGDBs/diagnostics, and remove completed sample intermediates') + sub.add_argument('--scratch_dir', help='Worker-local scratch directory for compact mode [Slurm: SLURM_TMPDIR; local: system temporary directory]') + sub.add_argument('--image', help='Pathway Tools SIF [registered by build_pt]') + from metapathways.pt_taxonomy import add_taxonomy_options + add_taxonomy_options(sub) + sub.add_argument('--no_transport_inference', action='store_true', help='Disable TIP transport inference') + from metapathways.pt_taxonomy import add_pruning_options + add_pruning_options(sub) + sub.add_argument('--ptools_memory', type=nextflow.memory, default=None, help='Optional PGDB memory override [same as --memory]') + return p + + +def files(directory): + directory = Path(directory).expanduser().resolve() + if not directory.is_dir(): + fail(f'Missing input directory: {directory}') + result = [] + for p in sorted(directory.iterdir()): + if p.name.startswith('.'): + continue + if not p.is_file(): + fail(f'Expected a flat input directory; unexpected entry: {p}') + result.append(p) + return result + + +def discover(args): + root = Path(args.input_file).expanduser().resolve() + separate = bool(args.reads_dir or args.mag_maps_dir) + assembly_dir = root if separate else root / 'assemblies' + if not separate: + if not root.is_dir(): + fail(f'Missing dataset directory: {root}') + for p in root.iterdir(): + if not p.name.startswith('.') and p.name not in ('assemblies', 'reads', 'mag_maps'): + fail(f'Unexpected dataset entry: {p}') + rows = {} + for p in files(assembly_dir): + match = FASTA.fullmatch(p.name) + if not match: + fail(f'Unrecognized assembly filename: {p}') + sample = match[1] + if sample in rows: + fail(f'Multiple assemblies match sample {sample}') + rows[sample] = dict(zip(FIELDS, (sample, str(p), 'none', '', '', ''))) + if args.no_reads and args.reads_dir or args.no_mags and args.mag_maps_dir: + fail('An input directory conflicts with its --no_reads/--no_mags flag') + if not args.no_reads: + groups = {} + for p in files(args.reads_dir or root / 'reads'): + match = READS.fullmatch(p.name) + if not match or match[1] not in rows: + fail(f'Unrecognized or unmatched reads: {p}') + group = groups.setdefault(match[1], {}) + if match[2] in group: + fail(f'Multiple read files match {match[1]} / {match[2]}') + group[match[2]] = str(p) + for sample, row in rows.items(): + group = groups.get(sample, {}) + if set(group) == {'R1', 'R2'}: + row.update(read_layout='paired', reads_1=group['R1'], reads_2=group['R2']) + elif set(group) in ({'single'}, {'interleaved'}): + mode = next(iter(group)) + row.update(read_layout=mode, reads_1=group[mode]) + else: + fail(f'Missing mates or ambiguous read layout for {sample}: {list(group)}') + if not args.no_mags: + for p in files(args.mag_maps_dir or root / 'mag_maps'): + if p.suffix != '.tsv' or p.stem not in rows: + fail(f'Unrecognized or unmatched MAG map: {p}') + rows[p.stem]['mag_map'] = str(p) + missing = [s for s, row in rows.items() if not row['mag_map']] + if missing: + fail(f'Missing MAG maps for: {", ".join(missing)}') + return list(rows.values()) + + +def read_manifest(filename): + source = Path(filename).expanduser().resolve() + with source.open(newline='') as handle: + reader = csv.DictReader(handle, delimiter='\t') + if reader.fieldnames != list(FIELDS): + fail(f'Manifest header must be exactly: {", ".join(FIELDS)}') + rows = [] + for line, row in enumerate(reader, 2): + if None in row or any(v is None for v in row.values()): + fail(f'Manifest line {line} must contain exactly six tab-separated fields') + for key in ('assembly', 'reads_1', 'reads_2', 'mag_map'): + if row[key]: + p = Path(row[key]).expanduser() + row[key] = str((source.parent / p).resolve()) + rows.append(row) + return rows + + +def map_entities(filename): + entities, contigs = {}, set() + with open(filename) as handle: + for line, text in enumerate(handle, 1): + fields = text.rstrip('\r\n').split('\t') + if len(fields) != 2 or not all(fields) or any(x != x.strip() for x in fields): + fail(f'{filename}:{line}: expected headerless contig_idmag_id') + contig, original = fields + normalized = original.replace('.', '_') + if not SAFE.fullmatch(normalized) or normalized == 'community' or 'non_binned' in normalized: + fail(f'{filename}:{line}: unsupported/reserved MAG ID {original!r}') + if normalized in entities and entities[normalized] != original: + fail(f'{filename}: MAG IDs collide after replacing periods with underscores: {original}') + if contig in contigs: + fail(f'{filename}:{line}: repeated contig assignment {contig}') + contigs.add(contig) + entities[normalized] = original + if not entities: + fail(f'MAG map is empty: {filename}') + return sorted(entities), contigs + + +def validate(rows): + if not rows: + fail('No samples were found') + ids, used = set(), {} + for row in rows: + sample = row['sample_id'] + if not SAFE.fullmatch(sample) or sample.lower() in RESERVED or sample in ids: + fail(f'Invalid, reserved, or duplicate sample ID: {sample!r}; use letters, digits and underscores, starting with a letter') + ids.add(sample) + mode, r1, r2 = row['read_layout'], row['reads_1'], row['reads_2'] + if mode not in ('paired', 'single', 'interleaved', 'none') or \ + (mode == 'paired' and not (r1 and r2)) or \ + (mode in ('single', 'interleaved') and (not r1 or r2)) or \ + (mode == 'none' and (r1 or r2)): + fail(f'{sample}: read_layout and read paths disagree') + for key in ('assembly', 'reads_1', 'reads_2', 'mag_map'): + if not row[key]: + if key == 'assembly': + fail(f'{sample}: missing assembly') + continue + p = Path(row[key]).expanduser().resolve() + if not p.is_file() or p.stat().st_size == 0 or not os.access(p, os.R_OK): + fail(f'{sample}: missing, empty or unreadable {key}: {p}') + stat = p.stat() + identity = (stat.st_dev, stat.st_ino) + if identity in used: + fail(f'{sample}: {key} reuses the same file as {used[identity]}') + used[identity] = f'{sample}/{key}' + row[key] = str(p) + if not FASTA.fullmatch(Path(row['assembly']).name): + fail(f'{sample}: assembly must end in .fa, .fna or .fasta, optionally .gz') + row['entities'], mapped = map_entities(row['mag_map']) if row['mag_map'] else ([], set()) + # Check map membership before submitting any jobs. FASTQs are not fully scanned. + found = set() + opener = gzip.open if row['assembly'].endswith('.gz') else open + with opener(row['assembly'], 'rt') as handle: + first = handle.readline() + if not first.startswith('>'): + fail(f'{sample}: assembly is not FASTA') + def header(text): + tokens = text[1:].split() + if not tokens: + fail(f'{sample}: empty FASTA identifier') + if tokens[0] in found: + fail(f'{sample}: duplicate FASTA identifier {tokens[0]}') + found.add(tokens[0]) + header(first) + for text in handle: + if text.startswith('>'): + header(text) + unknown = mapped - found + if unknown: + fail(f'{sample}: MAG map contigs absent from assembly: {", ".join(sorted(unknown)[:5])}') + return sorted(rows, key=lambda row: row['sample_id']) + + +def save_inputs(rows, output): + text = io.StringIO() + writer = csv.DictWriter(text, FIELDS, delimiter='\t', lineterminator='\n', extrasaction='ignore') + writer.writeheader() + writer.writerows(rows) + manifest = output / 'inputs.resolved.tsv' + if manifest.exists() and manifest.read_text() != text.getvalue(): + fail(f'{manifest} describes different inputs; use a new output directory') + entities_file = output / 'inputs.entities.json' + entities = {row['sample_id']: row['entities'] for row in rows} + if entities_file.exists() and json.loads(entities_file.read_text()) != entities: + fail('The set of MAG IDs changed; use a new output directory') + if not manifest.exists(): + # Refuse to combine an unrelated existing analysis with a new dataset. + if any(p.is_dir() and not p.name.startswith('.') and p.name not in ('logs',) for p in output.iterdir()): + fail('Use a new output directory for the first analysis_wf invocation') + manifest.write_text(text.getvalue()) + if not entities_file.exists(): + entities_file.write_text(json.dumps(entities, indent=2) + '\n') + staging = output / '.metapathways/analysis_wf/inputs' + staging.mkdir(parents=True, exist_ok=True) + staged = [] + for row in rows: + item = dict(row) + for key in ('assembly', 'reads_1', 'reads_2', 'mag_map'): + if not row[key]: + continue + directory = staging / row['sample_id'] / key + directory.mkdir(parents=True, exist_ok=True) + suffix = ('.fasta' if key == 'assembly' else '.tsv' if key == 'mag_map' else '.fastq') + if row[key].endswith('.gz'): + suffix += '.gz' + alias = directory / (row['sample_id'] + suffix) + if alias.is_symlink(): + if alias.resolve() != Path(row[key]): + fail(f'Staged input changed unexpectedly: {alias}') + elif alias.exists(): + fail(f'Unexpected file at staged input path: {alias}') + else: + alias.symlink_to(row[key]) + item[key] = str(alias) + staged.append(item) + return staged + + +def downstream(row, output, annotations, args, image): + sample = row['sample_id'] + base = output / sample + # Pathologic inputs follow annotation reports in the annotation DAG. + parent = next(t['id'] for t in annotations if t.get('context', {}).get('name') == 'PATHOLOGIC_INPUT') + tasks = [] + if row['mag_map']: + ms = base / 'magsplitter' + pf, orfmap = base / 'ptools/0.pf', base / 'ptools/orf_map.txt' + table = base / f'results/annotation_table/{sample}.ORF_annotation_table.txt' + feature_table = base / f'results/annotation_table/{sample}.ptinput.tsv' + mapping = base / f'preprocessed/{sample}.mapping.txt' + saved = ms / 'contig_to_mag.tsv' + cmd = ['magsplitter', '-p', str(pf), '-r', str(orfmap), '-c', str(table), + '-m', row['mag_map'], '-i', str(mapping), '-o', str(ms)] + copy_cmd = [sys.executable, '-c', 'import shutil,sys; shutil.copyfile(*sys.argv[1:])', row['mag_map'], str(saved)] + tasks.append(nextflow.task(f'{sample}:mag_split', f'{sample}:mag_split', + [shlex.join(['mkdir', '-p', str(ms)]), shlex.join(cmd), shlex.join(copy_cmd)], + [str(pf), str(orfmap), str(table), str(mapping), row['mag_map'], str(feature_table)], + [str(ms / 'results'), str(saved)], [parent], memory=args.memory, sample=sample, adopt_existing=False, + cache_version='authoritative-mag-coordinates-v1')) + if args.skip_ptools: + return tasks + script = shutil.which('pgdb_build_wf.py') or str(Path(__file__).resolve().parents[1] / 'dev/pgdb_build_wf.py') + if not Path(script).is_file(): + fail('pgdb_build_wf.py is missing; reinstall MetaPathways') + for entity in ['community'] + row['entities']: + community = entity == 'community' + inputs = base / 'ptools' if community else base / 'magsplitter/results' / entity + results = base / 'results/pgdb/community' if community else base / 'results/pgdb/MAGs' / entity + tag = sample if community else entity + cmd = [sys.executable, script, '--mp_out', str(base), '--tag', sample, '--entity', entity, '--image', image] + if getattr(args, 'compact_results', False): + cmd.append('--compact_results') + cmd.append('--taxprune' if args.taxprune else '--no_taxprune') + if args.no_transport_inference: + cmd.append('--no_transport_inference') + from metapathways.pt_taxonomy import resolve_taxon + taxon_id = resolve_taxon(args) + if taxon_id is not None: + cmd += ['--taxon_id', str(taxon_id)] + tasks.append(nextflow.task(f'{sample}:pgdb:{entity}', f'{sample}:pgdb:{entity}', [shlex.join(cmd)], + [image, str(inputs), str(base / f'results/annotation_table/{sample}.EC_RXN_map.tsv'), + str(base / f'results/annotation_table/{sample}.ptinput.tsv'), str(base / f'preprocessed/{sample}.fasta')], + [str(results / (tag + suffix)) for suffix in ('cyc.tar.bz2', '_pwy.tsv', '_pwy2orf.tsv')], + [parent if community else f'{sample}:mag_split'], cpus=1, memory=args.ptools_memory or args.memory, + sample=sample, entity=entity, allow_failure=not community, adopt_existing=False, + cache_version='sequence-backed-pgdb-compatibility-v5', + skip_if_missing=None if community else str(inputs / '0.pf'))) + from metapathways.pt_reactions import BLACKLIST + tasks[-1]['fingerprint_inputs'] = tasks[-1]['inputs'] + [str(BLACKLIST)] + [str(base / f'orf_prediction/{sample}.cds.gff')] + from metapathways.pt_reactions import compatibility_path + compatibility = Path(args.refdb_dir)/'functional_categories/ptools_reaction_compatibility.json' + if compatibility and compatibility.is_file(): + tasks[-1]['fingerprint_inputs'].append(str(compatibility)) + return tasks + + +def main(argv=None): + from metapathways.pipeline import prepare_annotation + from metapathways.pt_container import registered_image + p = parser() + args = p.parse_args(['analysis_wf'] + list(sys.argv[1:] if argv is None else argv)) + if args.scratch_dir and not args.compact_results: + fail('--scratch_dir requires --compact_results') + if args.compact_results and any(getattr(args, k, None) for k in ('keep_work', 'work_dir', 'conda_cache')): + fail('--compact_results cannot be combined with --keep_work, --work_dir or --conda_cache') + if not args.output_dir or not args.refdb_dir or not (args.input_file or args.manifest): + fail('Provide -o OUTPUT, -d MPDB and either -i INPUTS or --manifest FILE') + if args.manifest and any((args.input_file, args.reads_dir, args.mag_maps_dir, args.no_reads, args.no_mags)): + fail('--manifest cannot be combined with discovery flags') + if args.fwd_fastq or args.rev_fastq or args.interleaved or args.test or args.samples: + fail('Use per-sample reads and IDs from discovery/manifest, not -1/-2/--interleaved/--test/--samples') + if args.input_format != 'fasta': + fail('analysis_wf requires nucleotide FASTA assemblies') + if any(getattr(args, key) == 'skip' for key in ('PREPROCESS_INPUT', 'ORF_PREDICTION', 'FILTER_AMINOS', + 'FUNC_SEARCH', 'PARSE_FUNC_SEARCH', 'ANNOTATE_ORFS', 'CREATE_ANNOT_REPORTS', 'PATHOLOGIC_INPUT')): + fail('Required annotation stages cannot be skipped in analysis_wf; completed tasks are reused automatically') + output = Path(args.output_dir).expanduser().resolve() + refdb = Path(args.refdb_dir).expanduser().resolve() + if any(re.search(r'[^A-Za-z0-9_./-]', str(x)) for x in (output, refdb)): + fail('Output and MPDB paths must use letters, digits, underscores, hyphens, periods and slashes (legacy tool requirement)') + from metapathways._version import __version__ + print(f'RUNNING MetaPathways: v{__version__}', flush=True) + print(f'Output directory: {output}', flush=True) + print('Validating sample inputs, read layouts and genome maps...', flush=True) + rows = validate(read_manifest(args.manifest) if args.manifest else discover(args)) + from metapathways.compact_results import MARKER, resume_key, marker_state + for row in rows: + base = output/row['sample_id'] + if base.is_symlink(): + fail(f'Sample output must not be a symlink: {base}') + if (base/MARKER).exists() and not args.compact_results: + fail('This output contains compact samples; resume with --compact_results or use a new output directory') + if args.compact_results: + protected = [Path(r[k]).resolve() for r in rows + for k in ('assembly', 'reads_1', 'reads_2', 'mag_map') if r[k]] + protected += [refdb] + if args.image: + protected.append(Path(args.image).expanduser().resolve()) + if any(p == base or base in p.parents for p in protected): + fail(f'Compact mode requires inputs, MPDB and SIF outside sample output: {base}') + image = None + if not args.skip_ptools: + image = args.image or registered_image() + if not image or not Path(image).expanduser().is_file(): + fail('Build a Pathway Tools SIF with metapathways build_pt, pass --image, or explicitly --skip_ptools') + image = str(Path(image).expanduser().resolve()) + if args.compact_results and any((output/r['sample_id']) in Path(image).parents for r in rows): + fail('Compact mode requires the SIF outside sample output directories') + if not shutil.which('apptainer'): + fail('Apptainer is required for Pathway Tools') + if any(row['mag_map'] for row in rows) and not shutil.which('magsplitter'): + fail('MAGSplitter is required when MAG maps are supplied; see README installation instructions') + output.mkdir(parents=True, exist_ok=True) + control = output / '.metapathways/analysis_wf' + control.mkdir(parents=True, exist_ok=True) + with (control / 'planning.lock').open('a') as lock: + try: + fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError: + fail(f'Another analysis_wf invocation is using {output}') + print(f'Staging input links for {len(rows)} samples; references: {", ".join(args.annotation_dbs)}', flush=True) + staged = save_inputs(rows, output) + tasks = [] + for index, (original, row) in enumerate(zip(rows, staged), 1): + sample = row['sample_id'] + prefix = f'Planning [{index}/{len(rows)}] {sample}' + pgdbs = 'skipped' if args.skip_ptools else str(1 + len(row['entities'])) + print(f'{prefix}: reads={row["read_layout"]}; genome bins={len(row["entities"])}; ' + f'PGDBs={pgdbs}; compact={"yes" if args.compact_results else "no"}', flush=True) + first_task = len(tasks) + key = resume_key(args, original) if args.compact_results else None + state = marker_state(output/sample, key, args.force_redo) if args.compact_results else None + if state == 'complete': + print(f'{prefix}: completed compact results retained; no tasks scheduled', flush=True) + continue + if state == 'compacting': + tasks.append(compact_task(sample, output, key, [], args.memory)) + print(f'{prefix}: resuming interrupted cleanup only', flush=True) + continue + sample_args = copy.copy(args) + sample_args._analysis_planning = True + sample_args.input_file = row['assembly'] + sample_args.output_dir, sample_args.refdb_dir = str(output), str(refdb) + sample_args.fwd_fastq = row['reads_1'] or None + sample_args.rev_fastq = row['reads_2'] or None + sample_args.interleaved = row['read_layout'] == 'interleaved' + annotation, _ = prepare_annotation(sample_args, p) + # A combined analysis only reuses outputs whose receipts prove provenance. + for task in annotation: + task['adopt_existing'] = False + tasks.extend(annotation) + tasks.extend(downstream(row, output, annotation, args, image)) + if args.compact_results: + dependencies = [t['id'] for t in tasks if t.get('sample') == sample] + tasks.append(compact_task(sample, output, key, dependencies, args.memory)) + print(f'{prefix}: ready; {len(tasks) - first_task} tasks planned', flush=True) + if args.force_redo: + for task in tasks: + if task['status'] != 'skip': + task['status'] = 'redo' + print(f'Validated {len(rows)} samples. Resolved inputs: {output / "inputs.resolved.tsv"}', flush=True) + print(f'Planning complete: {len(tasks)} tasks.', flush=True) + if tasks: + nextflow.launch(tasks, output, args, 'analysis_wf', dryrun=args.dryrun) + if args.compact_results and not args.dryrun: + # No sample needs staged input links or MP task receipts once complete. + for name in ('inputs', 'receipts', 'tmp'): + shutil.rmtree(control/name, ignore_errors=True) + + +def compact_task(sample, output, key, dependencies, memory): + from metapathways.compact_results import MARKER + command = shlex.join([sys.executable, '-m', 'metapathways.compact_results', str(output/sample), key]) + return nextflow.task(f'{sample}:compact_results', f'{sample}:compact_results', [command], + outputs=[str(output/sample/MARKER)], dependencies=dependencies, + sample=sample, memory=memory, adopt_existing=False) diff --git a/metapathways/build_DBs/metacyc_build_ont.py b/metapathways/build_DBs/metacyc_build_ont.py index 7e90e9e..17ee3cc 100644 --- a/metapathways/build_DBs/metacyc_build_ont.py +++ b/metapathways/build_DBs/metacyc_build_ont.py @@ -56,8 +56,8 @@ print('Building MetaCyc Ontology...') ont_ids = {} ont_common = {} -stop_list = [] -for k in tqdm(class2types): +stop_list = set() +for k in tqdm(class2types, disable=not sys.stderr.isatty()): while k not in stop_list: if k in class2common: if k in ont_ids: @@ -80,11 +80,11 @@ else: new_maps.append(m) new_com_maps.append(com_m) - p_cnt =+ 1 + p_cnt += 1 ont_ids[k] = new_maps ont_common[k] = new_com_maps if ((map_cnt == p_cnt) | ('FRAMES' in parents)): - stop_list.append(k) + stop_list.add(k) else: parents = class2types[k] ont_ids[k] = [] @@ -96,9 +96,9 @@ class2common[k] ]) else: - stop_list.append(k) + stop_list.add(k) else: - stop_list.append(k) + stop_list.add(k) print('Saving MetaCyc Ontology...') path_ont_list = [] @@ -107,12 +107,13 @@ for path in path2types: for t in path2types[path]: for i,h in enumerate(ont_ids[t]): - c = ont_common[t][i] + h = list(h) + c = list(ont_common[t][i]) for p in h_prune_list: if p in h: h.remove(p) for p in c_prune_list: - if p in h: + if p in c: c.remove(p) common_name = path2common[path] path_ont_list.append([path, common_name, diff --git a/metapathways/build_DBs/metacyc_mapping_build.py b/metapathways/build_DBs/metacyc_mapping_build.py index 195ac84..71c344b 100755 --- a/metapathways/build_DBs/metacyc_mapping_build.py +++ b/metapathways/build_DBs/metacyc_mapping_build.py @@ -228,7 +228,7 @@ def build_rxnenz_dict(dat): c_enz_list = [] for c_tmp in com_flatlist_tmp: - e = prot2cat_dict[c][0] + e = prot2cat_dict[c_tmp][0] c_enz_list.extend(e) prot2enz_dict[m].extend(c_enz_list) @@ -245,6 +245,7 @@ def build_rxnenz_dict(dat): enzrxn_list.append([m, r]) enzrxns_df = pd.DataFrame(enzrxn_list, columns=['MC', 'RXN']) +enzrxns_df.drop_duplicates(inplace=True) enzrxns_df.to_csv(os.path.join(mc_dir, 'MetaCyc-monomer-rxn-pairs.tsv'), sep='\t', index=False ) @@ -263,8 +264,6 @@ def build_rxnenz_dict(dat): # Build Pathway to Compounds dictionary pwy_dict, pwy2cmp_dict = build_pwycmp_dict(pwy_dat, cmp_dict) -print(pwy_dict['METH-ACETATE-PWY']) -print(pwy2cmp_dict['METH-ACETATE-PWY']) # Create Pathway -> Reactions -> Compounds pwy_rxn_cmp_list = [] @@ -288,4 +287,3 @@ def build_rxnenz_dict(dat): pwy_rxn_cmp_df.to_csv(os.path.join(mc_dir, 'MetaCyc-PWY-RXN-CMP-map.tsv'), sep='\t', index=False ) - diff --git a/metapathways/cli_logging.py b/metapathways/cli_logging.py new file mode 100644 index 0000000..1fc76b0 --- /dev/null +++ b/metapathways/cli_logging.py @@ -0,0 +1,41 @@ +"""Keep the complete CLI transcript in the selected output directory.""" +from contextlib import contextmanager, redirect_stderr, redirect_stdout +from datetime import datetime, timezone +from pathlib import Path +import shlex +import sys +import uuid + + +class Tee: + def __init__(self, terminal, log): + self.terminal, self.log = terminal, log + + def write(self, value): + self.terminal.write(value) + self.log.write(value) + self.flush() + return len(value) + + def flush(self): + self.terminal.flush() + self.log.flush() + + def isatty(self): + return False + + def fileno(self): + return self.terminal.fileno() + + +@contextmanager +def transcript(directory, argv): + folder = Path(directory).expanduser().absolute() / 'logs' / 'cli' + folder.mkdir(parents=True, exist_ok=True) + stamp = datetime.now(timezone.utc).strftime('%Y%m%dT%H%M%SZ') + path = folder / (stamp + '-' + uuid.uuid4().hex[:8] + '.log') + with path.open('w') as log: + with redirect_stdout(Tee(sys.stdout, log)), redirect_stderr(Tee(sys.stderr, log)): + print('Command: ' + shlex.join(argv)) + print(f'CLI log: {path}') + yield diff --git a/metapathways/compact_results.py b/metapathways/compact_results.py new file mode 100644 index 0000000..ee51d7c --- /dev/null +++ b/metapathways/compact_results.py @@ -0,0 +1,139 @@ +"""Opt-in, terminal per-sample cleanup for analysis_wf. + +Keep report source files rather than a snapshot alone, so reports remain rebuildable. +Never follow directory symlinks when deleting generated output. +""" +import hashlib +import json +import os +from pathlib import Path +import sqlite3 +import sys + +from metapathways.nf_worker import atomic_json + +MARKER = 'compact-results.json' + + +def resume_key(args, row): + options = vars(args).copy() + # A compact sample cannot be silently reused under changed implementation. + code = hashlib.sha256() + for source in sorted(Path(__file__).parent.rglob('*.py')): + code.update(source.relative_to(Path(__file__).parent).as_posix().encode()) + code.update(source.read_bytes()) + from metapathways.pt_reactions import BLACKLIST + code.update(BLACKLIST.read_bytes()) + options['implementation_sha256'] = code.hexdigest() + for key in ('dryrun', 'force_redo', 'max_tasks', 'max_cpus', 'max_memory', + 'executor', 'account', 'partition', 'qos', 'reservation', + 'time_limit', 'submit_rate', 'scratch_dir'): + options.pop(key, None) + paths = [Path(row[k]) for k in ('assembly', 'reads_1', 'reads_2', 'mag_map') if row[k]] + from metapathways.pt_container import registered_image + image = getattr(args, 'image', None) or (None if args.skip_ptools else registered_image()) + if image: + paths.append(Path(image).expanduser()) + db = Path(args.refdb_dir).expanduser() + for folder in ('functional', 'functional_categories', 'taxonomic', 'ncbi_tree'): + paths.extend(p for p in (db/folder).rglob('*') if p.is_file()) + evidence = [(str(p.resolve()), p.stat().st_size, p.stat().st_mtime_ns) for p in sorted(set(paths))] + payload = json.dumps([options, row, evidence], sort_keys=True, default=str) + return hashlib.sha256(payload.encode()).hexdigest() + + +def marker_state(base, key, force=False): + marker = base/MARKER + if not marker.exists(): + return None + record = json.loads(marker.read_text()) + if force or record.get('resume_key') != key: + raise ValueError(f'{base} contains compact results; changed inputs/settings or --force_redo require a new output directory') + if record.get('state') not in ('compacting', 'complete'): + raise ValueError(f'Invalid compact-results marker: {marker}') + for relative in record['keep']: + if not (base/relative).is_file(): + raise ValueError(f'Compact result is missing {base/relative}; use a new output directory') + return record['state'] + + +def compact(base, key): + from metapathways.reporting import Importer, SCHEMA + base = Path(base) + if base.is_symlink() or not base.is_dir(): + raise ValueError(f'Expected a real sample output directory: {base}') + base = base.resolve() + state = marker_state(base, key) + if state == 'complete': + return + if state == 'compacting': + record = json.loads((base/MARKER).read_text()) + keep = set(record['keep']) + else: + for pattern in ('preprocessed/*.mapping.txt', 'results/annotation_table/*.functional_and_taxonomic_table.txt'): + if not list(base.glob(pattern)): + raise ValueError(f'Missing required report source {pattern} in {base}; cleanup cancelled') + # Validate all relationships and discover exactly what the explorer reads + # before deleting anything. Use disk-backed SQLite for large samples. + import tempfile + with tempfile.TemporaryDirectory(prefix='mp-compact-', dir=base) as tmp: + with sqlite3.connect(str(Path(tmp)/'check.sqlite')) as db: + db.execute('PRAGMA journal_mode=MEMORY') + db.execute('PRAGMA synchronous=OFF') + db.execute('PRAGMA foreign_keys=ON') + db.executescript(SCHEMA) + importer = Importer(db, base) + # Current-run summary is published only after the DAG finishes. + # Read completed PGDB receipts to exclude failed partial tables. + manifests = list((base.parent/'logs/analysis_wf').glob('*/tasks.json')) + if manifests: + manifest = max(manifests, key=lambda p: p.stat().st_mtime_ns) + for task in json.loads(manifest.read_text()): + if task.get('sample') == base.name and task.get('entity'): + receipt = Path(task['invocation_receipt']) + status = json.loads(receipt.read_text())['status'] + importer.pgdb_status[(base.name, task['entity'])] = status + importer.sample(base) + if db.execute('PRAGMA foreign_key_check').fetchone(): + raise ValueError(f'Invalid report relationships in {base}; cleanup cancelled') + keep = {p.relative_to(base).as_posix() for p in importer.source_stats} + # Retain final supporting tables and diagnostics, including failed MAG logs. + for directory, dirs, files in os.walk(base, followlinks=False): + dirs[:] = [d for d in dirs if not (Path(directory)/d).is_symlink()] + for name in files: + p = Path(directory)/name + rel = p.relative_to(base) + if (rel.parts[0] in ('logs', 'run_statistics', 'reports') or + (rel.parts[:2] == ('results', 'pgdb') and + (name.endswith('cyc.tar.bz2') or name.endswith('.tar.gz'))) or + p.suffix == '.log' or name.endswith('_log.txt') or + (rel.parts[0] == 'results' and p.suffix in ('.tsv', '.txt'))): + keep.add(rel.as_posix()) + for relative in keep: + p = base/relative + if p.is_symlink() or not p.is_file(): + raise ValueError(f'Report source must be a regular file before cleanup: {p}') + record = dict(state='compacting', resume_key=key, keep=sorted(keep), removed_bytes=0, removed_files=0) + atomic_json(base/MARKER, record) + keep.add(MARKER) + for directory, dirs, files in os.walk(base, topdown=False, followlinks=False): + for name in files: + p = Path(directory)/name + if p.relative_to(base).as_posix() not in keep: + if not p.is_symlink(): + record['removed_bytes'] += p.stat().st_size + p.unlink() + record['removed_files'] += 1 + for name in dirs: + p = Path(directory)/name + if p.is_symlink(): + p.unlink() + elif not any(p.iterdir()): + p.rmdir() + record['state'] = 'complete' + atomic_json(base/MARKER, record) + print(f"Compact results: {base.name}; removed {record['removed_files']} files ({record['removed_bytes']} bytes)", flush=True) + + +if __name__ == '__main__': + compact(Path(sys.argv[1]), sys.argv[2]) diff --git a/metapathways/compact_storage.py b/metapathways/compact_storage.py new file mode 100644 index 0000000..a706e15 --- /dev/null +++ b/metapathways/compact_storage.py @@ -0,0 +1,87 @@ +"""Task-local scratch and atomic archive publication for compact workflows.""" +from contextlib import contextmanager +import os +from pathlib import Path +import shutil +import tarfile +import tempfile +import uuid + + +def scratch_root(override=None): + # Resolve on the worker, never on the controller/login node. TMPDIR alone + # is not evidence of node-local storage on an arbitrary Slurm cluster. + candidate = override or os.environ.get('SLURM_TMPDIR') + if not candidate and os.environ.get('SLURM_JOB_ID'): + raise RuntimeError('Compact tasks require job-local scratch: set --scratch_dir ' + 'to your cluster node-local directory when SLURM_TMPDIR is unavailable') + root = Path(candidate or tempfile.gettempdir()).expanduser().resolve() + if not root.is_dir(): + raise RuntimeError(f'Scratch directory does not exist on this worker: {root}') + usage = shutil.disk_usage(root) + stats = os.statvfs(root) + print(f'Compact scratch: {root}; {usage.free / 2**30:.1f} GiB free; ' + f'{stats.f_favail} available inodes', flush=True) + if usage.free < 1024**3 or (stats.f_files and stats.f_favail < 1000): + raise RuntimeError(f'Insufficient scratch headroom at {root} (need at least 1 GiB and 1000 inodes)') + return root + + +@contextmanager +def task_scratch(override=None): + with tempfile.TemporaryDirectory(prefix='mp-task-', dir=scratch_root(override)) as work: + yield Path(work) + + +def publish_file(source, destination): + """Cross-filesystem copy, then atomic rename; never expose partial output.""" + source, destination = Path(source), Path(destination) + destination.parent.mkdir(parents=True, exist_ok=True) + temporary = destination.with_name('.' + destination.name + '.' + uuid.uuid4().hex + '.tmp') + try: + shutil.copyfile(source, temporary) + if temporary.stat().st_size != source.stat().st_size: + raise RuntimeError(f'Incomplete copy of {source}') + temporary.replace(destination) + finally: + temporary.unlink(missing_ok=True) + + +def archive_directory(source, destination): + """Bundle without deleting the source; callers delete only after success.""" + source, destination = Path(source), Path(destination) + destination.parent.mkdir(parents=True, exist_ok=True) + temporary = destination.with_name('.' + destination.name + '.' + uuid.uuid4().hex + '.tmp') + try: + with tarfile.open(temporary, 'w:gz', compresslevel=1, dereference=False) as archive: + archive.add(source, arcname=source.name) + temporary.replace(destination) + finally: + temporary.unlink(missing_ok=True) + + +def compact_pgdb(output, tag, build): + """Build and extract locally, publish report products, retain failure evidence.""" + output = Path(output) + output.mkdir(parents=True, exist_ok=True) + root = os.environ.get('METAPATHWAYS_COMPACT_SCRATCH') + if not root: + raise RuntimeError('Compact PGDB execution requires worker scratch setup') + with tempfile.TemporaryDirectory(prefix='pgdb-', dir=root) as temporary: + working = Path(temporary)/'result' + working.mkdir() + try: + build(str(working)) + names = [tag + suffix for suffix in ('cyc.tar.bz2', '_pwy.tsv', '_pwy2orf.tsv')] + for name in names: + if not (working/name).is_file() or not (working/name).stat().st_size: + raise RuntimeError(f'Missing PGDB result: {working/name}') + if (working/'diagnostics').exists(): + archive_directory(working/'diagnostics', output/'diagnostics.tar.gz') + for name in names: + publish_file(working/name, output/name) + except BaseException: + # Includes interrupted attempts when Python can still execute cleanup. + # A hard kill/node loss can only preserve already published records. + archive_directory(Path(temporary), output/f'failed-attempt-{uuid.uuid4().hex}.tar.gz') + raise diff --git a/metapathways/execution.py b/metapathways/execution.py index f5d49a8..fadcda7 100644 --- a/metapathways/execution.py +++ b/metapathways/execution.py @@ -13,6 +13,8 @@ import os import re import time + import importlib + import shlex from subprocess import Popen, PIPE, STDOUT from os import makedirs, listdir, _exit @@ -23,7 +25,6 @@ from metapathways import sysutil as sysutils from metapathways import general_utils as gutils - from metapathways import scripts as python_scripts except: print(""" Could not load some user defined module functions""") @@ -33,13 +34,13 @@ def execute_pipeline_stage(pipeline_command, \ extra_command=None, errorlogger=None, runstatslogger=None): - argv = [x.strip() for x in pipeline_command.split()] + argv = shlex.split(pipeline_command) funcname = re.sub(r".py$", "", argv[0]) funcname = re.sub(r"^.*/", "", funcname) args = argv[1:] - if hasattr(python_scripts, funcname): - methodtocall = getattr(getattr(python_scripts, funcname), funcname) + if funcname.startswith('MetaPathways_'): + methodtocall = getattr(importlib.import_module('metapathways.' + funcname), funcname) if extra_command == None: result = methodtocall( args, errorlogger=errorlogger, runstatslogger=runstatslogger) diff --git a/metapathways/jobscreator.py b/metapathways/jobscreator.py index 572a1d0..af128e5 100644 --- a/metapathways/jobscreator.py +++ b/metapathways/jobscreator.py @@ -29,7 +29,6 @@ PATHDELIM = sysutils.pathDelim() -@gutils.Singleton class Params: params = {} def __init__(self, params): @@ -52,13 +51,11 @@ def print_key_values(self): print(self.params) -@gutils.Singleton class Configs: def __init__(self, configs): for key, value in configs.items(): setattr(self, key, value) -@gutils.Singleton class ContextCreator: params = None configs = None @@ -261,7 +258,7 @@ def create_orf_prediction_cmd(self, s) : mode = self.params.get('ORF Prediction Arguments', 'orf_mode') - num_threads = str(self.configs.NUM_CPUS) + num_threads = self.configs.NUM_CPUS pyScript = self.configs.ORF_PREDICTION executable = self.configs.PRODIGAL_EXECUTABLE @@ -594,7 +591,7 @@ def create_scan_rRNA_seqs_cmd(self, s): if bar_exe == None: eprintf("ERROR\tCannot find barrnap\n") barnap_cmd = "%s --quiet --threads %s --outseq %s %s > %s"\ - %(bar_exe, str(num_threads), + %(bar_exe, num_threads, context.outputs['rRNA_barout_seq'], context.inputs['input_fasta'], context.outputs['rRNA_barout_gff']) context.commands = [barnap_cmd] @@ -619,7 +616,7 @@ def create_scan_rRNA_seqs_cmd(self, s): context = contextmod.Context() context.name = 'SCAN_rRNA:' + db - context.inputs = { 'rRNA_barout_seq':rRNA_barout_seq, 'dbsequences':dbsequences } + context.inputs = { 'rRNA_barout_seq':rRNA_barout_seq, 'rRNA_barout_gff':rRNA_barout_gff, 'dbsequences':dbsequences } context.inputs1 = { 'dbpath' : dbpath } context.outputs = { 'rRNA_blastout':rRNA_blastout, 'rRNA_stat_results': rRNA_stat_results } @@ -637,6 +634,7 @@ def create_scan_rRNA_seqs_cmd(self, s): bscore_cutoff, eval_cutoff, identity_cutoff, subunit, context.inputs['rRNA_barout_seq']) scan_cmd = scan_cmd + " -i " + context.outputs['rRNA_blastout'] + " -d " + context.inputs['dbsequences'] + scan_cmd += ' --query-gff ' + context.inputs['rRNA_barout_gff'] context.commands = [scan_cmd, blast_cmd] context.status = self.params.get('Pipeline Step Arguments', 'SCAN_rRNA') @@ -816,7 +814,7 @@ def create_genbank_file_cmd(self, s): genbank_file_status = self.params.get('Pipeline Step Arguments', 'GENBANK_FILE') if genbank_file_status in ['redo'] or\ - (genbank_file_status in ['yes'] and not s.hasGenbankFile() ): + genbank_file_status in ['yes']: cmd += ' --out-gbk ' + context.outputs['output_annot_gbk'] context.message = self._Message("GENBANK FILE" ) @@ -903,7 +901,7 @@ def create_ptinput_cmd(self, s): 'PATHOLOGIC_INPUT') - if context.status in ['redo'] or (context.status in ['yes'] and not s.hasPToolsInput() ): + if context.status in ['redo'] or context.status in ['yes']: cmd += ' --out-ptinput ' + s.output_fasta_pf_dir cmd += ' -n ' + context.inputs_optional['input_nucleotide_fasta'] cmd += ' --ncbi-tree ' + context.inputs1['ncbi_tree'] @@ -945,6 +943,7 @@ def create_report_files_cmd(self, s): context.outputs = { 'output_results_annotation_table_dir':s.output_results_annotation_table_dir, 'output_annot_table':output_annot_table, + 'annotation_taxonomy': s.output_results_annotation_table_dir + PATHDELIM + s.sample_name + '.annotation_taxonomy.tsv', } refdbs = self.get_dbs() @@ -963,12 +962,15 @@ def create_report_files_cmd(self, s): context.outputs['output_results_annotation_table_dir'], \ context.inputs['ncbi_taxonomy_tree'], \ ) - cmd = cmd + " -D " + s.blast_results_dir + " -s " + s.sample_name + " -a " + s.algorithm + cmd = cmd + " -s " + s.sample_name + " -a " + s.algorithm + for dbname in refdbs: + parsed_file = s.blast_results_dir + PATHDELIM + s.sample_name + "." + dbname + "." + s.algorithm + "out.parsed.txt" + cmd += " -d " + dbname + " -b " + parsed_file #add the command now, remove to disable in a hackish way context.commands = [cmd] context.status = self.params.get('Pipeline Step Arguments', - 'ANNOTATE_ORFS') + 'CREATE_ANNOT_REPORTS') context.message = self._Message("CREATING REPORT FILE FOR ORF ANNOTATION") contexts.append(context) return contexts @@ -984,6 +986,10 @@ def create_rpkm_cmd(self, s): if rpkm_input: fwd_fq = rpkm_input[0][0] rev_fq = rpkm_input[0][1] + if fwd_fq in ('', 'None'): + fwd_fq = None + if rev_fq in ('', 'None'): + rev_fq = None inter = rpkm_input[1] else: fwd_fq = None @@ -1052,8 +1058,10 @@ def create_rpkm_cmd(self, s): def __init__(self, params, configs): - self.params = gutils.Singleton(Params)(params) - self.configs = gutils.Singleton(Configs)(configs) + self.factory = {} + self.stageList = {} + self.params = Params(params) + self.configs = Configs(configs) self.initFactoryList() def getContexts(self, s, stage): @@ -1178,5 +1186,3 @@ def addJobs(self, s, block_mode = False): if block_mode == False: s.addContexts(contextBlock) - - diff --git a/metapathways/metacyc_db.py b/metapathways/metacyc_db.py new file mode 100644 index 0000000..ed55e9c --- /dev/null +++ b/metapathways/metacyc_db.py @@ -0,0 +1,181 @@ +"""Prepare MP references from a user's licensed MetaCyc export or Pathway Tools SIF.""" +import argparse +import csv +from datetime import datetime, timezone +import json +from pathlib import Path +import re +import shutil +import subprocess +import tempfile +import sys + +from metapathways.pt_container import digest, exec_command, save_json + + +REQUIRED = ('protseq.fsa', 'proteins.dat', 'enzrxns.dat', 'reactions.dat', + 'pathways.dat', 'compounds.dat', 'classes.dat') +TABLES = ('MetaCyc-monomer-rxn-pairs.tsv', 'MetaCyc-PWY-RXN-CMP-map.tsv', + 'MetaCyc_PWY_Ontology.tsv') + + +def source_path(source): + p = Path(source).expanduser().resolve() + if p.name == 'protseq.fsa': + p = p.parent + if p.is_file() and p.suffix.lower() == '.sif': + return p + if p.is_dir(): + missing = [name for name in REQUIRED if not (p/name).is_file()] + if not missing: + return p + raise ValueError('Incomplete MetaCyc data directory; missing ' + ', '.join(missing) + + '. Use the Pathway Tools SIF to export the matching flat files; protseq.fsa alone is insufficient. See https://hallamlab-metapathways.readthedocs.io/en/latest/pgdb-workflow.html#metacyc-from-pathway-tools.') + raise ValueError(f'MetaCyc source must be a local SIF or complete data directory: {p}; copy remote SFTP data locally first') + + +def make_tables(source, output): + """Use the repository's MetaCyc builders on a private, same-release copy.""" + source, output = Path(source), Path(output) + output.mkdir(parents=True, exist_ok=True) + fasta_ids = [] + with (source/'protseq.fsa').open() as stream: + for line in stream: + if line.startswith('>'): + match = re.match(r'>gnl\|META\|([^\s]+)', line) + if not match: + raise ValueError('Unexpected MetaCyc FASTA identifier: ' + line.strip()) + fasta_ids.append(match.group(1)) + fasta_set = set(fasta_ids) + if not fasta_ids or len(fasta_ids) != len(fasta_set): + raise ValueError('MetaCyc FASTA has no sequences or contains duplicate identifiers') + versions = set() + identifiers = {} + for filename in REQUIRED[1:]: + with (source/filename).open(errors='replace') as stream: + ids = set() + for line in stream: + if line.startswith('# Version:'): + versions.add(line.partition(':')[2].strip()) + elif line.startswith('UNIQUE-ID - '): + ids.add(line.removeprefix('UNIQUE-ID - ').strip()) + if not ids: + raise ValueError('Empty MetaCyc flat file: ' + filename) + identifiers[filename] = ids + if len(versions) > 1: + raise ValueError('Mixed MetaCyc releases in input flat files: ' + ', '.join(sorted(versions))) + missing = fasta_set - identifiers['proteins.dat'] + if missing: + raise ValueError('FASTA proteins missing from proteins.dat: ' + ', '.join(sorted(missing)[:10])) + scripts = Path(__file__).parent / 'build_DBs' + subprocess.run([sys.executable, str(scripts/'metacyc_mapping_build.py'), str(source)], check=True) + subprocess.run([sys.executable, str(scripts/'metacyc_build_ont.py'), str(source), str(output)], check=True) + for name in TABLES[:2]: + shutil.copy2(source/name, output/name) + counts = {} + for name in TABLES: + with (output/name).open() as stream: + reader = csv.DictReader(stream, delimiter='\t') + counts[name] = sum(1 for row in reader) + if not counts[name]: + raise ValueError('MetaCyc builder produced an empty table: ' + name) + with (output/TABLES[0]).open() as stream: + pairs = list(csv.DictReader(stream, delimiter='\t')) + if any(r['MC'] not in fasta_set or r['RXN'] not in identifiers['reactions.dat'] for r in pairs): + raise ValueError('MetaCyc mapping table references proteins or reactions absent from its source') + counts['proteins'] = len(fasta_ids) + counts['proteins_with_reaction'] = len({r['MC'] for r in pairs}) + counts['proteins_without_reaction'] = len(fasta_ids) - counts['proteins_with_reaction'] + return counts + + +def export_image(image, state): + state = Path(state) + (state/'export').mkdir(parents=True) + script = r'''set -eu +cp -a /opt/ptools-local-template /data/ptools-local +if test -f /opt/mp-ncbirc; then cp /opt/mp-ncbirc /data/.ncbirc; fi +meta_root=/opt/pathway-tools/aic-export/pgdbs/biocyc/metacyc +version=$(cat "$meta_root/default-version") +test -n "$version" +cp "$meta_root/$version/data/protseq.fsa" /data/export/protseq.fsa +printf '%s\n' "$version" > /data/export/version.txt +xvfb-run -a /opt/pathway-tools/pathway-tools -no-patch-download -no-cel-overview -disable-metadata-saving -nologfile -lisp -eval '(progn (with-organism (:org-id '\''META) (dump-frames-to-attribute-value-files "/data/export/")) (format t "~%MP-METACYC-EXPORTED~%") (exit))' +''' + # This exports the bundled reference; it does not create a sample PGDB. + log = state/'export.log' + with log.open('w') as out: + result = subprocess.run(exec_command(image, state, ['sh', '-ec', script]), + stdout=out, stderr=subprocess.STDOUT, timeout=1800) + text = log.read_text(errors='replace') + print(text, flush=True) + if result.returncode or 'MP-METACYC-EXPORTED' not in text.splitlines(): + raise RuntimeError('MetaCyc reference export failed; see ' + str(log)) + return source_path(state/'export') + + +def prepare(source, root, aligner): + source, root = source_path(source), Path(root).resolve() + if aligner not in ('fast', 'blast'): + raise ValueError('MetaCyc aligner must be fast or blast') + root.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory(prefix='.metacyc-build-', dir=root) as scratch: + work = Path(scratch) + if source.is_file(): + raw = export_image(source, work/'state') + else: + # Existing scripts write tables beside the FASTA; never modify the source. + raw = work/'source' + raw.mkdir() + for name in REQUIRED: + shutil.copy2(source/name, raw/name) + (raw/'version.txt').write_text((source/'version.txt').read_text() if (source/'version.txt').is_file() else source.parent.name) + + staged = work/'reference' + functional, categories = staged/'functional', staged/'functional_categories' + formatted = functional/'formatted' + formatted.mkdir(parents=True) + categories.mkdir() + counts = make_tables(raw, categories) + fasta = functional/'metacyc' + shutil.copy2(raw/'protseq.fsa', fasta) + prefix = formatted/'metacyc' + with Path(str(prefix)+'-names.txt').open('w') as out, fasta.open() as stream: + for line in stream: + if line.startswith('>'): + out.write(line) + command = (['fastdb', '-p', str(prefix), str(fasta)] if aligner == 'fast' else + ['makeblastdb', '-in', str(fasta), '-dbtype', 'prot', '-parse_seqids', '-out', str(prefix)]) + subprocess.run(command, check=True) + sentinel = Path(str(prefix) + ('.prj' if aligner == 'fast' else '.pdb')) + if not sentinel.is_file(): + raise RuntimeError('MetaCyc indexer did not produce ' + sentinel.name) + version_file = raw/'version.txt' + version = version_file.read_text().strip() if version_file.is_file() else raw.parent.name + manifest = dict(source=str(source), release=version, aligner=aligner, counts=counts, + created_at=datetime.now(timezone.utc).isoformat(), + source_sha256={name: digest(raw/name) for name in REQUIRED}) + if source.is_file(): + manifest['image_sha256'] = digest(source) + (categories/'MetaCyc_reldate.txt').write_text('Release: ' + version + '\n') + save_json(categories/'MetaCyc_provenance.json', manifest) + # All export, mapping and indexing checks finish before publication. + new_indexes = {file.name for file in formatted.glob('metacyc.*')} + for old in (root/'functional/formatted').glob('metacyc.*'): + if old.name not in new_indexes: + old.unlink() + for file in sorted(staged.rglob('*')): + if file.is_file(): + destination = root/file.relative_to(staged) + destination.parent.mkdir(parents=True, exist_ok=True) + file.replace(destination) + print('Prepared MetaCyc ' + version + ': ' + json.dumps(counts), flush=True) + + +if __name__ == '__main__': + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--source', required=True) + parser.add_argument('--output', required=True) + parser.add_argument('--aligner', choices=('fast', 'blast'), required=True) + args = parser.parse_args() + prepare(args.source, args.output, args.aligner) diff --git a/metapathways/metapathways_utils.py b/metapathways/metapathways_utils.py index 0f7e7f1..90c50ca 100644 --- a/metapathways/metapathways_utils.py +++ b/metapathways/metapathways_utils.py @@ -62,7 +62,7 @@ def exit_process(message=None, logger=None): if logger: logger.printf("ERROR\tExiting the Python code\n") logger.printf("ERROR\t" + message + "\n") - _exit(0) + raise SystemExit(1) def exit_step(message=None): diff --git a/metapathways/nextflow.py b/metapathways/nextflow.py new file mode 100644 index 0000000..6222d7c --- /dev/null +++ b/metapathways/nextflow.py @@ -0,0 +1,515 @@ +"""Nextflow scheduling behind the existing MetaPathways CLI. + +Stage tools continue writing their established output paths. Each Nextflow task +validates a durable receipt against those files instead of trusting a cached +sentinel when a user has removed or changed published outputs. +""" +import argparse +import fcntl +import hashlib +import json +import os +from pathlib import Path +import re +import shlex +import shutil +import subprocess +import sys +import uuid +import queue +import signal +import threading +import time + + +def positive(value): + n = int(value) + if n < 1: + raise argparse.ArgumentTypeError('must be a positive integer') + return n + + +def memory(value): + if not re.fullmatch(r'\d+(?:\.\d+)?\s*(?:KB|MB|GB|TB)', value, re.I): + raise argparse.ArgumentTypeError('use a size such as "16 GB"') + if float(re.match(r'[0-9.]+', value).group()) <= 0: + raise argparse.ArgumentTypeError('memory must be positive') + return value.upper() + + +def add_resources(parser, task_memory='16 GB'): + group = parser.add_argument_group('Execution Resources') + group.add_argument('--max_cpus', type=positive, default=None, + help='total CPU budget [local: available CPUs; Slurm: no aggregate cap]') + group.add_argument('--memory', type=memory, default=task_memory, + help=f'memory reservation per task [{task_memory}]') + group.add_argument('--max_memory', type=memory, default=None, + help='total memory budget [local: available memory; Slurm: no aggregate cap]') + group.add_argument('--max_tasks', type=positive, default=None, + help='maximum submitted tasks, including queued/running [local: CPU budget; Slurm: 4]') + + + group.add_argument('--executor', choices=['local', 'slurm'], default='local', + help='execution backend [local]; Slurm uses your logged-in cluster identity') + group.add_argument('--account', type=slurm_token, help='Slurm allocation/account') + group.add_argument('--partition', type=slurm_token, help='Slurm partition [cluster default]') + group.add_argument('--qos', type=slurm_token, help='Slurm quality of service') + group.add_argument('--reservation', type=slurm_token, help='Slurm reservation') + group.add_argument('--time_limit', type=duration, default='24h', help='Slurm walltime per task [24h]') + group.add_argument('--submit_rate', type=positive, default=6, help='maximum Slurm submissions per minute [6]') + group.add_argument('--work_dir', help='custom Nextflow work directory; retained after the run') + group.add_argument('--conda_cache', help='custom Conda cache directory; retained after the run') + group.add_argument('--keep_work', action='store_true', help='retain automatically allocated work/cache directories') + + +def slurm_token(value): + if not re.fullmatch(r'[A-Za-z0-9][A-Za-z0-9_.-]*', value): + raise argparse.ArgumentTypeError('use letters, digits, underscore, period or hyphen') + return value + + +def duration(value): + if not re.fullmatch(r'[1-9][0-9]*(?:s|m|h|d)', value): + raise argparse.ArgumentTypeError('use a positive duration such as 30m, 12h or 2d') + return value + + +def memory_bytes(value): + number, unit = re.fullmatch(r'([0-9.]+)\s*(KB|MB|GB|TB)', memory(value)).groups() + return int(float(number) * 1024 ** {'KB': 1, 'MB': 2, 'GB': 3, 'TB': 4}[unit]) + + +def ordered(tasks): + by_id = {t['id']: t for t in tasks} + if len(by_id) != len(tasks): + raise ValueError('Duplicate task identifiers') + result, visiting, seen = [], set(), set() + def visit(identifier): + if identifier not in by_id: + raise ValueError(f'Unknown dependency: {identifier}') + if identifier in visiting: + raise ValueError(f'Cyclic task dependency: {identifier}') + if identifier in seen: + return + visiting.add(identifier) + for dep in by_id[identifier]['dependencies']: + visit(dep) + visiting.remove(identifier) + seen.add(identifier) + result.append(by_id[identifier]) + for t in tasks: + visit(t['id']) + return result + + +def task(identifier, label, commands, inputs=(), outputs=(), dependencies=(), cpus=1, + memory='16 GB', **kwargs): + return dict(id=identifier, label=label, commands=commands, inputs=list(inputs), + outputs=list(outputs), dependencies=list(dependencies), cpus=cpus, + memory=memory, status='yes', **kwargs) + + +def groovy(value): + return "'" + str(value).replace('\\', '\\\\').replace("'", "\\'").replace('\n', '\\n') + "'" + + +def process_definition(t, name, manifest): + package_root = str(Path(__file__).resolve().parent.parent) + command = shlex.join([sys.executable, '-m', 'metapathways.nf_worker', str(manifest), t['id']]) + body = f'export PYTHONPATH={shlex.quote(package_root)}${{PYTHONPATH:+:$PYTHONPATH}}\n{command}\n' + count = len(t['dependencies']) + inputs = '\n'.join(f' val dependency_{i}' for i in range(1 if count > 64 else max(1, count))) + return f'''process {name} {{ + tag {groovy(t['label'])} + cpus {t['cpus']} + memory {groovy(t['memory'])} + cache false + input: +{inputs} + output: + path 'receipt.json' + script: + {groovy(body)} +}}''' + + +def render_modules(tasks, manifest, batch_size=64): + """Bound compiled script size without adding scheduling barriers. + + Each module exposes individual task completion channels. A consumer waits + only for its own dependencies, even when their producers share a module + with other tasks. All processes still use the same Nextflow executor. + """ + if batch_size < 1 or batch_size > 64: + raise ValueError('Workflow module size must be between 1 and 64') + tasks = ordered(tasks) + # Sample completion/cleanup has a large fan-in. Flattening every sample's + # channels into main.nf can exceed the JVM's 64 KiB method limit even though + # process definitions are batched. Keep independent sample DAGs in named + # subworkflows so both process code AND channel wiring stay bounded. + samples = list(dict.fromkeys(t.get('sample') for t in tasks)) + by_id = {t['id']: t for t in tasks} + if len(samples) > 1 and None not in samples and all( + by_id[dep].get('sample') == t['sample'] for t in tasks for dep in t['dependencies']): + files, includes, calls = {}, [], [] + for i, sample in enumerate(samples): + name = f'SAMPLE_{i:04d}' + subset = [t for t in tasks if t['sample'] == sample] + for relative, content in render_modules(subset, manifest, batch_size).items(): + if relative == 'main.nf': + content = content.replace('\nworkflow {\n', f'\nworkflow {name} {{\n') + files[f'samples/{name}/{relative}'] = content + includes.append(f"include {{ {name} }} from './samples/{name}/main'") + calls.append(f' {name}()') + files['main.nf'] = 'nextflow.enable.dsl=2\n\n' + '\n'.join(includes) + '\n\nworkflow {\n' + '\n'.join(calls) + '\n}\n' + return files + names = {t['id']: f'TASK_{i:04d}' for i, t in enumerate(tasks)} + batches = [tasks[i:i + batch_size] for i in range(0, len(tasks), batch_size)] + owner = {t['id']: i for i, batch in enumerate(batches) for t in batch} + exports = {dep for t in tasks for dep in t['dependencies'] if owner[dep] != owner[t['id']]} + files, includes, calls = {}, [], [] + for i, batch in enumerate(batches): + workflow = f'BATCH_{i:04d}' + external = list(dict.fromkeys(dep for t in batch for dep in t['dependencies'] if owner[dep] != i)) + bundled = len(external) > 64 + ports = {dep: ('upstream.' if bundled else '') + f'upstream_{j}' for j, dep in enumerate(external)} + blocks = [process_definition(t, names[t['id']], manifest) for t in batch] + lines = [f'workflow {workflow} {{'] + if external: + lines += [' take:'] + ([' upstream'] if bundled else [f' {ports[dep]}' for dep in external]) + lines.append(' main:') + for t in batch: + channels = [ports[dep] if owner[dep] != i else names[dep] + '.out' for dep in t['dependencies']] + args = ', '.join(channels) + if len(channels) > 64: + # One completion gate, with bounded operator argument counts. + # collect emits only after every prerequisite channel closes. + args = 'Channel.empty()' + for start in range(0, len(channels), 32): + args += '.mix(' + ', '.join(channels[start:start+32]) + ')' + args += '.collect()' + lines.append(f" {names[t['id']]}({args or 'Channel.value(true)'})") + emitted = [t['id'] for t in batch if t['id'] in exports] + if emitted: + lines += [' emit:'] + [f' done_{names[dep]} = {names[dep]}.out' for dep in emitted] + lines.append('}') + files[f'modules/{workflow}.nf'] = '\n\n'.join(blocks) + '\n\n' + '\n'.join(lines) + '\n' + includes.append(f"include {{ {workflow} }} from './modules/{workflow}'") + args = ', '.join(f'BATCH_{owner[dep]:04d}.out.done_{names[dep]}' for dep in external) + if bundled: + args = '[' + ', '.join(f'upstream_{j}: BATCH_{owner[dep]:04d}.out.done_{names[dep]}' + for j, dep in enumerate(external)) + ']' + calls.append(f' {workflow}({args})') + files['main.nf'] = 'nextflow.enable.dsl=2\n\n' + '\n'.join(includes) + '\n\nworkflow {\n' + '\n'.join(calls) + '\n}\n' + return files + + +def local_capacity(): + cpus = len(os.sched_getaffinity(0)) + available = None + try: + for line in Path('/proc/meminfo').read_text().splitlines(): + if line.startswith('MemAvailable:'): + available = int(line.split()[1]) * 1024 + quota, period = Path('/sys/fs/cgroup/cpu.max').read_text().split() + if quota != 'max': + cpus = min(cpus, max(1, int(quota) // int(period))) + except (OSError, ValueError): + pass + try: + limit = Path('/sys/fs/cgroup/memory.max').read_text().strip() + if limit != 'max': + remaining = max(1, int(limit) - int(Path('/sys/fs/cgroup/memory.current').read_text())) + available = min(available, remaining) if available is not None else remaining + except (OSError, ValueError): + pass + return cpus, (f'{max(1, available // (1024 * 1024))} MB' if available is not None else None) + + +def configuration(tasks, args, conda_cache): + backend = getattr(args, 'executor', 'local') + local_cpus, local_memory = local_capacity() if backend == 'local' else (None, None) + cpus = getattr(args, 'max_cpus', None) or local_cpus + limit_memory = getattr(args, 'max_memory', None) or local_memory + if any(t['cpus'] < 1 or (cpus is not None and t['cpus'] > cpus) for t in tasks): + raise ValueError('Task threads must be positive and no greater than --max_cpus') + largest_memory = max(memory_bytes(t['memory']) for t in tasks) + if limit_memory and largest_memory > memory_bytes(limit_memory): + raise ValueError('A task requests more memory than --max_memory') + config = [f'process.executor = {groovy(backend)}', "process.errorStrategy = 'finish'", + 'process.maxRetries = 0', f'conda.cacheDir = {groovy(str(conda_cache / "envs"))}'] + if backend == 'local': + if any(getattr(args, key, None) for key in ('account', 'partition', 'qos', 'reservation')): + raise ValueError('Slurm flags require --executor slurm') + config.append(f'executor.cpus = {cpus}') + if limit_memory: + config.append(f'executor.memory = {groovy(limit_memory)}') + # Default Nextflow queueSize (100) can otherwise underutilize large hosts. + jobs = getattr(args, 'max_tasks', None) or cpus + else: + if not getattr(args, 'account', None): + raise ValueError('Slurm requires --account') + # Slurm does not use executor.cpus/memory as aggregate limits. Bound + # submitted jobs conservatively using the largest task request instead. + jobs = getattr(args, 'max_tasks', None) or 4 + if cpus is not None: + jobs = min(jobs, cpus // max(t['cpus'] for t in tasks)) + if limit_memory is not None: + jobs = min(jobs, memory_bytes(limit_memory) // largest_memory) + # All MP tools use threads within one node, never distributed MPI ranks. + # Some sites reject sbatch submissions that omit an explicit node count. + options = ['--nodes=1'] + for key in ('account', 'qos', 'reservation'): + if getattr(args, key, None): + options.append('--' + key + '=' + slurm_token(getattr(args, key))) + if getattr(args, 'partition', None): + config.append(f'process.queue = {groovy(slurm_token(args.partition))}') + config += [f'process.clusterOptions = {groovy(" ".join(options))}', + f'process.time = {groovy(duration(getattr(args, "time_limit", "24h")))}', + f'executor.submitRateLimit = {groovy(str(getattr(args, "submit_rate", 6)) + "/1min")}', + "executor.queueStatInterval = '1min'", "executor.pollInterval = '10sec'"] + config.append(f'executor.queueSize = {jobs}') + return '\n'.join(config) + '\n', dict(executor=backend, max_cpus=cpus, max_memory=limit_memory, max_tasks=jobs) + + +def stream_run(command, cwd, env, tasks, console): + """Mirror controller and worker output to the terminal and a durable log.""" + lines = queue.Queue() + process = subprocess.Popen(command, cwd=cwd, env=env, stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, text=True, errors='replace', bufsize=1) + def read_controller(): + for line in process.stdout: + lines.put(line) + reader = threading.Thread(target=read_controller, daemon=True) + reader.start() + offsets = {t['id']: 0 for t in tasks} + def emit(line): + print(line, end='', flush=True) + console.write(line) + console.flush() + def drain(): + while True: + try: + emit(lines.get_nowait()) + except queue.Empty: + break + for t in tasks: + path = Path(t['log']) + if path.exists(): + with path.open(errors='replace') as f: + f.seek(offsets[t['id']]) + for line in f: + emit(f"[{t['label']}] {line}") + offsets[t['id']] = f.tell() + try: + while process.poll() is None: + drain() + time.sleep(0.2) + reader.join() + drain() + except BaseException: + process.send_signal(signal.SIGINT) + try: + process.wait(timeout=30) + except subprocess.TimeoutExpired: + process.terminate() + raise + finally: + process.stdout.close() + if process.returncode: + raise subprocess.CalledProcessError(process.returncode, command) + + +def launch(tasks, output_dir, args, name, dryrun=False): + if not tasks: + raise ValueError('No tasks were selected') + tasks = ordered(tasks) + output_dir = Path(output_dir).resolve() + state = output_dir / '.metapathways' / name + state.mkdir(parents=True, exist_ok=True) + with (state / 'controller.lock').open('w') as lock: + try: + fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError: + raise RuntimeError(f'Another {name} command is already using {output_dir}') + run_id = uuid.uuid4().hex + run_dir = output_dir / 'logs' / name / run_id + run_dir.mkdir(parents=True) + scratch = state / 'tmp' / run_id + work_override = getattr(args, 'work_dir', None) + cache_override = getattr(args, 'conda_cache', None) + work = (Path(work_override).expanduser().resolve() / (name + '-' + run_id) if work_override else scratch / 'work') + cache = Path(cache_override).expanduser().resolve() if cache_override else scratch / 'conda' + config_text, resources = configuration(tasks, args, cache) + for t in tasks: + t['compact_results'] = bool(getattr(args, 'compact_results', False)) + t['scratch_dir'] = getattr(args, 'scratch_dir', None) + key = hashlib.sha256(t['id'].encode()).hexdigest() + t['receipt'] = str(state / 'receipts' / (key + '.json')) + t['log'] = str(run_dir / 'tasks' / (key + '.log')) + t['invocation_receipt'] = str(run_dir / 'tasks' / (key + '.json')) + manifest = run_dir / 'tasks.json' + manifest.write_text(json.dumps(tasks, indent=2) + '\n') + nf = run_dir / 'main.nf' + for relative, content in render_modules(tasks, manifest).items(): + target = run_dir / relative + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(content) + config = run_dir / 'nextflow.config' + config.write_text(config_text) + from metapathways._version import __version__ + summary = dict(mp_version=__version__, resources=resources, work_dir=str(work), conda_cache=str(cache), status='PLANNED') + summary_path = run_dir / 'summary.json' + summary_path.write_text(json.dumps(summary, indent=2) + '\n') + if dryrun: + for t in tasks: + print(f"{t['label']}: cpus={t['cpus']} memory={t['memory']} after={','.join(t['dependencies']) or '-'}") + for cmd in t['commands']: + print(' ' + cmd) + print(f'Execution plan: {manifest}') + return run_dir + executable = shutil.which('nextflow') + if not executable: + raise RuntimeError('Nextflow is required; activate the MetaPathways environment.') + if resources['executor'] == 'slurm': + missing = [c for c in ('sbatch', 'squeue', 'scancel') if not shutil.which(c)] + if missing: + raise RuntimeError('Run from your Slurm login node; missing commands: ' + ', '.join(missing)) + session = scratch / 'session' + session.mkdir(parents=True) + work.mkdir(parents=True, exist_ok=True) + cache.mkdir(parents=True, exist_ok=True) + env = os.environ.copy() + env.update(NXF_ANSI_LOG='false', PYTHONUNBUFFERED='1', + NXF_CONDA_CACHEDIR=str(cache / 'envs'), CONDA_PKGS_DIRS=str(cache / 'packages')) + # Isolate resource limits from ambient ~/.nextflow/config overrides. + command = [executable, '-C', str(config), '-log', str(run_dir / 'nextflow.log'), + 'run', str(nf), '-work-dir', str(work), + '-with-trace', str(run_dir / 'trace.tsv'), '-with-report', str(run_dir / 'report.html'), + '-with-timeline', str(run_dir / 'timeline.html')] + cpu_budget = resources['max_cpus'] if resources['max_cpus'] is not None else 'per-job requests' + print(f"MetaPathways: {name}; executor={resources['executor']}; CPU budget={cpu_budget}; task limit={resources['max_tasks']}", flush=True) + print(f'Logs and resource reports: {run_dir}', flush=True) + interrupted = False + try: + with (run_dir / 'console.log').open('w') as console: + stream_run(command, session, env, tasks, console) + summary['status'] = 'SUCCESS' + except BaseException as exc: + interrupted = isinstance(exc, (KeyboardInterrupt, SystemExit)) + summary['status'] = 'INTERRUPTED' if interrupted else 'FAILED' + summary['error'] = str(exc) + raise + finally: + records = [] + for t in tasks: + receipt = Path(t['invocation_receipt']) + if receipt.exists(): + record = json.loads(receipt.read_text()) + record['label'] = t['label'] + records.append(record) + else: + records.append(dict(task=t['id'], label=t['label'], status='NOT_STARTED')) + for t, record in zip(tasks, records): + for key in ('sample', 'entity'): + if key in t: + record[key] = t[key] + summary['tasks'] = records + summary_path.write_text(json.dumps(summary, indent=2) + '\n') + compact_exit = getattr(args, 'compact_results', False) and not interrupted + # Preserve wrapper diagnostics even for scheduler/bootstrap failures + # before a worker could open its own log. + if compact_exit: + from metapathways.compact_storage import archive_directory + archive_directory(work, run_dir/'nextflow_tasks.tar.gz') + if (run_dir/'tasks').is_dir(): + archive_directory(run_dir/'tasks', run_dir/'task_logs.tar.gz') + shutil.rmtree(run_dir/'tasks') + else: + for path in work.rglob('.command.*'): + if path.is_file(): + target = run_dir / 'nextflow_tasks' / path.relative_to(work) + target.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(path, target) + if compact_exit: + # Keep trace, task plans and diagnostics; discard generated NF code. + nf.unlink(missing_ok=True) + config.unlink(missing_ok=True) + shutil.rmtree(run_dir / 'modules', ignore_errors=True) + shutil.rmtree(run_dir / 'samples', ignore_errors=True) + if not getattr(args, 'keep_work', False) and not interrupted: + shutil.rmtree(scratch) + else: + print(f'Work/cache retained: {scratch}', flush=True) + failed = [r['label'] for r in records if r.get('status') == 'FAILED'] + if failed: + print(f'Completed with {len(failed)} optional PGDB failures: ' + ', '.join(failed)) + print(f'Finished {name}. Logs: {run_dir}', flush=True) + return run_dir + + +def annotation_tasks(samples, params, configs, memory_value): + from metapathways import jobscreator + creator = jobscreator.JobCreator(params, configs) + threaded = {'ORF_PREDICTION', 'FUNC_SEARCH', 'SCAN_rRNA', 'SCAN_tRNA', 'COMPUTE_TPM'} + tasks = [] + for key in sorted(samples): + s = samples[key] + creator.addJobs(s, block_mode=True) + s.writeParamsToRunLogs(type('Parameters', (), {'params': params})()) + previous, group_ids, last_group = [], [], None + for block, contexts in enumerate(s.getContextBlocks()): + for i, c in enumerate(contexts): + group = c.name if c.name == 'SCAN_rRNA:barrnap' else c.name.split(':')[0] + if group != last_group: + previous, group_ids, last_group = group_ids, [], group + identifier = f'{s.sample_name}:{block}:{i}:{c.name}' + t = task(identifier, f'{s.sample_name}:{c.name}', c.commands, + c.inputs.values(), c.outputs.values(), previous, + int(configs['NUM_CPUS']) if c.name.split(':')[0] in threaded else 1, + memory_value, kind='annotation', context=vars(c), + sample_output=s.output_dir, sample=s.sample_name, block=block) + t['status'] = c.status + t['adopt_existing'] = c.name != 'COMPUTE_TPM' + if c.name.startswith('FUNC_SEARCH:') or c.name == 'COMPUTE_REFSCORES': + t['cache_version'] = 'private-fast-scratch-v1' + t['adopt_existing'] = False + if c.name == 'PATHOLOGIC_INPUT': + t['cache_version'] = 'normalized-ec-input-v1' + t['adopt_existing'] = False + if c.name == 'COMPUTE_TPM': + # bwaFolder is a destination for this task's intermediate + # BAM/count files, not a biological input. Fingerprinting + # it makes a successful run invalidate its own receipt. + t['inputs'] = [value for name, value in c.inputs.items() + if name != 'bwaFolder'] + if c.temps.get('rev_fq'): + t['inputs'].append(c.temps['rev_fq']) + # Auxiliary inputs affect resumability but may be optional references. + t['fingerprint_inputs'] = t['inputs'] + list(getattr(c, 'inputs1', {}).values()) + list(getattr(c, 'inputs_optional', {}).values()) + if c.name.startswith('FUNC_SEARCH:'): + db = c.name.split(':', 1)[1] + t['fingerprint_inputs'].append(str(Path(configs['REFDBS']) / 'functional/formatted' / db) + '.*') + if c.name.startswith('SCAN_rRNA:') and c.name != 'SCAN_rRNA:barrnap': + t['fingerprint_inputs'].append(c.inputs1['dbpath'] + '.*') + if c.name == 'PATHOLOGIC_INPUT': + t['outputs'] += [str(Path(s.output_dir) / 'ptools/0.pf'), str(Path(s.output_dir) / 'ptools/orf_map.txt'), + str(Path(s.output_dir) / 'results/annotation_table' / f'{s.sample_name}.EC_RXN_map.tsv')] + if c.name == 'CREATE_ANNOT_REPORTS': + t['outputs'].append(str(Path(s.output_dir) / 'results/annotation_table' / f'{s.sample_name}.ORF_annotation_table.txt')) + t['outputs'] = [p for p in t['outputs'] if not Path(p).is_dir()] + + # Read abundance is optional, not a missing-input failure. + if c.name == 'COMPUTE_TPM' and not c.inputs.get('fwd_fq'): + t['status'] = 'skip' + if c.name == 'COMPUTE_TPM': + t['outputs'] += [str(Path(s.output_dir) / 'results/rpkm' / f'{s.sample_name}.contig_counts.tsv'), + str(Path(s.output_dir) / 'results/rpkm' / f'{s.sample_name}.orf_counts.tsv')] + tasks.append(t) + group_ids.append(identifier) + # Some legacy stages intentionally update an earlier stage's output (GBK). + # The final producer owns its fingerprint; earlier stages still require it. + owner = {p: t['id'] for t in tasks for p in t['outputs']} + for t in tasks: + t['cache_outputs'] = [p for p in t['outputs'] if owner[p] == t['id']] + return tasks diff --git a/metapathways/nf_databases.py b/metapathways/nf_databases.py new file mode 100644 index 0000000..5bb5cd0 --- /dev/null +++ b/metapathways/nf_databases.py @@ -0,0 +1,171 @@ +"""Database preparation DAG for the public build_db command (no Snakemake).""" +from pathlib import Path +import shlex +import sys +from metapathways.nextflow import task + + +def metacyc_task(root, source, aligner, memory='16 GB', dependencies=()): + """Shared by explicit build_db selection and the build_pt dependent export.""" + from metapathways.metacyc_db import TABLES + root, source = Path(root).expanduser().resolve(), Path(source).expanduser().resolve() + ext = '.prj' if aligner == 'fast' else '.pdb' + outputs = [root/'functional/metacyc', root/('functional/formatted/metacyc'+ext), + root/'functional/formatted/metacyc-names.txt', + root/'functional_categories/MetaCyc_provenance.json', root/'functional_categories/MetaCyc_reldate.txt'] + outputs.extend(root/'functional_categories'/name for name in TABLES) + command = shlex.join([sys.executable, '-m', 'metapathways.metacyc_db', '--source', str(source), + '--output', str(root), '--aligner', aligner]) + return task('prepare_metacyc', 'Prepare licensed MetaCyc reference', [command], [str(source)], + list(map(str, outputs)), dependencies, memory=memory, adopt_existing=False, + cache_outputs=[str(root/'functional/formatted/metacyc.*')]) + + +def plan(root, databases, aligner, test=False, memory='16 GB', metacyc_source=None, skip_pt_screen=False, screen_image=None, resources=None): + root = Path(root).resolve() + scripts = Path(__file__).parent / 'build_DBs' + q = lambda p: shlex.quote(str(p)) + python = q(sys.executable) + tasks = [] + dirs = ['functional/formatted', 'functional_categories', 'ncbi_tree', 'taxonomic/formatted', '.metapathways'] + ready = root / '.metapathways/database-directories.ready' + tasks.append(task('directories', 'Prepare database directories', + ['mkdir -p ' + ' '.join(q(root/d) for d in dirs) + f'; test -e {q(ready)} || touch {q(ready)}'], outputs=[str(ready)], memory=memory)) + # Recreate missing directories without changing the download dependency timestamp. + tasks[-1]['status'] = 'redo' + + def add(identifier, commands, inputs, outputs, deps): + tasks.append(task(identifier, identifier, commands, map(str, inputs), map(str, outputs), deps, memory=memory)) + if identifier.startswith('index_'): + # Track every index shard so removal of a non-sentinel file invalidates reuse. + prefix = str(tasks[-1]['outputs'][0]).rsplit('.', 1)[0] + tasks[-1]['cache_outputs'] = [prefix + '.*'] + tasks[-1]['adopt_existing'] = False + return identifier + + def fetch(identifier, url, dest, compressed=False): + dest = Path(dest) + download = str(dest) + '.download' + command = f'wget -O {q(download)} {q(url)}\n' + if compressed: + command += f'gzip -dc {q(download)} > {q(str(dest)+".tmp")}\nmv {q(str(dest)+".tmp")} {q(dest)}\nrm {q(download)}' + else: + command += f'mv {q(download)} {q(dest)}' + return add(identifier, [command], [ready], [dest], ['directories']) + + if 'metacyc' in databases: + from metapathways.metacyc_db import source_path + from metapathways.pt_container import registered_image + source = metacyc_source or registered_image() + if not source: + raise ValueError('MetaCyc requires a licensed source: run build_pt first, or provide --metacyc_source with a complete data directory or SIF. See https://hallamlab-metapathways.readthedocs.io/en/latest/pgdb-workflow.html#metacyc-from-pathway-tools.') + source = source_path(source) + tasks.append(metacyc_task(root, source, aligner, memory, ['directories'])) + if not skip_pt_screen: + image = screen_image or (str(source) if source.is_file() else registered_image()) + if not image: + raise ValueError('MetaCyc screening requires a PTools SIF: run build_pt, provide --screen_image, or explicitly use --skip_pt_screen') + tasks.append(screen_task(root, image, memory, resources=resources)) + databases = [db for db in databases if db != 'metacyc'] + if not databases: + # Adding MetaCyc to an existing MPDB must not refresh unrelated references. + return tasks + + silvas = ['SILVA_LSU_test', 'SILVA_SSU_test'] if test else [ + 'SILVA_138.1_LSURef_NR99_tax_silva_trunc', 'SILVA_138.1_SSURef_NR99_tax_silva_trunc'] + for db in silvas: + fasta, prefix = root/'taxonomic'/db, root/'taxonomic/formatted'/db + deps = ['directories'] + if not test: + deps = [fetch('fetch_'+db, f'https://www.arb-silva.de/fileadmin/silva_databases/release_138_1/Exports/{db}.fasta.gz', fasta, True)] + add('index_'+db, [f'makeblastdb -in {q(fasta)} -dbtype nucl -parse_seqids -out {q(prefix)}'], + [fasta], [str(prefix)+'.ndb'], deps) + add('names_'+db, [f"grep '^>' {q(fasta)} > {q(str(prefix)+'-names.txt')}"], [fasta], [str(prefix)+'-names.txt'], deps) + + if not test: + note = root/'functional_categories/SILVA_reldate.txt' + add('release_silva', [f'date -u > {q(note)}\nprintf "Release: 138.1\\n" >> {q(note)}'], + [], [note], ['fetch_'+db for db in silvas]) + + ec = root/'functional_categories/enzyme.dat' + ec_dep = fetch('fetch_enzyme', 'https://ftp.expasy.org/databases/enzyme/enzyme.dat', ec) + ec_note = root/'functional_categories/Expasy_reldate.txt' + add('enzyme_release', [f'wget -O {q(str(ec_note)+".tmp")} https://ftp.expasy.org/databases/enzyme/enzclass.txt\n' + f'date -u > {q(ec_note)}\ngrep "^Release:" {q(str(ec_note)+".tmp")} >> {q(ec_note)}\nrm {q(str(ec_note)+".tmp")}'], + [ec], [ec_note], [ec_dep]) + tax_archive = root/'ncbi_tree/new_taxdump.tar.gz' + tax_dep = fetch('fetch_taxonomy', 'https://ftp.ncbi.nlm.nih.gov/pub/taxonomy/new_taxdump/new_taxdump.tar.gz', tax_archive) + add('taxonomy_tree', [f'tar xzf {q(tax_archive)} -C {q(root/"ncbi_tree")}\n{python} {q(scripts/"taxmap_build.py")} {q(root/"ncbi_tree")}'], + [tax_archive], [root/'ncbi_tree/ncbi_taxonomy_tree.txt'], [tax_dep]) + unirefs = [db for db in databases if db in ('uniref50', 'uniref90')] + if unirefs: + idmap = root/'functional_categories/idmapping.dat.gz' + idmap_dep = fetch('fetch_idmapping', 'https://ftp.ebi.ac.uk/pub/databases/uniprot/current_release/knowledgebase/idmapping/idmapping.dat.gz', idmap) + for db in databases: + fasta, prefix = root/'functional'/db, root/'functional/formatted'/db + deps = ['directories'] + if not test: + if db == 'swissprot': + url = 'https://ftp.ebi.ac.uk/pub/databases/uniprot/current_release/knowledgebase/complete/uniprot_sprot.fasta.gz' + deps = [fetch('fetch_'+db, url, fasta, True)] + fetch('release_'+db, 'https://ftp.ebi.ac.uk/pub/databases/uniprot/current_release/knowledgebase/complete/reldate.txt', root/'functional_categories/SwissProt_reldate.txt') + elif db in unirefs: + url = f'https://ftp.ebi.ac.uk/pub/databases/uniprot/uniref/{db}/{db}.fasta.gz' + deps = [fetch('fetch_'+db, url, fasta, True)] + fetch('release_'+db, f'https://ftp.ebi.ac.uk/pub/databases/uniprot/uniref/{db}/{db}.release_note', root/f'functional_categories/{db}_reldate.txt') + elif db == 'cazy': + deps = [fetch('fetch_cazy', 'https://bcb.unl.edu/dbCAN2/download/Databases/V12/CAZyDB.07262023.fa', fasta)] + note = root/'functional_categories/CAZyDB_reldate.txt' + add('release_cazy', [f'date -u > {q(note)}\nprintf "Release: V12_07262023\\n" >> {q(note)}'], + [fasta], [note], deps) + elif db == 'eggnog' and not fasta.is_file(): + raise ValueError(f'Provide the eggNOG protein FASTA at {fasta}; the previous database builder did not define an eggNOG download source.') + ext = '.pdb' if aligner == 'blast' else '.prj' + cmd = (f'makeblastdb -in {q(fasta)} -dbtype prot -parse_seqids -out {q(prefix)}' if aligner == 'blast' + else f'fastdb -p {q(prefix)} {q(fasta)}') + add('index_'+db, [cmd], [fasta], [str(prefix)+ext], deps) + add('names_'+db, [f"grep '^>' {q(fasta)} > {q(str(prefix)+'-names.txt')}"], [fasta], [str(prefix)+'-names.txt'], deps) + if db in ('swissprot', 'swissprot_test'): + add('ec_'+db, [f'{python} {q(scripts/(db+"_mapper.py"))} {q(ec)} {q(root/"functional_categories")}'], + [fasta, ec], [root/f'functional_categories/EC_map.{db}.tsv'], deps+[ec_dep]) + elif db in unirefs: + filtered = root/f'functional_categories/{db}_idmapping.dat' + # A shared download feeds independent maps; no concurrent overwrites. + add('ec_'+db, [f"gzip -dc {q(idmap)} | grep {q('UniRef'+db[6:]+'_')} > {q(filtered)}\n" + f'{python} {q(scripts/(db+"_mapper.py"))} {q(ec)} {q(root/"functional_categories")}'], + [fasta, ec, idmap], [root/f'functional_categories/EC_map.{db}.tsv'], deps+[ec_dep, idmap_dep]) + return tasks + + +def screen_task(root, image, memory='16 GB', resources=None): + slots, reservation = screen_resources(resources, memory) + root, image = Path(root).expanduser().resolve(), Path(image).expanduser().resolve() + output = root/'.metapathways/ptools-screens'/image.name + mapping = root/'functional_categories/MetaCyc-monomer-rxn-pairs.tsv' + # The screen manifest also validates mapping identity before reusing receipts. + command = shlex.join([sys.executable, '-m', 'metapathways.pt_screen', '-d', str(root), + '-o', str(output), '--image', str(image), '--publish', '--database_screen', '--max_tasks', str(slots)]) + return task('screen_metacyc', 'Screen MetaCyc reaction compatibility', [command], + [str(image), str(mapping)], [str(root/'functional_categories/ptools_reaction_compatibility.json')], + ['prepare_metacyc'], memory=reservation, cpus=slots, adopt_existing=False) + + +def screen_resources(args, per_container_memory): + from metapathways.nextflow import local_capacity, memory_bytes + local = getattr(args, 'executor', 'local') == 'local' + available_cpus, available_memory = local_capacity() if local else (None, None) + cpu_budget = getattr(args, 'max_cpus', None) or available_cpus + memory_budget = getattr(args, 'max_memory', None) or available_memory + requested = getattr(args, 'max_tasks', None) or (cpu_budget if local else 4) + slots = min(requested, cpu_budget) if cpu_budget else requested + per_container = memory_bytes(per_container_memory) + if memory_budget: + capacity = memory_bytes(memory_budget)//per_container + if capacity < 1: + raise ValueError('A screening container requests more memory than the available screening budget') + slots = min(slots, capacity) + # The outer task reserves resources for every nested single-CPU PTools container. + reservation = f'{slots * per_container / (1024 ** 2):.6f} MB' + print(f'Reaction screening: {slots} concurrent containers; 1 CPU and ' + f'{per_container_memory} per container; total reservation {reservation}', flush=True) + return slots, reservation diff --git a/metapathways/nf_worker.py b/metapathways/nf_worker.py new file mode 100644 index 0000000..5ca6464 --- /dev/null +++ b/metapathways/nf_worker.py @@ -0,0 +1,184 @@ +"""Execute one planned task; preserve MP outputs and check resumability.""" +import glob +import fcntl +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import time +import traceback +import tempfile +from contextlib import ExitStack + + +def fingerprint(paths): + result = [] + for value in paths: + if not value: + result.append([value, None]) + continue + matches = sorted(glob.glob(str(value))) or [str(value)] + for name in matches: + p = Path(name) + if not p.exists() and Path(name + '.gz').exists(): + p = Path(name + '.gz') + if p.exists(): + st = p.stat() + result.append([str(p), st.st_size, st.st_mtime_ns]) + if p.is_dir(): + # A directory mtime alone does not detect edits to its files. + for child in sorted(p.rglob('*')): + if child.is_file(): + st = child.stat() + result.append([str(child), st.st_size, st.st_mtime_ns]) + else: + result.append([str(p), None]) + return result + + +def available(paths): + return bool(paths) and all(value and (Path(value).exists() or Path(str(value)+'.gz').exists()) for value in paths) + + +def atomic_json(p, obj): + p = Path(p) + p.parent.mkdir(parents=True, exist_ok=True) + # Concurrent nodes must never share a staging filename, even when a + # filesystem does not provide cross-node advisory locking. + temp = None + try: + with tempfile.NamedTemporaryFile(mode='w', encoding='utf-8', + prefix=f'.{p.name}.', suffix='.tmp', + dir=p.parent, delete=False) as handle: + temp = Path(handle.name) + json.dump(obj, handle, indent=2) + handle.write('\n') + temp.replace(p) + finally: + if temp is not None: + temp.unlink(missing_ok=True) + + +def execute(t): + print(f"Starting {t['label']} (cpus={t['cpus']}, memory={t['memory']})", flush=True) + start = time.monotonic() + receipt_path = Path(t['receipt']) + previous = json.loads(receipt_path.read_text()) if receipt_path.exists() else None + signature_data = {k: t.get(k) for k in ['commands', 'context', 'cpus', 'kind']} + if 'cache_version' in t: + signature_data['cache_version'] = t['cache_version'] + if signature_data['context']: + signature_data['context'] = {k: v for k, v in signature_data['context'].items() if k not in ('status', 'message')} + signature = hashlib.sha256(json.dumps(signature_data, sort_keys=True).encode()).hexdigest() + input_state = fingerprint(t.get('fingerprint_inputs', t['inputs'])) + output_state = fingerprint(t.get('cache_outputs', t['outputs'])) + status = 'FAILED' + error = '' + def log(status, duration): + if t.get('kind') == 'annotation': + p = Path(t['sample_output']) / 'metapathways_steps_log.txt' + with p.open('a') as f: + f.write(f"{t['context']['name']}\t{status} - Time elapsed: {duration:.2f} seconds\n") + try: + if t['status'] == 'skip' or (t.get('skip_if_missing') and not Path(t['skip_if_missing']).is_file()): + status = 'SKIPPED' + elif (t['status'] != 'redo' and available(t['outputs']) and + ((previous and previous['status'] == 'SUCCESS' and previous['signature'] == signature + and previous['inputs'] == input_state and previous['outputs'] == output_state) + or (previous is None and t.get('adopt_existing', True)))): + status = 'ALREADY_COMPUTED' + else: + if any(not value or (not glob.glob(str(value)) and not Path(str(value)+'.gz').exists()) for value in t['inputs']): + raise RuntimeError(f"Missing required input: {t['inputs']}") + # Invalidate the previous success before touching outputs, even if killed. + atomic_json(receipt_path, dict(status='RUNNING')) + for command in t['commands']: + print('Command: ' + command, flush=True) + if t.get('kind') == 'annotation': + from types import SimpleNamespace + from metapathways.context import Context + from metapathways.metapathways_utils import WorkflowLogger + from metapathways.execution import execute as run_stage + c = Context() + c.__dict__.update(t['context']) + c.removeOutput() + base = Path(t['sample_output']) + s = SimpleNamespace(errorlogger=WorkflowLogger(str(base/'errors_warnings_log.txt'), open_mode='a'), + runstatslogger=WorkflowLogger(str(base/'run_statistics'/f"{t['sample']}.run.stats.txt"), open_mode='a')) + result = run_stage(s, c) + if result is None or result[0] != 0: + raise RuntimeError(f'Stage failed: {result}') + else: + for cmd in t['commands']: + subprocess.run(['bash', '-euo', 'pipefail', '-c', cmd], check=True) + if not available(t['outputs']): + raise RuntimeError(f"Task did not produce expected outputs: {t['outputs']}") + status = 'SUCCESS' + except Exception as exc: + error = str(exc) + traceback.print_exc() + duration = time.monotonic() - start + log(status, duration) + record = dict(task=t['id'], status=status, elapsed_seconds=duration, signature=signature, + inputs=input_state, outputs=fingerprint(t.get('cache_outputs', t['outputs'])), error=error) + # Preserve the cache's SUCCESS on a cache hit; record this invocation separately. + if status != 'SKIPPED': + saved = dict(record, status='SUCCESS' if status == 'ALREADY_COMPUTED' else status) + atomic_json(receipt_path, saved) + atomic_json('receipt.json', record) + if t.get('invocation_receipt'): + atomic_json(t['invocation_receipt'], record) + print(f"{t['label']}: {status} ({duration:.2f}s)", flush=True) + if status == 'FAILED' and not t.get('allow_failure', False): + return 1 + if status == 'FAILED': + print(f"Optional-entity task failed; inspect its diagnostics: {t['label']}: {error}", file=sys.stderr) + return 0 + + +def main(argv=None): + argv = list(sys.argv[1:] if argv is None else argv) + direct = argv and argv[0] == '--execute' + if direct: + argv.pop(0) + manifest, identifier = argv + tasks = json.loads(Path(manifest).read_text()) + t = next(t for t in tasks if t['id'] == identifier) + if direct: + with ExitStack() as contexts: + if t.get('compact_results') and not t['id'].endswith(':compact_results'): + from metapathways.compact_storage import task_scratch + work = contexts.enter_context(task_scratch(t.get('scratch_dir'))) + os.environ['TMPDIR'] = str(work) + os.environ['METAPATHWAYS_COMPACT_SCRATCH'] = str(work) + tempfile.tempdir = None + if t.get('host_serial'): + lock = contexts.enter_context(open(f'/tmp/metapathways-ptools-{os.getuid()}.lock', 'a')) + fcntl.flock(lock, fcntl.LOCK_EX) + return execute(t) + logfile = Path(t['log']) + logfile.parent.mkdir(parents=True, exist_ok=True) + env = os.environ.copy() + env["METAPATHWAYS_STREAM_TOOLS"] = "1" + # Prevent hidden BLAS/OpenMP pools from exceeding the scheduled CPU request. + for key in ('OMP_NUM_THREADS', 'OPENBLAS_NUM_THREADS', 'MKL_NUM_THREADS', 'NUMEXPR_NUM_THREADS'): + env[key] = str(t['cpus']) + with logfile.open('w') as log: + process = subprocess.Popen([sys.executable, '-u', '-m', 'metapathways.nf_worker', + '--execute', manifest, identifier], env=env, + stdout=subprocess.PIPE, stderr=subprocess.STDOUT, + text=True, errors='replace', bufsize=1) + try: + for line in process.stdout: + log.write(line) + log.flush() + print(line, end='', flush=True) + return process.wait() + finally: + process.stdout.close() + + +if __name__ == '__main__': + sys.exit(main()) diff --git a/metapathways/pipeline.py b/metapathways/pipeline.py index cc7d661..269eee1 100755 --- a/metapathways/pipeline.py +++ b/metapathways/pipeline.py @@ -27,6 +27,7 @@ from metapathways import sampledata as sampledata from metapathways import general_utils as gutils from metapathways import _version + from metapathways import nextflow except: print("""Could not load some user defined module functions""") print(traceback.print_exc(10)) @@ -58,14 +59,14 @@ -def runParser(): +def runParser(command='run'): parser = argparse.ArgumentParser(description='MetaPathways command-line tool for annotation of contigs.') subparsers = parser.add_subparsers(dest="command") run_parser = subparsers.add_parser( - 'run', + command, description='Minimum REQUIRED Command:\n' 'MetaPathways run -i INPUT_FILE -o OUTPUT_DIR -d REFDB_DIR\n\n', - usage='Metapathways run [options]', formatter_class=argparse.RawTextHelpFormatter) + usage=f'Metapathways {command} [options]', formatter_class=argparse.RawTextHelpFormatter) # Minimum required args req_args = run_parser.add_argument_group('Minimum Required Arguments') @@ -171,12 +172,15 @@ def runParser(): pipe_args.add_argument('--force_redo', action="store_true", default=False, help="Redo all steps [False]") + nextflow.add_resources(run_parser) + run_parser.add_argument('--dryrun', action='store_true', help='show the execution plan without running tasks') + # Other arguments misc_args = run_parser.add_argument_group('Miscellaneous Arguments') misc_args.add_argument("-s", "--samples", nargs='+', action="append", default=[], help="process only specific samples, space-separated list") - misc_args.add_argument("-t", "--threads", default=1, type=int, - help="max number of cores to use in multithreaded steps [1]") + misc_args.add_argument("-t", "--threads", default=8, type=nextflow.positive, + help="threads per capable tool [8], capped by --max_cpus; serial stages use one CPU") misc_args.add_argument("-v", "--verbose", action="store_true", default=False, help="print more information on the stdout") misc_args.add_argument("--test", action="store_true", help="use test values for all arguments") @@ -198,6 +202,7 @@ def msParser(): mag_parser.add_argument("-m", "--contig_mag_map", dest="mag_map", required=True, help="TSV file that contains contig-to-MAG mapping [REQUIRED]") + nextflow.add_resources(mag_parser) return parser @@ -216,31 +221,42 @@ def ptParser(): help="Custom name for ePGDB [optional]") ptools_parser.add_argument("--container", action="store_true", dest="container", default=False, help="Flag only used in containerized env [special flag]") - ptools_parser.add_argument('--taxprune', action="store_true", dest="taxprune", default=False, - help='Set taxonomic pruning in pathway tools to True') + from metapathways.pt_taxonomy import add_pruning_options + add_pruning_options(ptools_parser) + from metapathways.pt_taxonomy import add_taxonomy_options + add_taxonomy_options(ptools_parser) + ptools_parser.add_argument('--no_transport_inference', action='store_true', help='Disable TIP transport inference (SIF only)') + ptools_parser.add_argument('--entity', help='Build only community or the specified MAG ID') + ptools_parser.add_argument('--image', help='Pathway Tools SIF [registered by build_pt]') + from metapathways.nextflow import add_resources + add_resources(ptools_parser, task_memory='4 GB') return parser def blParser(DBS_FUNC, DBS_FUNC_DEFAULT, ALIGNERS): parser = argparse.ArgumentParser(description='automated database install') db = parser.add_argument_group(title="database arguments") - db.add_argument("-d", "--refdb_dir", metavar="PATH", required=False, default='./', + db.add_argument("-d", "--refdb_dir", metavar="PATH", required=False, default=None, help="path to save the reference DB, [DEFAULT \"./\"]") db.add_argument("--func", metavar="CATEGORICAL", nargs='*', required=False, default=DBS_FUNC_DEFAULT, help=f"functional references, select any combination from {DBS_FUNC}, [DEFAULT {DBS_FUNC_DEFAULT}]") db.add_argument("-a", "--aligner", required=False, default="fast", help=f"local aligner to index for, select one of {ALIGNERS}, [DEFAULT fast]") + db.add_argument('--skip_pt_screen', action='store_true', help='Skip default PTools reaction compatibility screening for MetaCyc') + db.add_argument('--screen_image', help='PTools SIF for screening a MetaCyc directory [registered SIF]') + db.add_argument('--metacyc_source', help='licensed MetaCyc data directory or Pathway Tools SIF [registered SIF when --func includes metacyc]') # "options" group - parser.add_argument("-t", "--threads", metavar="INT", type=int, required=False, default=1, - help="max number of cores to use in multithreaded steps [1]") + parser.add_argument("-t", "--threads", metavar="INT", type=nextflow.positive, required=False, default=None, + help="total database-build CPU budget [available CPUs]") parser.add_argument("--dryrun", action="store_true", default=False, required=False, - help="dry run snakemake") + help="show the database execution plan without running tasks") parser.add_argument("--snakemake", nargs='*', required=False, default=[], - help="additional snakemake cli args in the form of KEY=\"VALUE\" or KEY (no leading dashes)") + help="legacy compatibility flags; use the resource flags for new runs") - parser.add_argument("--test", action="store_true", help="use test values for all arguments") + parser.add_argument("--test", action="store_true", help="build test SwissProt/SILVA references; use -d for the prepared MPDB directory") + nextflow.add_resources(parser) return parser @@ -284,7 +300,7 @@ def check_for_error_in_input_file_name(shortname, globalerrorlogger=None): gutils.eprintf("ERROR\t%s\n",errmessage) if globalerrorlogger: globalerrorlogger.printf("ERROR\t%s\n",errmessage) - mputils.exit_process() + raise ValueError(errmessage) return False @@ -399,10 +415,15 @@ def create_arg_dict(parser): def run(): argv = sys.argv - gutils.eprintf("%-10s:%s\n" % ('COMMAND', ' '.join(argv))) parser = runParser() args = parser.parse_args() - # Check for the --test flag and set test values if present + tasks, output_dir = prepare_annotation(args, parser) + nextflow.launch(tasks, output_dir, args, 'run', dryrun=args.dryrun) + print('MetaPathways processing complete.') + + +def prepare_annotation(args, parser): + # Shared planner: creates tasks without starting any tools. if args.test: args.refdb_dir = str(pathlib.Path(path.abspath(__file__)).parent.joinpath("regtests/test_db")) args.rRNA_refdbs = ["SILVA_SSU_test", "SILVA_LSU_test"] @@ -427,6 +448,12 @@ def run(): elif args.refdb_dir == None: parser.error(f"Reference DB is a required argument: \"-d\", \"--refdb_dir\"") + for key in ('input_file', 'output_dir', 'refdb_dir', 'fwd_fastq', 'rev_fastq'): + value = getattr(args, key, None) + if value and value != 'None': + setattr(args, key, str(pathlib.Path(value).expanduser().absolute())) + budget = args.max_cpus or (args.threads if args.executor == 'slurm' else nextflow.local_capacity()[0]) + args.threads = min(args.threads, budget) if args.force_redo: steps_list = ['PREPROCESS_INPUT', 'ORF_PREDICTION', 'FILTER_AMINOS', 'SCAN_rRNA', 'SCAN_tRNA', 'FUNC_SEARCH', 'PARSE_FUNC_SEARCH', 'ANNOTATE_ORFS', @@ -461,8 +488,9 @@ def run(): """ load the sample inputs it expects either a fasta file or a directory containing fasta and yaml file pairs """ - print("output dir", output_dir) - globalerrorlogger = mputils.WorkflowLogger(mputils.generate_log_fp(output_dir, basefile_name = 'global_errors_warnings'), open_mode='w') + if not getattr(args, '_analysis_planning', False): + print(f"Output directory: {output_dir}", flush=True) + globalerrorlogger = mputils.WorkflowLogger(mputils.generate_log_fp(output_dir, basefile_name = 'global_errors_warnings'), open_mode='a') input_output_list = {} if path.isfile(input_fp): """ check if it is a file """ @@ -475,7 +503,7 @@ def run(): """ must be an error """ gutils.eprintf("ERROR\tNo valid input sample file or directory containing samples exists .!") gutils.eprintf("ERROR\tAs provided as arguments in the -in option.!\n") - mputils.exit_process("ERROR\tAs provided as arguments in the -in option.!\n") + parser.error(f'Input path does not exist: {input_fp}') """ these are the subset of sample to process if specified in case of an empty subset process all the sample """ @@ -489,7 +517,7 @@ def run(): #stop on invalid samples if not halt_on_invalid_input(input_output_list, filetypes, sample_subset): - globalerrorlogger.printf("ERROR\tInvalid inputs found. Check for file with bad format or characters!\n") + parser.error('Invalid input sequence format; see the error log.') # make sure the sample files are found report_missing_filenames(input_output_list, sample_subset, logger=globalerrorlogger) @@ -501,7 +529,7 @@ def run(): if not diagnoze.staticDiagnose(params, config, logger = globalerrorlogger): gutils.eprintf("ERROR\tFailed to pass the test for required scripts and inputs before run\n") globalerrorlogger.printf("ERROR\tFailed to pass the test for required scripts and inputs before run\n") - return + parser.exit(1, "Input or dependency checks failed; see the error log.\n") samplesData = {} # PART1 before the blast @@ -537,7 +565,8 @@ def run(): try: # load the sample information - print(f"RUNNING MetaPathways: v{__version__}") + if not getattr(args, '_analysis_planning', False): + print(f"RUNNING MetaPathways: v{__version__}", flush=True) if len(input_output_list): for input_file in sorted_input_output_list: sample_output_dir = input_output_list[input_file] @@ -566,51 +595,32 @@ def run(): s.prepareToRun() samplesData[input_file] = s - # load the sample information - mpsteps.run_metapathways( - samplesData, - sample_output_dir, - output_dir, - globallogger = globalerrorlogger, - command_line_params = command_line_params, - params = params, - status_update_callback = status_update_callback, - config_settings = configs, - run_type = run_type, - block_mode = block_mode - ) + tasks = nextflow.annotation_tasks(samplesData, params, configs, args.memory) + return tasks, output_dir else: gutils.eprintf("ERROR\tNo valid input files/Or no files specified to process in folder %s!\n",gutils.sQuote(input_fp) ) - globalerrorlogger.printf("ERROR\tNo valid input files to process in folder %s!\n",gutils.sQuote(input_fp) ) + raise ValueError(f'No valid inputs in {input_fp}') - except: - mputils.exit_process(str(traceback.format_exc(10)), logger= globalerrorlogger ) + except Exception as exc: + parser.exit(1, f"MetaPathways run failed: {exc}\n") - gutils.eprintf(" *********** \n") - gutils.eprintf("INFO : FINISHED PROCESSING THE SAMPLES \n") - gutils.eprintf(" THE END \n") - gutils.eprintf(" *********** \n") - - def build_db(): argv = sys.argv - gutils.eprintf("%-10s:%s\n" % ('COMMAND', ' '.join(argv))) - DBS_FUNC = "swissprot cazy eggnog uniref50 uniref90".split(" ") + DBS_FUNC = "swissprot cazy eggnog uniref50 uniref90 metacyc".split(" ") DBS_FUNC_DEFAULT = "swissprot".split(" ") ALIGNERS = "fast blast".split(" ") parser = blParser(DBS_FUNC, DBS_FUNC_DEFAULT, ALIGNERS) args = parser.parse_args(argv[2:]) - build_dbs_src = pathlib.Path(path.abspath(__file__)).parent.joinpath("build_DBs") # Check for the --test flag and set test values if present if args.test: - args.refdb_dir = pathlib.Path(path.abspath(__file__)).parent.joinpath("regtests/test_db") + args.refdb_dir = args.refdb_dir or pathlib.Path(path.abspath(__file__)).parent.joinpath("regtests/test_db") args.func = ["swissprot_test"] args.aligner = "fast" args.threads = 1 - snakefile = f"""{build_dbs_src}/Snakefile_test""" + # Set required arguments to False when --test is used for action in parser._actions: @@ -619,7 +629,7 @@ def build_db(): # Validate required arguments if --test is not used else: - snakefile = f"""{build_dbs_src}/Snakefile""" + missing_required = [] for action in parser._actions: if action.required and getattr(args, action.dest, None) is None: @@ -628,6 +638,7 @@ def build_db(): if missing_required: parser.error(f"The following arguments are required: {', '.join(missing_required)}") + args.refdb_dir = args.refdb_dir or "./" input_error = False help_printed = False @@ -652,90 +663,36 @@ def _error(message: str): if input_error: sys.exit(1) - # --------------------------------------------------------------------------- - # setup folders, params, and wget passwords for snakemake - - ref_db_dir = pathlib.Path(args.refdb_dir).absolute() - if not ref_db_dir.exists(): makedirs(ref_db_dir) - smk_temp_dir = ref_db_dir.joinpath("temp_cache") - - smk_args = [ - "--latency-wait 0", - "--keep-going", - "--rerun-incomplete", - ] - if args.dryrun: smk_args.append("--dryrun") - for smk_arg in args.snakemake: - if "=" in smk_arg: - smk_arg_tokens = smk_arg.split("=") - parsed_smk_arg = f"--{smk_arg_tokens[0]} {'='.join(smk_arg_tokens[1:])}" - else: - parsed_smk_arg = f"--{smk_arg}" - smk_args.append(parsed_smk_arg) - - smk_config = dict( - ref_db_dir=ref_db_dir, - script_path=build_dbs_src, - aligner=alinger, - functional_db_names=','.join(selected_dbs_functional) - ) - - #if "metacyc" in selected_dbs_functional: - # gutils.eprintf("Your selected database type requires a MetaCyc download...\n") - # # Get the username and password securely - # metacyc_user = input("Enter your MetaCyc username: ") - # metacyc_pswd = getpass.getpass("Enter your MetaCyc password: ") - # add_netrc_entry("brg-files.ai.sri.com", f"{metacyc_user}", f"{metacyc_pswd}") - - # --------------------------------------------------------------------------- - # run snakemake - - # set $XDG_CACHE_HOME so that snakemake doesn't polute $HOME - cmd = f"""\ - mkdir -p {smk_temp_dir} - export XDG_CACHE_HOME={smk_temp_dir} - snakemake -p -s "{snakefile}" \ - -d {ref_db_dir} \ - --cores {args.threads} \ - --config {' '.join([f'{k}="{v}"' for k, v in smk_config.items()])} \ - {' '.join(smk_args)} \ - && rm -r {smk_temp_dir} - """ - cmd = " ".join(l for l in cmd.split(" ") if l != "").strip() # remove indetation - gutils.eprintf("-"*30+"\n") - gutils.eprintf(cmd) - gutils.eprintf("\n"+"-"*30+"\n") - system(cmd) - -def add_netrc_entry(machine, login, password): - # Get the user's home directory - home_directory = path.expanduser("~") - - # Define the path to the .netrc file - netrc_path = path.join(home_directory, ".netrc") - - # todo: overwrite previous, otherwise may be stuck with wrong password - # Check if the .netrc file already exists - if path.exists(netrc_path): - # Read the existing .netrc file - with open(netrc_path, 'r') as netrc_file: - existing_entries = netrc_file.read() - # Check if an entry for the specified machine already exists - if f"machine {machine}" in existing_entries: - print(f"An entry for '{machine}' already exists in .netrc. Not adding a duplicate entry.") - return - - # If it doesn't exist or the entry doesn't exist, open the .netrc file in append mode - with open(netrc_path, 'a') as netrc_file: - # Write the new entry to the file - netrc_file.write(f"machine {machine}\n") - netrc_file.write(f"login {login}\n") - netrc_file.write(f"password {password}\n") + from metapathways.nf_databases import plan + # Preserve the old build_db -t total-core limit, while run -t controls searches. + if args.max_cpus is None and args.threads is not None: + args.max_cpus = args.threads + force = False + for item in args.snakemake: + key, _, value = item.partition('=') + if key in ('cores', 'jobs') and value: + args.max_cpus = nextflow.positive(value) + elif key in ('dryrun', 'dry-run'): + args.dryrun = True + elif key == 'forceall': + force = True + elif key not in ('keep-going', 'rerun-incomplete', 'printshellcmds', 'latency-wait'): + parser.error(f'Legacy --snakemake option {key!r} has no supported Nextflow equivalent; use resource flags') + try: + if args.metacyc_source and 'metacyc' not in args.func: + parser.error('--metacyc_source requires --func metacyc (optionally alongside other databases)') + tasks = plan(args.refdb_dir, args.func, args.aligner, test=args.test, memory=args.memory, + metacyc_source=args.metacyc_source, skip_pt_screen=args.skip_pt_screen, screen_image=args.screen_image, resources=args) + if force: + for t in tasks: + t['status'] = 'redo' + nextflow.launch(tasks, args.refdb_dir, args, 'build_db', dryrun=args.dryrun) + except (ValueError, RuntimeError, OSError, subprocess.CalledProcessError) as exc: + parser.exit(1, f'build_db: {exc}\n') def mag_split(): argv = sys.argv - gutils.eprintf("%-10s:%s\n" % ('COMMAND', ' '.join(argv))) parser = msParser() args = parser.parse_args() @@ -743,6 +700,7 @@ def mag_split(): pf_file = path.join(args.output_dir, 'ptools/0.pf') orf_map = path.join(args.output_dir, 'ptools/orf_map.txt') orf_contig_map = glob.glob(path.join(args.output_dir, 'results/annotation_table/*.ORF_annotation_table.txt'))[0] + feature_table = orf_contig_map.removesuffix('.ORF_annotation_table.txt') + '.ptinput.tsv' contig_map = glob.glob(path.join(args.output_dir, 'preprocessed/*.mapping.txt'))[0] mag_map = args.mag_map ms_outdir = path.join(args.output_dir, 'magsplitter') @@ -754,56 +712,93 @@ def mag_split(): cmd_str = f' magsplitter -p {pf_file} -r {orf_map} -c {orf_contig_map} -m {mag_map} -i {contig_map} -o {ms_outdir}' gutils.eprintf(cmd_str + '\n') - run_command_with_realtime_output(cmd) + import shlex + cmd = [str(pathlib.Path(v).resolve()) if i in (2, 4, 6, 8, 10, 12) else v for i, v in enumerate(cmd)] + saved_map = str(pathlib.Path(ms_outdir).resolve() / 'contig_to_mag.tsv') + copy_map = shlex.join([sys.executable, '-c', 'import shutil,sys; from pathlib import Path; a,b=map(Path,sys.argv[1:]); shutil.copyfile(a,b) if a.resolve()!=b.resolve() else None', str(pathlib.Path(mag_map).resolve()), saved_map]) + tasks = [nextflow.task('mag-split', 'Split annotations into MAGs', [shlex.join(cmd), copy_map], + [str(pathlib.Path(p).resolve()) for p in (pf_file, orf_map, orf_contig_map, mag_map, contig_map, feature_table)], + [str(pathlib.Path(ms_outdir).resolve() / 'results'), saved_map], + cpus=1, memory=args.memory, adopt_existing=False, + cache_version='authoritative-mag-coordinates-v1')] + try: + nextflow.launch(tasks, args.output_dir, args, 'mag_split') + except (ValueError, RuntimeError, OSError, subprocess.CalledProcessError) as exc: + parser.exit(1, f'mag_split: {exc}\n') + def ptools(): argv = sys.argv - gutils.eprintf("%-10s:%s\n" % ('COMMAND', ' '.join(argv))) parser = ptParser() args = parser.parse_args() - if args.tag: - tag = args.tag - else: - tag = path.basename(args.output_dir.rstrip('/')) - cmd = ['pgdb_build_wf.py', '--mp_out', args.output_dir, '--tag', tag] - if args.container: - container = args.container - cmd.append('--container') - if args.taxprune: - taxprune = args.taxprune - cmd.append('--taxprune') - - cmd_str = ' '.join(cmd) - gutils.eprintf("Building ePGDBs:") - gutils.eprintf(cmd_str + '\n') - - run_command_with_realtime_output(cmd) - - -def run_command_with_realtime_output(command): - """Runs a command and prints its output in real-time.""" - process = subprocess.Popen(command, stdout=subprocess.PIPE, stderr=subprocess.STDOUT) - - # Read and print the output as it is produced - while True: - output_bytes = process.stdout.readline() - if output_bytes: - print(output_bytes.decode('utf-8', 'replace').strip()) - else: - break # Exit the loop if no more output - - # Wait for the subprocess to complete - process.wait() - - if process.returncode != 0: - print("\nThe command ended with an error.") - + from metapathways.pt_container import registered_image + from metapathways import nextflow + image = args.image or (None if args.container else registered_image()) + if args.executor == 'slurm' and not image: + parser.error('Slurm Pathway Tools tasks require a SIF; run build_pt or pass --image') + if image: + image = str(pathlib.Path(image).expanduser().resolve()) + if image and not pathlib.Path(image).is_file(): + parser.error(f'Pathway Tools image does not exist: {image}') + output = pathlib.Path(args.output_dir).resolve() + tag = args.tag or output.name + script = shutil.which('pgdb_build_wf.py') + if not script: + candidate = pathlib.Path(__file__).resolve().parent.parent / 'dev/pgdb_build_wf.py' + if not candidate.is_file(): + parser.error('pgdb_build_wf.py is missing; reinstall MetaPathways') + script = str(candidate) + import shlex + tasks = [] + entities = [('community', tag, output / 'ptools', output / 'results/pgdb/community')] + entities += [(p.name, p.name, p, output / 'results/pgdb/MAGs' / p.name) + for p in sorted((output / 'magsplitter/results').glob('*')) + if p.is_dir() and 'non_binned' not in p.name] + if args.no_transport_inference and not image: + parser.error('--no_transport_inference requires a Pathway Tools SIF') + if args.entity: + entities = [entry for entry in entities if entry[0] == args.entity] + if not entities: + parser.error('Unknown PGDB entity: ' + args.entity) + for entity, entity_tag, inputs, results in entities: + cmd = [sys.executable, script, '--mp_out', str(output), '--tag', tag, + '--entity', entity] + if image: + cmd += ['--image', image] + elif args.container: + cmd.append('--container') + cmd.append('--taxprune' if args.taxprune else '--no_taxprune') + if args.no_transport_inference: + cmd.append('--no_transport_inference') + from metapathways.pt_taxonomy import resolve_taxon + taxon_id = resolve_taxon(args) + if taxon_id is not None: + cmd += ['--taxon_id', str(taxon_id)] + tasks.append(nextflow.task('pgdb-' + entity, entity, [shlex.join(cmd)], + ([image] if image else []) + [str(inputs), str(output / 'results/annotation_table'), + str(output / 'preprocessed' / (output.name + '.fasta'))], + [str(results / (entity_tag + suffix)) for suffix in ('cyc.tar.bz2', '_pwy.tsv', '_pwy2orf.tsv')], + cpus=1, memory=args.memory, allow_failure=entity != 'community', adopt_existing=False, + cache_version='sequence-backed-pgdb-compatibility-v5', host_serial=not bool(image))) + from metapathways.pt_reactions import BLACKLIST + tasks[-1]['fingerprint_inputs'] = tasks[-1]['inputs'] + [str(BLACKLIST)] + [str(output / 'orf_prediction' / (output.name + '.cds.gff'))] + from metapathways.pt_reactions import compatibility_path + compatibility = compatibility_path(output) + if compatibility and compatibility.is_file(): + tasks[-1]['fingerprint_inputs'].append(str(compatibility)) + if not image: + args.max_tasks = 1 + print('Native Pathway Tools tasks are serialized; build_pt enables isolated parallel runs.') + try: + nextflow.launch(tasks, output, args, 'ptools') + except (ValueError, RuntimeError, OSError, subprocess.CalledProcessError) as exc: + parser.exit(1, f'ptools: {exc}\n') + return def help(): print(f"""\ MetaPathways: v{__version__} - https://metapathways.readthedocs.io https://github.com/hallamlab/MetaPathways Syntax: MetaPathways COMMAND [OPTIONS] @@ -811,10 +806,15 @@ def help(): Where COMMAND is one of : help version + prepare_test build_db + build_pt + screen_pt run + analysis_wf mag_split ptools + report for addional help, use: MetaPathways COMMAND -h @@ -824,21 +824,83 @@ def version(): print(f"MetaPathways v{__version__}") +def build_pt(): + from metapathways.pt_container import main as build_image + try: + build_image(sys.argv[2:]) + except (ValueError, RuntimeError, OSError, subprocess.CalledProcessError) as exc: + print(f'build_pt: {exc}', file=sys.stderr) + sys.exit(1) + + +def screen_pt(): + from metapathways.pt_screen import main as screen_main + screen_main(sys.argv[2:]) + + +def report(): + from metapathways.report_server import main as report_main + report_main(sys.argv[2:]) + + +def prepare_test(): + from metapathways.test_data import main as test_main + test_main(sys.argv[2:]) + + +def analysis_wf(): + from metapathways.analysis_workflow import main as analysis_main + analysis_main(sys.argv[2:]) + + def main(): if len(sys.argv) <= 1: help() return - { - "help": help, - "version": version, - "build_db" : build_db, - "run": run, - "mag_split": mag_split, - "ptools": ptools - }.get( - sys.argv[1], - help # default - )() + command = sys.argv[1] + fn = {'prepare_test': prepare_test, 'help': help, 'version': version, 'build_db': build_db, 'build_pt': build_pt, + 'screen_pt': screen_pt, 'run': run, 'analysis_wf': analysis_wf, 'mag_split': mag_split, 'ptools': ptools, 'report': report}.get(command, help) + log_dir = None + if command in ('run', 'analysis_wf', 'mag_split', 'ptools', 'build_db', 'build_pt', 'screen_pt', 'report') and not any(x in sys.argv for x in ('-h', '--help')): + probe = argparse.ArgumentParser(add_help=False) + probe.add_argument('-o', '--output_dir') + probe.add_argument('-d', '--refdb_dir') + probe.add_argument('--test', action='store_true') + options, _ = probe.parse_known_args(sys.argv[2:]) + log_dir = options.refdb_dir or '.' if command == 'build_db' else options.output_dir + if options.test and command in ('run', 'build_db'): + log_dir = './test' if command == 'run' else options.refdb_dir or pathlib.Path(__file__).parent / 'regtests/test_db' + if command == 'build_pt' and not log_dir: + from metapathways.pt_container import parser as pt_parser + log_dir = pt_parser().get_default('output_dir') + def invoke(): + try: + fn() + if command in ('run', 'analysis_wf', 'mag_split', 'ptools') and log_dir and '--dryrun' not in sys.argv: + from metapathways.reporting import build_report + report_root = pathlib.Path(log_dir).resolve() + parent_report = report_root.parent / 'reports/schema.json' + if parent_report.is_file(): + import json + if str(report_root) in json.loads(parent_report.read_text()).get('sample_paths', []): + report_root = report_root.parent + build_report(report_root) + except (ValueError, RuntimeError, OSError, subprocess.CalledProcessError) as exc: + print(f'MetaPathways: {exc}', file=sys.stderr) + sys.exit(1) + except KeyboardInterrupt: + print('MetaPathways interrupted; see the retained logs and work directory.', file=sys.stderr) + sys.exit(130) + except Exception: + traceback.print_exc() + sys.exit(1) + if log_dir: + from metapathways.cli_logging import transcript + with transcript(log_dir, sys.argv): + invoke() + else: + invoke() + # the main function of metapaths if __name__ == "__main__": diff --git a/metapathways/protein_taxonomy.py b/metapathways/protein_taxonomy.py new file mode 100644 index 0000000..7bbd192 --- /dev/null +++ b/metapathways/protein_taxonomy.py @@ -0,0 +1,50 @@ +"""Taxonomy provenance for individual protein hits and within-database LCA.""" +import re + +NOT_COMPUTED = 'Not computed' +UNCLASSIFIED = 'Unclassified' + + +def supports_taxonomy(database): + return any(name in database.lower() for name in ('swissprot', 'uniref', 'eggnog')) + + +def hit_taxid(hit, database): + database = database.lower() + if 'eggnog' in database: + match = re.match(r'^(\d+)\.', hit.get('target', '')) + else: + field = 'OX' if 'swissprot' in database else 'TaxID' if 'uniref' in database else None + if field is None: + return None + # Parsed reports historically replace '=' with spaces; accept both forms. + text = ' '.join(str(hit.get(k, '')) for k in ('product', 'comment')) + match = re.search(r'\b' + field + r'(?:\s*=\s*|\s+)(\d+)(?=\s|$)', text) + return match.group(1) if match else None + + +def taxon_label(lca, taxid): + if taxid is None or str(taxid) not in lca.taxid_to_ptaxid: + return UNCLASSIFIED + taxid = str(taxid) + return lca.get_preferred_taxonomy(taxid) or lca.id_to_name[taxid] + + +def hit_taxonomy(hit, database, lca): + if not supports_taxonomy(database): + return '', NOT_COMPUTED + taxid = hit_taxid(hit, database) + return taxid or '', taxon_label(lca, taxid) + + +def raw_lca(hits, database, lca): + """Only valid IDs from score-qualified hits within this database enter LCA.""" + eligible = [h for h in hits if h['bitscore'] >= lca.lca_min_score] + if not eligible: + return None + threshold = max(h['bitscore'] for h in eligible) * (1 - lca.lca_top_percent / 100) + ids = {hit_taxid(h, database) for h in eligible if h['bitscore'] >= threshold} + ids = sorted(i for i in ids if i in lca.taxid_to_ptaxid) + if not ids: + return None + return str(lca.getTaxonomy(ids, taxid=True, return_id=True)) diff --git a/metapathways/pt_container.py b/metapathways/pt_container.py new file mode 100644 index 0000000..5e0027f --- /dev/null +++ b/metapathways/pt_container.py @@ -0,0 +1,440 @@ +"""Build and register a private Pathway Tools image from a local installer.""" +import argparse +from datetime import datetime, timezone +import hashlib +from html.parser import HTMLParser +import json +import os +from pathlib import Path +import platform +import re +import shlex +import shutil +import subprocess +import sys +import tempfile +import time +from collections import deque +import uuid +from urllib.request import build_opener, HTTPRedirectHandler + +from metapathways import nextflow + + +# Changing the recipe produces a distinct image, even for the same installer. +RECIPE = r'''Bootstrap: docker +From: ubuntu:22.04 + +%files + installer /opt/pt-installer + official-patches /opt/mp-official-patches + +%post + set -eu + export DEBIAN_FRONTEND=noninteractive + apt-get update + apt-get install -y --no-install-recommends ca-certificates xterm openssl libxml2 xvfb xauth libxm4 libssl-dev procps bzip2 ncbi-blast+ + mkdir -p /data /opt/bin + chmod 700 /opt/pt-installer + unset DISPLAY + printf '/opt/pathway-tools\n/data\n\nn\nY\nn\n\n' | /opt/pt-installer + test -x /opt/pathway-tools/pathway-tools + rm /opt/pt-installer + # Install only the unmodified, release-specific vendor files (SRI FAQ 7.4). + pt_version=$(cat /opt/mp-official-patches/version) + patch_dir=/opt/pathway-tools/aic-export/pathway-tools/ptools/$pt_version/patches + test -d "$patch_dir" + set -- "$patch_dir"/bin-* + test "$#" -eq 1 && test -d "$1" + for patch in /opt/mp-official-patches/files/*; do + case "$patch" in + *.fasl) cp "$patch" "$1/" ;; + *) cp "$patch" "$patch_dir/" ;; + esac + done + # Load official patches while the filesystem is still writable. + xvfb-run -a /opt/pathway-tools/pathway-tools -no-patch-download -lisp -eval '(progn (format t "~%MP-PT-READY~%") (exit))' > /opt/mp-pt-patch-startup.log 2>&1 || { cat /opt/mp-pt-patch-startup.log; exit 1; } + cat /opt/mp-pt-patch-startup.log + grep -qx 'MP-PT-READY' /opt/mp-pt-patch-startup.log + printf '[ncbi]\nData=/usr/share/ncbi/data\n' > /opt/mp-ncbirc + test -d /usr/share/ncbi/data + blastp -version + makeblastdb -version + dpkg-query -W ncbi-blast+ > /opt/mp-blast-version.txt + # Keep a pristine template; each invocation mounts its own writable /data. + cp -a /data/ptools-local /opt/ptools-local-template + rm -rf /var/lib/apt/lists/* + +%environment + export PATH=/opt/pathway-tools:$PATH + +%runscript + exec /opt/pathway-tools/pathway-tools "$@" + +%labels + org.metapathways.purpose PathwayTools +''' + + +def digest(path): + h = hashlib.sha256() + with Path(path).open('rb') as f: + for block in iter(lambda: f.read(8 * 1024 * 1024), b''): + h.update(block) + return h.hexdigest() + + +class _NoRedirect(HTTPRedirectHandler): + def redirect_request(self, req, fp, code, msg, headers, newurl): + raise ValueError('Official patch URL redirected; refusing an unverified patch source: ' + newurl) + + +class _PatchLinks(HTMLParser): + def __init__(self): + super().__init__() + self.names = set() + + def handle_starttag(self, tag, attrs): + if tag == 'a': + href = dict(attrs).get('href', '') + # No arbitrary URLs, parent paths, nested directories or query strings. + if re.fullmatch(r'[A-Za-z0-9][A-Za-z0-9_.-]*\.(?:fasl|lisp|tar\.gz)', href): + self.names.add(href) + + +def installer_version(installer, explicit=None): + match = re.search(r'pathway-tools-(\d+\.\d+)', Path(installer).name) + version = explicit or (match.group(1) if match else None) + if not version or not re.fullmatch(r'\d+\.\d+', version): + raise ValueError('Cannot determine Pathway Tools release; use --ptools_version (for example 29.5) for a renamed installer') + if explicit and match and explicit != match.group(1): + raise ValueError('--ptools_version disagrees with the installer filename') + return version + + +def download_patches(version, destination): + """Snapshot SRI's official release feed; never silently use partial downloads.""" + if not re.fullmatch(r'\d+\.\d+', version): + raise ValueError('Invalid Pathway Tools release') + url = f'https://bioinformatics.ai.sri.com/ptools/{version}/Linux-64/patches/' + destination = Path(destination) + files = destination / 'files' + files.mkdir(parents=True) + opener = build_opener(_NoRedirect()) + try: + with opener.open(url, timeout=60) as response: + listing = response.read() + links = _PatchLinks() + links.feed(listing.decode('utf-8')) + if not links.names: + raise ValueError('Vendor listing contained no recognized patch files') + manifest = dict(source=url, version=version, fetched_at=datetime.now(timezone.utc).isoformat(), + listing_sha256=hashlib.sha256(listing).hexdigest(), files=[]) + for name in sorted(links.names): + print('Downloading official Pathway Tools patch: ' + name, flush=True) + target = files / name + with opener.open(url + name, timeout=60) as response, target.open('wb') as stream: + shutil.copyfileobj(response, stream) + if target.stat().st_size == 0: + raise ValueError('Empty patch: ' + name) + manifest['files'].append(dict(name=name, url=url + name, sha256=digest(target))) + manifest['snapshot_sha256'] = hashlib.sha256(json.dumps(manifest['files'], sort_keys=True).encode()).hexdigest() + (destination / 'version').write_text(version + '\n') + (destination / 'index.html').write_bytes(listing) + save_json(destination / 'manifest.json', manifest) + return manifest + except Exception as exc: + raise RuntimeError(f'Official Pathway Tools patches could not be obtained from {url}; build stopped: {exc}') from exc + + +def registry_path(): + return Path(os.environ.get('XDG_CONFIG_HOME', Path.home() / '.config')) / 'metapathways' / 'ptools.json' + + +def registered_image(): + override = os.environ.get('METAPATHWAYS_PTOOLS_IMAGE') + if override: + image = Path(override).expanduser().resolve() + elif registry_path().exists(): + image = Path(json.loads(registry_path().read_text())['image']) + else: + return None + if not image.is_file(): + raise ValueError(f'Registered Pathway Tools image is missing: {image}; run metapathways build_pt again') + return str(image) + + +def save_json(path, data): + path = Path(path) + path.parent.mkdir(parents=True, exist_ok=True) + with tempfile.NamedTemporaryFile(mode='w', dir=path.parent, delete=False) as f: + temp = Path(f.name) + json.dump(data, f, indent=2) + f.write('\n') + try: + temp.replace(path) + finally: + temp.unlink(missing_ok=True) + + +def exec_command(image, state, command): + """Use a task-private data directory, PID namespace, home and /tmp.""" + state = Path(state).resolve() + state.mkdir(parents=True, exist_ok=True) + if any(c in str(state) for c in ':,\n'): + raise ValueError('Apptainer state path cannot contain a colon, comma or newline') + executable = shutil.which('apptainer') + if not executable: + raise RuntimeError('Apptainer is required and must be on PATH') + # --containall otherwise puts /tmp and /var/tmp in the small session + # filesystem, not on the disk backing our task-private home directory. + # Keep scratch private per invocation, including Xvfb filesystem sockets. + scratch = state / 'container-work' + scratch.mkdir(mode=0o700, exist_ok=True) + return [executable, 'exec', '--containall', '--cleanenv', + '--workdir', str(scratch), + '--home', f'{state}:/data', '--pwd', '/data', str(image), *command] + + +def validate(image, directory): + script = ('cp -a /opt/ptools-local-template /data/ptools-local; ' + 'cp /opt/mp-ncbirc /data/.ncbirc; ' + 'blastp -version; makeblastdb -version; ' + 'printf ">mp_validation\\nMKWVTFISLLFLFSSAYSRGVFRRDTHKSEIAHRFKDLGE\\n" > /data/check.faa; ' + 'makeblastdb -in /data/check.faa -dbtype prot -out /data/check-db; ' + 'blastp -query /data/check.faa -db /data/check-db -outfmt 6 -out /data/check.tsv; ' + 'test -s /data/check.tsv; ' + 'exec xvfb-run -a /opt/pathway-tools/pathway-tools ' + '-no-patch-download -lisp -eval \'(progn (format t "~%MP-PT-READY~%") (exit))\'') + result = subprocess.run(exec_command(image, directory, ['sh', '-ec', script]), + check=False, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, + text=True, timeout=300) + if result.returncode: + raise RuntimeError(f'Pathway Tools startup failed (exit {result.returncode}): ' + result.stdout[-4000:]) + if 'MP-PT-READY' not in [line.strip() for line in result.stdout.splitlines()]: + raise RuntimeError('Pathway Tools startup did not emit its validation marker: ' + result.stdout[-4000:]) + if re.search(r'Error determining path to BLAST|blastall or blastp could not be located', result.stdout, re.I): + raise RuntimeError('Pathway Tools could not locate the installed BLAST executables: ' + result.stdout[-4000:]) + return result.stdout + + +def _stream_container(command, log): + """Tee output to the task console and an attempt log; keep bounded error text.""" + tail = deque(maxlen=100) + with log.open('w') as stream: + with subprocess.Popen(command, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, + text=True, errors='replace') as process: + for line in process.stdout: + stream.write(line) + stream.flush() + print(line, end='', flush=True) + tail.append(line) + return process.wait(), ''.join(tail) + + +def _run_pgdb_container(command, state, status): + """Retry only explicit mount failures before the container shell starts.""" + attempts = status['container_attempts'] = [] + for attempt in range(1, 4): + (state / 'stage.txt').write_text('container startup\n') + log = state / f'container-attempt-{attempt}.log' + started = time.monotonic() + code, tail = _stream_container(command, log) + stage = (state / 'stage.txt').read_text().strip() + retryable = (code != 0 and stage == 'container startup' + and 'container creation failed' in tail.lower() + and 'mount hook function failure' in tail.lower()) + record = dict(attempt=attempt, exit_code=code, stage=stage, + elapsed_seconds=time.monotonic() - started, log=log.name, + retryable=retryable) + attempts.append(record) + if code == 0: + return + if not retryable or attempt == 3: + raise subprocess.CalledProcessError(code, command) + delay = (5, 15)[attempt - 1] + record['retry_delay_seconds'] = delay + print(f'Apptainer startup mount failure (attempt {attempt}/3); ' + f'retrying in {delay}s. Pathway Tools has not started.', flush=True) + time.sleep(delay) + + +def run_pgdb(image, inputs, outputs, tag, taxprune=True, sample_output=None, transport_inference=True): + """Run one PGDB with no shared host Pathway Tools state.""" + if not re.fullmatch(r'[A-Za-z_][A-Za-z0-9_-]*', tag): + raise ValueError('PGDB tag must start with a letter or underscore and contain only letters, digits, underscores or hyphens') + output = Path(outputs).resolve() + output.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory(prefix='.pt-run-', dir=output.parent) as work: + state = Path(work) + shutil.copytree(inputs, state / 'input') + (state / 'output').mkdir() + script = '''set -eu +printf 'container setup\\n' > /data/stage.txt +cp -a /opt/ptools-local-template /data/ptools-local +if test -f /opt/mp-ncbirc; then cp /opt/mp-ncbirc /data/.ncbirc; fi +mkdir -p /data/blastdb /tmp/.X11-unix +# --containall isolates /tmp, but not Linux abstract sockets. Use only the +# private filesystem X socket so concurrent containers cannot collide. +printf 'build\n' > /data/stage.txt +xvfb-run -a -e /data/build-xvfb.log -s '-screen 0 1280x1024x24 -nolisten local' /opt/pathway-tools/pathway-tools -patho /data/input "$@" -no-web-cel-overview -no-patch-download -no-cel-overview -disable-metadata-saving -nologfile +''' + lisp = f"(progn (with-organism (:org-id '{tag}) (dump-frames-to-attribute-value-files (org-data-dir)))(exit))" + script += "printf 'export\\n' > /data/stage.txt\n" + script += shlex.join(['xvfb-run', '-a', '-e', '/data/export-xvfb.log', + '-s', '-screen 0 1280x1024x24 -nolisten local', '/opt/pathway-tools/pathway-tools', + '-no-patch-download', '-no-cel-overview', '-nologfile', '-eval', lisp]) + '\n' + script += "printf 'archive\\n' > /data/stage.txt\n" + script += shlex.join(['tar', '-cjf', f'/data/output/{tag}cyc.tar.bz2', + '-C', f'/data/ptools-local/pgdbs/user/{tag.lower()}cyc', '.']) + '\n' + command = ['sh', '-ec', script, 'run-pgdb'] + if transport_inference: + command.append('-tip') + if not taxprune: + command.append('-no-taxonomic-pruning') + invocation = exec_command(image, state, command) + diagnostics = output / 'diagnostics' / uuid.uuid4().hex + status = dict(tag=tag, image=str(image), command=invocation, status='FAILED') + try: + if sample_output is not None: + from metapathways.pt_sequences import attach_sequences + (state / 'stage.txt').write_text('input preparation\n') + attach_sequences(state / 'input', sample_output) + from metapathways.pt_reactions import filter_reactions + filter_reactions(state / 'input', image=image, sample_output=sample_output) + _run_pgdb_container(invocation, state, status) + archive = state / 'output' / f'{tag}cyc.tar.bz2' + if not archive.is_file() or not archive.stat().st_size: + raise RuntimeError(f'Pathway Tools produced no PGDB archive for {tag}') + archive.replace(output / archive.name) + status['status'] = 'SUCCESS' + except BaseException as exc: + status['error'] = str(exc) + status['exit_code'] = getattr(exc, 'returncode', None) + # Pathologic redirects its own errors away from the parent's stderr. + private_log = state / 'input/pathologic.log' + if private_log.is_file(): + from collections import deque + with private_log.open(errors='replace') as stream: + tail = ''.join(deque(stream, maxlen=80)) + print(f'Pathway Tools internal log tail ({tag}):\n{tail}', flush=True) + raise + finally: + # Preserve internal diagnostics before TemporaryDirectory removes state, + # on success, failure, or an interrupted subprocess. + diagnostics.mkdir(parents=True, exist_ok=True) + stage = state / 'stage.txt' + status['stage'] = stage.read_text().strip() if stage.is_file() else 'container startup' + for source in state.rglob('*'): + if source.is_file() and (source.suffix.lower() in ('.log', '.err', '.out') + or source.name in ('stage.txt', 'sequence-input.json', 'ptools-reaction-filter.json', 'organism-params.dat') + or 'reports' in source.relative_to(state).parts): + destination = diagnostics / source.relative_to(state) + destination.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(source, destination) + if status['status'] != 'SUCCESS': + # Retain the last on-disk PGDB for separate recovery attempts. + # Moving avoids copying a potentially large database before cleanup. + user_pgdbs = state / 'ptools-local/pgdbs/user' + if user_pgdbs.is_dir(): + retained = diagnostics / 'failed-pgdbs' + shutil.move(str(user_pgdbs), str(retained)) + status['retained_pgdbs'] = str(retained) + shutil.copytree(state / 'input', diagnostics / 'input', dirs_exist_ok=True) + save_json(diagnostics / 'execution.json', status) + print(f'Pathway Tools {status["status"]} during {status["stage"]}; diagnostics: {diagnostics}', flush=True) + + +def build(installer, image, installer_sha256, threads, version=None): + """Worker entry point: never publish or register a failed build.""" + installer, image = Path(installer).resolve(), Path(image).resolve() + if digest(installer) != installer_sha256: + raise ValueError('Installer changed after the build was planned') + executable = shutil.which('apptainer') + if not executable: + raise RuntimeError('Apptainer is required and must be on PATH') + image.parent.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory(prefix='.pt-build-', dir=image.parent) as work: + work = Path(work) + # A fixed relative name avoids definition-file quoting of arbitrary paths. + (work / 'installer').symlink_to(installer) + patches = download_patches(installer_version(installer, version), work / 'official-patches') + definition = work / 'pathway-tools.def' + definition.write_text(RECIPE) + partial = work / 'pathway-tools.sif' + subprocess.run([executable, 'build', '--fakeroot', '--mksquashfs-args', + f'-processors {threads}', str(partial), str(definition)], + cwd=work, check=True) + validation_log = validate(partial, work / 'validation') + metadata = dict(installer_sha256=installer_sha256, image=str(image), + image_sha256=digest(partial), recipe_sha256=hashlib.sha256(RECIPE.encode()).hexdigest(), + official_patches=patches, validation_log=validation_log) + partial.replace(image) + save_json(str(image) + '.json', metadata) + image.with_suffix('.def').write_text(RECIPE) + + +def parser(): + p = argparse.ArgumentParser(prog='metapathways build_pt', + description='Build a Pathway Tools SIF from a local Linux installer using Nextflow and Apptainer.') + p.add_argument('-i', '--installer', required=True, help='local Pathway Tools Linux x86-64 installer') + p.add_argument('--ptools_version', help='release number if the installer was renamed; otherwise inferred from its filename') + p.add_argument('-d', '--refdb_dir', help='also export and prepare the licensed MetaCyc reference in this MPDB after building the SIF') + p.add_argument('--skip_pt_screen', action='store_true', help='Skip default reaction compatibility screening when building MetaCyc with -d') + p.add_argument('-a', '--aligner', choices=('fast', 'blast'), default='fast', help='MetaCyc reference index format with -d [fast]') + default = Path(os.environ.get('XDG_DATA_HOME', Path.home() / '.local/share')) / 'metapathways/containers' + p.add_argument('-o', '--output_dir', default=str(default), help='container directory [~/.local/share/metapathways/containers]') + p.add_argument('-t', '--threads', type=nextflow.positive, default=2, help='CPUs for image compression [2]; Pathway Tools uses one CPU') + p.add_argument('--dryrun', action='store_true', help='write and show the build plan without building or registering') + nextflow.add_resources(p, task_memory='4 GB') + return p + + +def main(argv=None): + p = parser() + args = p.parse_args(argv) + installer = Path(args.installer).expanduser().resolve() + if not installer.is_file(): + p.error(f'Installer does not exist: {installer}') + if platform.system() != 'Linux' or platform.machine() not in ('x86_64', 'AMD64'): + p.error('Building this Pathway Tools image requires Linux x86-64') + if not args.dryrun: + for executable in ('apptainer', 'nextflow'): + if not shutil.which(executable): + p.error(f'{executable} must be installed and on PATH') + checksum = digest(installer) + recipe_hash = hashlib.sha256(RECIPE.encode()).hexdigest() + try: + version = installer_version(installer, args.ptools_version) + except ValueError as exc: + p.error(str(exc)) + output = Path(args.output_dir).expanduser().resolve() + # Each explicit build refreshes the vendor feed and creates a separate SIF. + # Never overwrite an image that existing analysis receipts/running jobs use. + build_id = uuid.uuid4().hex + image = output / f'pathway-tools-{version}-{checksum[:12]}-{recipe_hash[:8]}-{build_id[:12]}.sif' + command = shlex.join([sys.executable, '-m', 'metapathways.pt_container', '--worker', + str(installer), str(image), checksum, str(args.threads), version]) + t = nextflow.task('build-pt-' + checksum + recipe_hash + build_id, 'Build Pathway Tools ' + version, + [command], [str(installer)], [str(image), str(image) + '.json'], + cpus=args.threads, memory=args.memory, adopt_existing=False) + tasks = [t] + if args.refdb_dir: + from metapathways.nf_databases import metacyc_task, screen_task + tasks.append(metacyc_task(args.refdb_dir, image, args.aligner, args.memory, [t['id']])) + if not args.skip_pt_screen: + tasks.append(screen_task(args.refdb_dir, image, args.memory, resources=args)) + nextflow.launch(tasks, output, args, 'build_pt', dryrun=args.dryrun) + if not args.dryrun: + metadata = json.loads(Path(str(image) + '.json').read_text()) + if metadata['image_sha256'] != digest(image): + raise RuntimeError('Built image checksum does not match its metadata') + save_json(registry_path(), metadata) + print(f'Pathway Tools image: {image}\nRegistered in: {registry_path()}') + + +if __name__ == '__main__': + if len(sys.argv) > 1 and sys.argv[1] == '--worker': + build(*sys.argv[2:5], int(sys.argv[5]), sys.argv[6]) + else: + main() diff --git a/metapathways/pt_ec.py b/metapathways/pt_ec.py new file mode 100644 index 0000000..23deffa --- /dev/null +++ b/metapathways/pt_ec.py @@ -0,0 +1,19 @@ +"""Normalize multi-valued EC annotations without inventing EC assignments.""" +import re + + +def normalize_ecs(values): + """Return distinct EC tokens in input order; preserve provisional identifiers.""" + if values is None: + return [] + if isinstance(values, str): + values = [values] + result = [] + seen = set() + for value in values: + for token in re.split(r'[,;|]', str(value)): + token = token.strip() + if token and token not in seen: + seen.add(token) + result.append(token) + return result diff --git a/metapathways/pt_exports.py b/metapathways/pt_exports.py new file mode 100644 index 0000000..7c6884c --- /dev/null +++ b/metapathways/pt_exports.py @@ -0,0 +1,51 @@ +"""Handle Pathway Tools' empty-pathway export without altering the raw PGDB.""" +from pathlib import Path + + +EMPTY_EXPORT_ERROR = ( + "Error: The flat-file generation program didn't specify what kind of data " + "to put in this file." +) +PATHWAY_COLUMNS = ( + 'SAMPLE', 'PWY_NAME', 'PWY_COMMON_NAME', 'PWY_SCORE', 'NUM_REACTIONS', + 'NUM_COVERED_REACTIONS', 'ORF_COUNT', 'ORFS', +) +REPORT_COLUMNS = ( + 'Pathway Name', 'Pathway Frame-id', 'Pathway Class Name', + 'Pathway Class Frame-id', 'Pathway Score', 'Pathway Frequency Score', + 'Pathway Abundance', 'Reason to Keep', 'Pathway URL', +) + + +def write_verified_empty_pathways(flatpath, outfile): + """Return True only for an independently verified empty pathway export. + + Some Pathway Tools versions export an error sentence for an empty frame + class. Do not pass this sentence to Camelot or silently discard parse errors. + Require a header-only final pathway report and an empty inference evidence + list from this PGDB. Raw exports and archives remain unchanged. + """ + flatpath = Path(flatpath) + lines = [line.strip() for line in (flatpath / 'pathways.dat').read_text().splitlines() + if line.strip() and not line.lstrip().startswith('#')] + if lines and lines != [EMPTY_EXPORT_ERROR]: + return False # Ordinary exports (including other errors) go to Camelot. + reports = flatpath.parent / 'reports' + summaries = sorted(reports.glob('pathways-report_*.txt')) + if not summaries: + raise ValueError('Cannot verify empty pathways.dat: missing final pathway report') + # Multiple reports can reflect repeated inference. Fail closed if any differs. + for report in summaries: + rows = [line.strip() for line in report.read_text().splitlines() + if line.strip() and not line.lstrip().startswith('#')] + if len(rows) != 1 or tuple(part.strip() for part in rows[0].split('|')) != REPORT_COLUMNS: + raise ValueError(f'Cannot verify empty pathways.dat: nonempty or invalid {report.name}') + evidence = (reports / 'pwy-evidence-list.dat').read_text() + if 'This file contains all inferred pathways and super-pathways' not in evidence or any( + line.strip() and not line.lstrip().startswith(';;;') + for line in evidence.splitlines() + ): + raise ValueError('Cannot verify empty pathways.dat: nonempty or invalid pathway evidence list') + Path(outfile).write_text('\t'.join(PATHWAY_COLUMNS) + '\n') + print('Pathway export: verified zero pathways; writing an empty pathway table', flush=True) + return True diff --git a/metapathways/pt_reactions.py b/metapathways/pt_reactions.py new file mode 100644 index 0000000..d6293e8 --- /dev/null +++ b/metapathways/pt_reactions.py @@ -0,0 +1,90 @@ +"""Filter known unsafe explicit reaction assignments in private PGDB staging.""" +import hashlib +import json +from pathlib import Path + +BLACKLIST = Path(__file__).parent / 'resources/ptools_reaction_blacklist.json' + + +def compatibility_path(sample_output): + if sample_output is None: + return None + log = Path(sample_output)/'metapathways_run_log.txt' + if log.is_file(): + for line in log.read_text().splitlines(): + if line.startswith('Minimum Required Arguments:refdb_dir\t'): + return Path(line.split('\t', 1)[1])/'functional_categories/ptools_reaction_compatibility.json' + return None + + +def image_digest_cached(image, sample_output): + """Hash once per sample/image identity, even when many MAG workers start together.""" + from metapathways.pt_screen import digest + if sample_output is None: + return digest(image) + import fcntl + from metapathways.nf_worker import atomic_json + image = Path(image).resolve() + stat = image.stat() + identity = [str(image), stat.st_size, stat.st_mtime_ns, stat.st_ctime_ns] + cache = Path(sample_output)/'.metapathways/ptools-image-digest.json' + cache.parent.mkdir(parents=True, exist_ok=True) + with cache.with_suffix('.lock').open('a') as lock: + fcntl.flock(lock, fcntl.LOCK_EX) + if cache.is_file(): + record = json.loads(cache.read_text()) + if record['identity'] == identity: + return record['sha256'] + checksum = digest(image) + atomic_json(cache, dict(identity=identity, sha256=checksum)) + return checksum + + +def filter_reactions(inputs, image=None, sample_output=None): + """Keep features and names intact; remove only listed METACYC assignments.""" + inputs = Path(inputs) + payload = BLACKLIST.read_bytes() + blacklist = json.loads(payload) + selected = str(BLACKLIST) + if image: + checksum = image_digest_cached(image, sample_output) + blacklist = {r: entry for r, entry in blacklist.items() + if entry.get('image_sha256') == checksum} + compatibility = compatibility_path(sample_output) + if compatibility and compatibility.is_file(): + from metapathways.pt_screen import digest + data = json.loads(compatibility.read_text()) + mapping = compatibility.parent/'MetaCyc-monomer-rxn-pairs.tsv' + if image and data['image_sha256'] == image_digest_cached(image, sample_output) and mapping.is_file() and data['mapping_sha256'] == digest(mapping): + payload = compatibility.read_bytes() + blacklist = data['reactions'] + selected = str(compatibility) + else: + print('Pathway Tools compatibility list does not match the image/mapping; using bundled known-trigger fallback', flush=True) + removals = [] + for file in sorted(inputs.glob('*.pf')): + feature = None + kept = [] + changed = False + for line in file.read_text().splitlines(keepends=True): + fields = line.rstrip('\r\n').split('\t', 1) + if fields[0] == 'ID' and len(fields) == 2: + feature = fields[1] + if fields[0] == 'METACYC' and len(fields) == 2 and fields[1].strip() in blacklist: + reaction = fields[1].strip() + removals.append(dict(file=file.name, feature_id=feature, reaction=reaction, + reason=blacklist[reaction]['reason'])) + changed = True + continue + kept.append(line) + if line.strip() == '//': + feature = None + if changed: + file.write_text(''.join(kept)) + audit = dict(blacklist_source=selected, blacklist_sha256=hashlib.sha256(payload).hexdigest(), removed=removals) + (inputs/'ptools-reaction-filter.json').write_text(json.dumps(audit, indent=2)+'\n') + if removals: + affected = {(r['feature_id'], r['reaction']) for r in removals} + print(f'Pathway Tools reaction blacklist: removed {len(affected)} explicit feature/reaction ' + 'assignments from staged inputs; annotations and feature records retained', flush=True) + return audit diff --git a/metapathways/pt_screen.py b/metapathways/pt_screen.py new file mode 100644 index 0000000..dcaf4a9 --- /dev/null +++ b/metapathways/pt_screen.py @@ -0,0 +1,256 @@ +"""Maintainer compatibility screen for explicit MetaCyc reaction assignments.""" +import argparse +from concurrent.futures import ThreadPoolExecutor +import csv +import hashlib +import json +import os +from pathlib import Path +import re +import signal +import subprocess +import tempfile +import time + +from metapathways.pt_container import exec_command, registered_image +from metapathways.nf_worker import atomic_json + +SCREEN_VERSION = 1 + + +def digest(file): + h = hashlib.sha256() + with Path(file).open('rb') as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b''): + h.update(chunk) + return h.hexdigest() + + +def reactions_from_db(directory): + file = Path(directory)/'functional_categories/MetaCyc-monomer-rxn-pairs.tsv' + with file.open() as stream: + rows = csv.DictReader(stream, delimiter='\t') + if not {'MC', 'RXN'} <= set(rows.fieldnames or []): + raise ValueError(f'Missing MC/RXN columns: {file}') + reactions = sorted({row['RXN'].strip() for row in rows if row['RXN'].strip()}) + if not reactions: + raise ValueError(f'No reaction assignments found: {file}') + if any(not re.fullmatch(r'[A-Za-z0-9_.:+-]+', r) for r in reactions): + raise ValueError('Unsafe reaction identifier in mapping table') + return reactions, file + + +def write_inputs(directory, reactions, tag): + directory.mkdir(parents=True) + count = max(1, len(reactions)) + gene = 'ATG' + 'GCT'*78 + 'TAA' + sequence = gene * count + (directory/'contig.fasta').write_text('>contig\n'+sequence+'\n') + records = [] + for i in range(count): + records.append(f'ID\tG{i+1}\nNAME\tG{i+1}\nSTARTBASE\t{i*len(gene)+1}\n' + f'ENDBASE\t{(i+1)*len(gene)}\nFUNCTION\tUncharacterized protein\n' + 'PRODUCT-TYPE\tP\n' + + (f'METACYC\t{reactions[i]}\n' if reactions else '') + '//\n') + (directory/'contig.pf').write_text(''.join(records)) + (directory/'genetic-elements.dat').write_text('ID\tcontig\nNAME\tcontig\nTYPE\t:CONTIG\n' + 'CODON-TABLE\t11\nANNOT-FILE\tcontig.pf\nSEQ-FILE\tcontig.fasta\n//\n') + (directory/'organism-params.dat').write_text(f'ID\t{tag}\nSTORAGE\tFILE\nNAME\t{tag}\n' + 'STRAIN\t1\nRANK\t|species|\nNCBI-TAXON-ID\t131567\n') + + +def execute_attempt(image, directory, reactions, timeout, scratch=None): + """Use unfiltered synthetic inputs; preserve logs and a real success marker.""" + tag = 'MPscreen' + write_inputs(directory/'input', reactions, tag) + start = time.monotonic() + with tempfile.TemporaryDirectory(prefix='mp-pt-screen-', dir=scratch) as temporary: + state = Path(temporary) + # Bind only this attempt's inputs into its private writable home. + import shutil + shutil.copytree(directory/'input', state/'input') + (state/'screen.lisp').write_text("""(progn + (handler-case + (progn + (batch-pathologic "1.0" "/data/input/" + :download-publications? nil :do-overview? nil :web-cel-ov? nil + :taxonomic-pruning? t :tip? t :suppress-metadata-saving? t + :standard-streams? t :debug? t :trap-errors? nil) + (format t "~%MP-SCREEN-SUCCESS~%") (finish-output) (exit)) + (error (e) (format t "~%MP-SCREEN-ERROR ~A~%" e) + (finish-output) (excl:exit 1)))) +""") + script = """set -eu +cp -a /opt/ptools-local-template /data/ptools-local +if test -f /opt/mp-ncbirc; then cp /opt/mp-ncbirc /data/.ncbirc; fi +exec xvfb-run -a -s '-screen 0 1280x1024x24 -nolisten local' /opt/pathway-tools/pathway-tools -no-patch-download -no-cel-overview -nologfile -lisp -load /data/screen.lisp +""" + with (directory/'console.log').open('w') as log: + process = subprocess.Popen(exec_command(image, state, ['sh', '-ec', script]), + stdout=log, stderr=subprocess.STDOUT, start_new_session=True) + timed_out = False + try: + process.wait(timeout=timeout) + except subprocess.TimeoutExpired: + timed_out = True + finally: + if process.poll() is None: + os.killpg(process.pid, signal.SIGTERM) + try: + process.wait(timeout=10) + except subprocess.TimeoutExpired: + os.killpg(process.pid, signal.SIGKILL) + process.wait() + for source in state.rglob('*.log'): + destination = directory/'diagnostics'/source.relative_to(state) + destination.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(source, destination) + text = (directory/'console.log').read_text(errors='replace') + if timed_out or process.returncode < 0 or process.returncode in (137, 143): + status = 'INCONCLUSIVE' + elif process.returncode == 0 and 'MP-SCREEN-SUCCESS' in text.splitlines(): + status = 'PASS' + elif 'MP-SCREEN-ERROR' in text: + status = 'FAIL' + else: + status = 'INCONCLUSIVE' + return dict(status=status, exit_code=process.returncode, timed_out=timed_out, + duration_seconds=round(time.monotonic()-start, 3)) + + +class Screen: + def __init__(self, args, runner=execute_attempt): + self.args, self.runner = args, runner + self.output = Path(args.output_dir).expanduser().resolve() + if getattr(args, 'database_screen', False): + mapping = Path(args.refdb_dir)/'functional_categories/MetaCyc-monomer-rxn-pairs.tsv' + self.output /= digest(mapping) + self.output.mkdir(parents=True, exist_ok=True) + self.image = Path(args.image).expanduser().resolve() + self.reactions, mapping = reactions_from_db(args.refdb_dir) + if args.reactions: + missing = set(args.reactions)-set(self.reactions) + if missing: + raise ValueError('Reactions absent from MPDB: '+', '.join(sorted(missing))) + self.reactions = sorted(set(args.reactions)) + self.identity = dict(screen_version=SCREEN_VERSION, image_sha256=digest(self.image), + mapping_sha256=digest(mapping), reactions=self.reactions, batch_size=args.batch_size, + confirm_runs=args.confirm_runs, timeout=args.timeout) + manifest = self.output/'screen.json' + if manifest.exists() and json.loads(manifest.read_text()) != self.identity: + raise ValueError('Screen inputs/settings changed; use a new output directory') + atomic_json(manifest, self.identity) + self.candidates, self.interactions, self.inconclusive = {}, [], [] + + def attempt(self, reactions, label): + key = hashlib.sha256(json.dumps([reactions, label]).encode()).hexdigest()[:20] + directory = self.output/'attempts'/key + receipt = directory/'result.json' + if receipt.exists(): + previous = json.loads(receipt.read_text()) + if previous['status'] != 'INCONCLUSIVE': + return previous + if directory.exists(): + # An interrupted attempt has no valid receipt; preserve its evidence. + directory.rename(directory.with_name(key+'-interrupted-'+str(time.time_ns()))) + directory.mkdir(parents=True) + print(f'Screen {label}: {len(reactions)} reactions; {directory}', flush=True) + result = self.runner(self.image, directory, reactions, self.args.timeout, self.args.scratch_dir) + result.update(reactions=reactions, label=label, directory=str(directory)) + atomic_json(receipt, result) + print(f'Screen {label}: {result["status"]}', flush=True) + return result + + def isolate(self, reactions): + result = self.attempt(reactions, 'screen') + if result['status'] == 'PASS': + return [] + if result['status'] != 'FAIL': + self.inconclusive.append(result) + return [] + if len(reactions) > 1: + half = len(reactions)//2 + failures = self.isolate(reactions[:half])+self.isolate(reactions[half:]) + if not failures: + children = [] + for child in (reactions[:half], reactions[half:]): + key = hashlib.sha256(json.dumps([child, 'screen']).encode()).hexdigest()[:20] + children.append(json.loads((self.output/'attempts'/key/'result.json').read_text())) + if all(child['status'] == 'PASS' for child in children): + self.interactions.append(result) + else: + self.inconclusive.append(result) + return failures + checks = [result] + for i in range(1, self.args.confirm_runs): + checks.append(self.attempt(reactions, f'confirmation-{i}')) + control = self.attempt([], 'control-'+reactions[0]) + if all(r['status'] == 'FAIL' for r in checks) and control['status'] == 'PASS': + self.candidates[reactions[0]] = dict(reason='Repeated isolated PTools build failure; no-reaction control passed', + evidence=[r['directory'] for r in checks], control=control['directory'], + image_sha256=self.identity['image_sha256'], review_required=True) + return reactions + self.inconclusive.extend(checks+[control]) + return [] + + def run(self): + if self.attempt([], 'baseline')['status'] != 'PASS': + raise RuntimeError('Baseline PGDB failed; inspect its logs before screening reactions') + batches = [self.reactions[i:i+self.args.batch_size] + for i in range(0, len(self.reactions), self.args.batch_size)] + # Every attempt has its own home and receipt. Results are consolidated after jobs finish. + with ThreadPoolExecutor(max_workers=self.args.max_tasks) as pool: + list(pool.map(self.isolate, batches)) + atomic_json(self.output/'blacklist-candidates.json', self.candidates) + atomic_json(self.output/'summary.json', dict(reactions=len(self.reactions), + initial_batches=len(batches), candidates=len(self.candidates), + interaction_failures=self.interactions, inconclusive=self.inconclusive, + image_sha256=self.identity['image_sha256'], + note='Explicit-ID synthetic screen; passing does not certify all annotations or reaction combinations.')) + if getattr(self.args, 'publish', False): + if self.inconclusive or self.interactions: + raise RuntimeError('Screen is unresolved; compatibility list was not published. Inspect summary.json and resume.') + atomic_json(Path(self.args.refdb_dir)/'functional_categories/ptools_reaction_compatibility.json', + dict(image_sha256=self.identity['image_sha256'], + mapping_sha256=self.identity['mapping_sha256'], reactions=self.candidates, + screen_version=SCREEN_VERSION)) + print(f'Screen finished: {len(self.candidates)} blacklist candidates; ' + f'{len(self.interactions)} batch interaction failures; ' + f'{len(self.inconclusive)} inconclusive results. Review {self.output}', flush=True) + + +def positive(value): + value = int(value) + if value < 1: + raise argparse.ArgumentTypeError('must be positive') + return value + + +def main(argv=None): + p = argparse.ArgumentParser(description='Screen explicit MPDB reaction assignments against a licensed PTools SIF') + p.add_argument('-d', '--refdb_dir', required=True) + p.add_argument('-o', '--output_dir', required=True) + p.add_argument('--image', default=registered_image(), help='PTools SIF [registered build_pt image]') + p.add_argument('--database_screen', action='store_true', help=argparse.SUPPRESS) + p.add_argument('--publish', action='store_true', help='Save a completed full screen compatibility list in the MPDB') + p.add_argument('--reactions', nargs='+', help='Optional reaction IDs for a targeted screen') + p.add_argument('--batch_size', type=positive, default=100) + p.add_argument('--max_tasks', type=positive, default=1, help='Concurrent isolated containers [1]') + p.add_argument('--confirm_runs', type=positive, default=2) + p.add_argument('--timeout', type=positive, default=1800, help='Seconds per build [1800]; timeout is inconclusive') + p.add_argument('--scratch_dir', help='Local temporary storage for private PGDB builds') + args = p.parse_args(argv) + if not args.image or not Path(args.image).expanduser().is_file(): + p.error('Provide a valid --image or register one with build_pt') + if args.publish and args.reactions: + p.error('--publish requires the complete MPDB reaction set') + if args.confirm_runs < 2: + p.error('--confirm_runs must be at least 2') + try: + Screen(args).run() + except (OSError, ValueError, RuntimeError) as exc: + p.exit(1, f'screen_pt: {exc}\n') + + +if __name__ == '__main__': + main() diff --git a/metapathways/pt_sequences.py b/metapathways/pt_sequences.py new file mode 100644 index 0000000..991f62c --- /dev/null +++ b/metapathways/pt_sequences.py @@ -0,0 +1,153 @@ +"""Attach real contig sequences to compact PathoLogic input in private staging.""" +import csv +import json +import re +from pathlib import Path +from metapathways.pt_ec import normalize_ecs + + +def normalize_trna_name(name): + """Separate MP's amino-acid label from a known anticodon; retain locus IDs.""" + return re.sub( + r'(\.tRNA\d+-(?:Ala|Arg|Asn|Asp|Cys|Gln|Glu|Gly|His|Ile|Leu|Lys|' + r'Met|Phe|Pro|Ser|Thr|Trp|Tyr|Val|fMet|SeC|Sec|Pyl))([ACGTU]{3})$', + r'\1-\2', name) + + +def set_organism_taxon(inputs, taxon_id): + """Apply an explicit NCBI taxon override to private PathoLogic input.""" + if not isinstance(taxon_id, int) or isinstance(taxon_id, bool) or taxon_id < 1: + raise ValueError('NCBI taxon ID must be a positive integer') + params = Path(inputs) / 'organism-params.dat' + lines = params.read_text().splitlines() + lines = [line for line in lines if line.split(None, 1)[:1] != ['NCBI-TAXON-ID']] + lines.append(f'NCBI-TAXON-ID\t{taxon_id}') + params.write_text('\n'.join(lines) + '\n') + print(f'Pathway Tools organism taxon override: {taxon_id}', flush=True) + + +def prodigal_codon_tables(gff): + """Read per-contig genetic codes from Prodigal's original model metadata.""" + tables, contig = {}, None + if not gff.is_file(): + return tables + for line in gff.read_text().splitlines(): + if line.startswith('# Sequence Data:'): + match = re.search(r'seqhdr="([^" ]+)', line) + contig = match.group(1) if match else None + elif line.startswith('# Model Data:'): + match = re.search(r'transl_table=(\d+)(?:;|$)', line) + if contig is None or match is None: + raise ValueError(f'Missing Prodigal contig/genetic-code metadata in {gff}') + code = int(match.group(1)) + if code not in (1, 2, 3, 4, 5, 6, 9, 10, 11, 12, 13, 14, 15): + raise ValueError(f'Unsupported PathoLogic codon table {code} for {contig}') + if contig in tables and tables[contig] != code: + raise ValueError(f'Conflicting Prodigal codon tables for {contig}') + tables[contig] = code + return tables + + +def attach_sequences(inputs, sample_output): + inputs, base = Path(inputs), Path(sample_output) + sample = base.name + table = base/'results/annotation_table'/f'{sample}.ptinput.tsv' + fasta = base/'preprocessed'/f'{sample}.fasta' + for file in (table, fasta): + if not file.is_file(): + raise ValueError(f'Cannot prepare sequence-backed Pathway Tools input; missing {file}') + with table.open() as stream: + reader = csv.DictReader(stream, delimiter='\t') + if not {'orf_id', 'seqname', 'start', 'end', 'strand'} <= set(reader.fieldnames or []): + raise ValueError(f'Pathway Tools annotation table lacks feature coordinates/contig/strand: {table}') + features = {} + for row in reader: + if not row['orf_id'] or not row['seqname'] or row['orf_id'] in features: + raise ValueError('Missing or duplicate feature/contig in annotation table: ' + row['orf_id']) + start, end = int(row['start']), int(row['end']) + if start < 1 or end < start or row['strand'] not in ('+', '-'): + raise ValueError('Invalid source coordinates/strand for ' + row['orf_id']) + if row['strand'] == '-': + start, end = end, start + features[row['orf_id']] = (row['seqname'], start, end) + records, seen, corrected = {}, set(), 0 + normalized_ec_records = 0 + normalized_trna_names = {} + texts, lines = [], [] + for line in (inputs/'0.pf').read_text().splitlines(): + if line.strip() == '//': + texts.append('\n'.join(lines)) + lines = [] + else: + lines.append(line) + if any(line.strip() for line in lines): + raise ValueError('Unterminated Pathway Tools feature record') + for text in texts: + if not text.strip(): + continue + fields = dict(line.split('\t', 1) for line in text.strip().splitlines() if '\t' in line) + identifier = fields.get('ID') + if identifier in seen or identifier not in features: + raise ValueError('Duplicate or unmapped Pathway Tools feature: ' + str(identifier)) + seen.add(identifier) + contig, start, end = features[identifier] + # MAG splitter expands representative annotations to member IDs but retains + # the representative's coordinates. Restore each member's actual locus. + corrected += (fields.get('STARTBASE'), fields.get('ENDBASE')) != (str(start), str(end)) + fixed = [line for line in text.strip().splitlines() if not line.startswith(('STARTBASE\t', 'ENDBASE\t'))] + if fields.get('PRODUCT-TYPE', '').upper() == 'TRNA': + old_name = fields.get('NAME', '') + new_name = normalize_trna_name(old_name) + if new_name != old_name: + fixed = ['NAME\t' + new_name if line.startswith('NAME\t') else line for line in fixed] + normalized_trna_names[identifier] = dict(original=old_name, staged=new_name) + old_ecs = [line.split('\t', 1)[1] for line in fixed if line.startswith('EC\t')] + ecs = normalize_ecs(old_ecs) + normalized_ec_records += old_ecs != ecs + fixed = [line for line in fixed if not line.startswith('EC\t')] + fixed.extend('EC\t' + ec for ec in ecs) + fixed.extend([f'STARTBASE\t{start}', f'ENDBASE\t{end}']) + records.setdefault(contig, []).append(('\n'.join(fixed)+'\n//\n', start, end)) + if not records: + raise ValueError('Pathway Tools input contains no feature records') + sequences, name, parts = {}, None, [] + def save(): + if name in records: + if name in sequences: + raise ValueError('Duplicate contig sequence: ' + name) + sequences[name] = ''.join(parts) + with fasta.open() as stream: + for line in stream: + if line.startswith('>'): + save() + name, parts = line[1:].split()[0], [] + elif name in records: + parts.append(line.strip()) + save() + missing = set(records)-set(sequences) + if missing: + raise ValueError('Missing Pathway Tools contig sequences: ' + ', '.join(sorted(missing)[:10])) + # Validate everything before replacing the genetic-element manifest. + for contig, features in records.items(): + for _, start, end in features: + if min(start, end) < 1 or max(start, end) > len(sequences[contig]): + raise ValueError(f'Pathway Tools coordinates {start}..{end} outside {contig} (length {len(sequences[contig])})') + codon_tables = prodigal_codon_tables(base/'orf_prediction'/f'{sample}.cds.gff') + elements = [] + for number, contig in enumerate(sorted(records), 1): + identifier = f'contig_{number}' + (inputs/(identifier+'.pf')).write_text(''.join(record[0] for record in records[contig])) + seq = sequences[contig] + (inputs/(identifier+'.fasta')).write_text('>'+identifier+'\n' + '\n'.join(seq[i:i+80] for i in range(0, len(seq), 80))+'\n') + code = f'CODON-TABLE\t{codon_tables[contig]}\n' if contig in codon_tables else '' + elements.append(f'ID\t{identifier}\nNAME\t{contig}\nTYPE\t:CONTIG\n{code}ANNOT-FILE\t{identifier}.pf\nSEQ-FILE\t{identifier}.fasta\n//\n') + (inputs/'genetic-elements.dat').write_text(''.join(elements)) + (inputs/'sequence-input.json').write_text(json.dumps(dict(contigs=len(sequences), features=len(seen), + restored_coordinates=corrected, normalized_ec_records=normalized_ec_records, + normalized_trna_names=normalized_trna_names, + codon_tables={c: codon_tables[c] for c in records if c in codon_tables}, fasta=str(fasta), feature_table=str(table)), indent=2)+'\n') + if normalized_trna_names: + print(f'Separated anticodon triplets in {len(normalized_trna_names)} tRNA names; feature IDs and original inputs retained', flush=True) + if normalized_ec_records: + print(f'Normalized EC entries in {normalized_ec_records} features; original inputs retained', flush=True) + print(f'Attached {len(sequences)} contig sequences to {len(seen)} Pathway Tools features; restored {corrected} feature coordinates', flush=True) diff --git a/metapathways/pt_taxonomy.py b/metapathways/pt_taxonomy.py new file mode 100644 index 0000000..273d03e --- /dev/null +++ b/metapathways/pt_taxonomy.py @@ -0,0 +1,49 @@ +"""Readable aliases for explicit Pathway Tools organism taxon overrides.""" +import argparse + +SCOPES = {'all': 131567, 'bacteria': 2, 'archaea': 2157, 'eukaryotes': 2759} +ALIASES = {'euks': 'eukaryotes'} + + +def positive_taxon(value): + try: + taxon = int(value) + except (TypeError, ValueError): + raise argparse.ArgumentTypeError('taxon ID must be a positive integer') + if taxon < 1: + raise argparse.ArgumentTypeError('taxon ID must be a positive integer') + return taxon + + +def scope_name(value): + if value == 'prokaryotes': + raise argparse.ArgumentTypeError( + 'prokaryotes combines Bacteria and Archaea and has no supported single ' + 'NCBI taxon here; choose bacteria, archaea, or all (includes eukaryotes)') + return ALIASES.get(value, value) + + +def add_taxonomy_options(parser): + group = parser.add_mutually_exclusive_group() + group.add_argument('--taxon_id', type=positive_taxon, + help='Override PGDB NCBI taxon in private inputs; applies to every selected entity') + group.add_argument('--taxonomic_scope', type=scope_name, choices=tuple(SCOPES), + help='Named PGDB taxon override; all means cellular life; euks aliases eukaryotes. ' + 'Taxonomic pruning is enabled by default. ' + 'Default: all (cellular life).') + + +def resolve_taxon(args): + scope = getattr(args, 'taxonomic_scope', None) + taxon = getattr(args, 'taxon_id', None) + if scope and taxon is not None: + raise ValueError('--taxon_id and --taxonomic_scope are mutually exclusive') + return SCOPES[scope] if scope else (taxon if taxon is not None else SCOPES['all']) + + +def add_pruning_options(parser): + group = parser.add_mutually_exclusive_group() + group.add_argument('--taxprune', dest='taxprune', action='store_true', default=True, + help='Enable taxonomic pruning [default]') + group.add_argument('--no_taxprune', dest='taxprune', action='store_false', + help='Disable taxonomic pruning and perform unpruned rescoring') diff --git a/metapathways/regtests/cami_test/Gastrointestinal_5.selection.log b/metapathways/regtests/cami_test/Gastrointestinal_5.selection.log new file mode 100644 index 0000000..20566c6 --- /dev/null +++ b/metapathways/regtests/cami_test/Gastrointestinal_5.selection.log @@ -0,0 +1,11 @@ +[M::mm_idx_gen::0.004*1.29] collected minimizers +[M::mm_idx_gen::0.012*1.60] sorted minimizers +[M::main::0.012*1.60] loaded/built the index for 3 target sequence(s) +[M::mm_mapopt_update::0.012*1.60] mid_occ = 1000 +[M::mm_idx_stat] kmer size: 21; skip: 11; is_hpc: 0; #seq: 3 +[M::mm_idx_stat::0.012*1.57] distinct minimizers: 24598 (99.13% are singletons); average occurrences: 1.009; average spacing: 6.045; total length: 150000 +[M::worker_pipeline::0.918*2.05] mapped 333334 sequences +[M::worker_pipeline::1.077*1.98] mapped 166666 sequences +[M::main] Version: 2.31-r1302 +[M::main] CMD: /data/home/ryan/repos/metapathways-setup/environment/bin/minimap2 -ax sr -t 2 /data/tmp/mp-cami-reviewer-0sinn_1_/assembly.fasta /data/tmp/mp-cami-reviewer-0sinn_1_/r1.fq /data/tmp/mp-cami-reviewer-0sinn_1_/r2.fq +[M::main] Real time: 1.109 sec; CPU: 2.170 sec; Peak RSS: 0.198 GB diff --git a/metapathways/regtests/cami_test/README.md b/metapathways/regtests/cami_test/README.md new file mode 100644 index 0000000..88cf1ab --- /dev/null +++ b/metapathways/regtests/cami_test/README.md @@ -0,0 +1,40 @@ +# Tiny CAMI II test inputs + +Use the [test walkthrough](https://hallamlab-metapathways.readthedocs.io/en/latest/test.html) for normal MP commands. + + + +This bundle contains three different samples from the CAMI II human-associated short-read dataset ([Meyer et al., 2022](#cami-references)): Urogenital_22, Gastrointestinal_5 and Skin_28. Each has three 50,000-base assembly regions from three source genomes, intact paired simulated reads, and a headerless contig-to-genome map. Total input size is about 2.4 MiB compressed. No Pathway Tools installer, image, MetaCyc reference or licensed PGDB is included. + +- `single.tsv`: Urogenital_22. +- `pair.tsv`: Gastrointestinal_5 and Skin_28. +- `all.tsv`: all three. +- `inputs/`: strict automatic-discovery layout containing only assemblies, reads and genome maps. +- `provenance.json`: selection method, source files, original GSA mapping rows, retained coordinates, aligned-pair counts and SHA256 hashes. +- `validation.json`: input validation and predicted gene counts; the complete MP workflow has not been executed as part of preparing this bundle. +- `*.selection.log`: minimap2 selection diagnostics. + +Manifest paths are relative to the manifest file. Copy the whole bundle when moving it. Original contig and read identifiers are preserved; contigs are cropped to their first 50,000 bases. Source genome IDs are prefixed with `CAMI_` and punctuation becomes underscores for MP-compatible entity names. Coordinates in `provenance.json` explicitly distinguish original source mapping intervals from retained contig-relative intervals. + +For each sample, the preparation script ranks contigs at least 20 kb long by source read density, chooses one contig from each of three different genomes, and retains at most 50 kb per contig. It scans the first 250,000 original interleaved read pairs, aligns them to the retained regions with minimap2, and keeps at most 3,000 complete pairs with at least one primary alignment at MAPQ 20 or higher. Both original mates and qualities are retained even if only one mate maps. Every retained contig must have at least ten selected pairs aligning to it. The cap and selection deliberately bias coverage; this is interface-test data, not an accuracy or abundance benchmark. Three short genome fragments are not three complete MAGs. + +Maintainers can reproduce the data from an MP manifest pointing at the original local CAMI downloads: + +```bash +python scripts/prepare_cami_test.py \ + --source-manifest /path/to/full-cami-manifest.tsv \ + --output /path/to/new-cami-test \ + --minimap2 /path/to/minimap2 +``` + +Run from the source checkout. The output directory must not already exist. Source assemblies must have a sibling `gsa_mapping.tsv`, and the selected sample read inputs must be the original interleaved FASTQs. Python and minimap2 are required. Temporary read subsets and alignments are cleaned up after preparation. Reproduction requires the original large CAMI files; tests use the bundled small files directly. + +## Source attribution + +Cropping, read selection, mate separation and MP-format maps are the modifications made here. The data retain their source terms; the MP software license does not relicense third-party data. The source collection is CAMI II; the original CAMI paper credits the initiative ([Sczyrba et al., 2017](#cami-references)). + +## CAMI references + +- **CAMI:** Sczyrba, A., Hofmann, P., Belmann, P., et al. (2017). *Critical Assessment of Metagenome Interpretation—a benchmark of metagenomics software*. **Nature Methods 14**(11), 1063–1071. [DOI: 10.1038/nmeth.4458](https://doi.org/10.1038/nmeth.4458). [CAMI project website](https://cami-challenge.org/). +- **CAMI II:** Meyer, F., Fritz, A., Deng, Z.-L., et al. (2022). *Critical Assessment of Metagenome Interpretation: the second round of challenges*. **Nature Methods 19**(4), 429–440. [DOI: 10.1038/s41592-022-01431-4](https://doi.org/10.1038/s41592-022-01431-4). [CAMI project website](https://cami-challenge.org/). +- **Source dataset for the MP test subset:** CAMI II multi-sample human microbiome dataset. [Dataset DOI: 10.4126/FRL01-006425518](https://doi.org/10.4126/FRL01-006425518). The bundled inputs are selected and cropped subsets of this collection; their exact transformations and file hashes are recorded in the bundle provenance. diff --git a/metapathways/regtests/cami_test/Skin_28.selection.log b/metapathways/regtests/cami_test/Skin_28.selection.log new file mode 100644 index 0000000..466a20b --- /dev/null +++ b/metapathways/regtests/cami_test/Skin_28.selection.log @@ -0,0 +1,11 @@ +[M::mm_idx_gen::0.006*1.23] collected minimizers +[M::mm_idx_gen::0.010*1.46] sorted minimizers +[M::main::0.010*1.46] loaded/built the index for 3 target sequence(s) +[M::mm_mapopt_update::0.010*1.46] mid_occ = 1000 +[M::mm_idx_stat] kmer size: 21; skip: 11; is_hpc: 0; #seq: 3 +[M::mm_idx_stat::0.011*1.43] distinct minimizers: 20161 (75.80% are singletons); average occurrences: 1.243; average spacing: 5.987; total length: 150000 +[M::worker_pipeline::0.985*1.90] mapped 333334 sequences +[M::worker_pipeline::1.201*1.87] mapped 166666 sequences +[M::main] Version: 2.31-r1302 +[M::main] CMD: /data/home/ryan/repos/metapathways-setup/environment/bin/minimap2 -ax sr -t 2 /data/tmp/mp-cami-reviewer-0sinn_1_/assembly.fasta /data/tmp/mp-cami-reviewer-0sinn_1_/r1.fq /data/tmp/mp-cami-reviewer-0sinn_1_/r2.fq +[M::main] Real time: 1.221 sec; CPU: 2.263 sec; Peak RSS: 0.199 GB diff --git a/metapathways/regtests/cami_test/Urogenital_22.selection.log b/metapathways/regtests/cami_test/Urogenital_22.selection.log new file mode 100644 index 0000000..1168cc5 --- /dev/null +++ b/metapathways/regtests/cami_test/Urogenital_22.selection.log @@ -0,0 +1,11 @@ +[M::mm_idx_gen::0.004*1.25] collected minimizers +[M::mm_idx_gen::0.012*1.59] sorted minimizers +[M::main::0.012*1.60] loaded/built the index for 3 target sequence(s) +[M::mm_mapopt_update::0.012*1.60] mid_occ = 1000 +[M::mm_idx_stat] kmer size: 21; skip: 11; is_hpc: 0; #seq: 3 +[M::mm_idx_stat::0.012*1.57] distinct minimizers: 25092 (99.95% are singletons); average occurrences: 1.001; average spacing: 5.975; total length: 150000 +[M::worker_pipeline::0.917*2.03] mapped 333334 sequences +[M::worker_pipeline::1.119*1.97] mapped 166666 sequences +[M::main] Version: 2.31-r1302 +[M::main] CMD: /data/home/ryan/repos/metapathways-setup/environment/bin/minimap2 -ax sr -t 2 /data/tmp/mp-cami-reviewer-0sinn_1_/assembly.fasta /data/tmp/mp-cami-reviewer-0sinn_1_/r1.fq /data/tmp/mp-cami-reviewer-0sinn_1_/r2.fq +[M::main] Real time: 1.132 sec; CPU: 2.218 sec; Peak RSS: 0.198 GB diff --git a/metapathways/regtests/cami_test/all.tsv b/metapathways/regtests/cami_test/all.tsv new file mode 100644 index 0000000..9b11279 --- /dev/null +++ b/metapathways/regtests/cami_test/all.tsv @@ -0,0 +1,4 @@ +sample_id assembly read_layout reads_1 reads_2 mag_map +Urogenital_22 inputs/assemblies/Urogenital_22.fasta.gz paired inputs/reads/Urogenital_22_R1.fastq.gz inputs/reads/Urogenital_22_R2.fastq.gz inputs/mag_maps/Urogenital_22.tsv +Gastrointestinal_5 inputs/assemblies/Gastrointestinal_5.fasta.gz paired inputs/reads/Gastrointestinal_5_R1.fastq.gz inputs/reads/Gastrointestinal_5_R2.fastq.gz inputs/mag_maps/Gastrointestinal_5.tsv +Skin_28 inputs/assemblies/Skin_28.fasta.gz paired inputs/reads/Skin_28_R1.fastq.gz inputs/reads/Skin_28_R2.fastq.gz inputs/mag_maps/Skin_28.tsv diff --git a/metapathways/regtests/cami_test/inputs/assemblies/Gastrointestinal_5.fasta.gz b/metapathways/regtests/cami_test/inputs/assemblies/Gastrointestinal_5.fasta.gz new file mode 100644 index 0000000..c5ec223 Binary files /dev/null and b/metapathways/regtests/cami_test/inputs/assemblies/Gastrointestinal_5.fasta.gz differ diff --git a/metapathways/regtests/cami_test/inputs/assemblies/Skin_28.fasta.gz b/metapathways/regtests/cami_test/inputs/assemblies/Skin_28.fasta.gz new file mode 100644 index 0000000..5a56ffb Binary files /dev/null and b/metapathways/regtests/cami_test/inputs/assemblies/Skin_28.fasta.gz differ diff --git a/metapathways/regtests/cami_test/inputs/assemblies/Urogenital_22.fasta.gz b/metapathways/regtests/cami_test/inputs/assemblies/Urogenital_22.fasta.gz new file mode 100644 index 0000000..50daa5c Binary files /dev/null and b/metapathways/regtests/cami_test/inputs/assemblies/Urogenital_22.fasta.gz differ diff --git a/metapathways/regtests/cami_test/inputs/mag_maps/Gastrointestinal_5.tsv b/metapathways/regtests/cami_test/inputs/mag_maps/Gastrointestinal_5.tsv new file mode 100644 index 0000000..90746cc --- /dev/null +++ b/metapathways/regtests/cami_test/inputs/mag_maps/Gastrointestinal_5.tsv @@ -0,0 +1,3 @@ +S5C2519 CAMI_OTU_97_1154_0 +S5C5928 CAMI_OTU_97_68_0 +S5C5950 CAMI_OTU_97_44483_0 diff --git a/metapathways/regtests/cami_test/inputs/mag_maps/Skin_28.tsv b/metapathways/regtests/cami_test/inputs/mag_maps/Skin_28.tsv new file mode 100644 index 0000000..9c0f81b --- /dev/null +++ b/metapathways/regtests/cami_test/inputs/mag_maps/Skin_28.tsv @@ -0,0 +1,3 @@ +S28C1233 CAMI_OTU_97_44585_0 +S28C1662 CAMI_OTU_97_34494_0 +S28C1722 CAMI_OTU_97_37297_0 diff --git a/metapathways/regtests/cami_test/inputs/mag_maps/Urogenital_22.tsv b/metapathways/regtests/cami_test/inputs/mag_maps/Urogenital_22.tsv new file mode 100644 index 0000000..12928f1 --- /dev/null +++ b/metapathways/regtests/cami_test/inputs/mag_maps/Urogenital_22.tsv @@ -0,0 +1,3 @@ +S22C5358 CAMI_OTU_97_19349_0 +S22C8829 CAMI_OTU_97_578_0 +S22C9504 CAMI_OTU_97_39620_0 diff --git a/metapathways/regtests/cami_test/inputs/reads/Gastrointestinal_5_R1.fastq.gz b/metapathways/regtests/cami_test/inputs/reads/Gastrointestinal_5_R1.fastq.gz new file mode 100644 index 0000000..256a9cd Binary files /dev/null and b/metapathways/regtests/cami_test/inputs/reads/Gastrointestinal_5_R1.fastq.gz differ diff --git a/metapathways/regtests/cami_test/inputs/reads/Gastrointestinal_5_R2.fastq.gz b/metapathways/regtests/cami_test/inputs/reads/Gastrointestinal_5_R2.fastq.gz new file mode 100644 index 0000000..6534bc6 Binary files /dev/null and b/metapathways/regtests/cami_test/inputs/reads/Gastrointestinal_5_R2.fastq.gz differ diff --git a/metapathways/regtests/cami_test/inputs/reads/Skin_28_R1.fastq.gz b/metapathways/regtests/cami_test/inputs/reads/Skin_28_R1.fastq.gz new file mode 100644 index 0000000..36a130b Binary files /dev/null and b/metapathways/regtests/cami_test/inputs/reads/Skin_28_R1.fastq.gz differ diff --git a/metapathways/regtests/cami_test/inputs/reads/Skin_28_R2.fastq.gz b/metapathways/regtests/cami_test/inputs/reads/Skin_28_R2.fastq.gz new file mode 100644 index 0000000..e3c5b3d Binary files /dev/null and b/metapathways/regtests/cami_test/inputs/reads/Skin_28_R2.fastq.gz differ diff --git a/metapathways/regtests/cami_test/inputs/reads/Urogenital_22_R1.fastq.gz b/metapathways/regtests/cami_test/inputs/reads/Urogenital_22_R1.fastq.gz new file mode 100644 index 0000000..b511ff1 Binary files /dev/null and b/metapathways/regtests/cami_test/inputs/reads/Urogenital_22_R1.fastq.gz differ diff --git a/metapathways/regtests/cami_test/inputs/reads/Urogenital_22_R2.fastq.gz b/metapathways/regtests/cami_test/inputs/reads/Urogenital_22_R2.fastq.gz new file mode 100644 index 0000000..92c596f Binary files /dev/null and b/metapathways/regtests/cami_test/inputs/reads/Urogenital_22_R2.fastq.gz differ diff --git a/metapathways/regtests/cami_test/pair.tsv b/metapathways/regtests/cami_test/pair.tsv new file mode 100644 index 0000000..dde44a5 --- /dev/null +++ b/metapathways/regtests/cami_test/pair.tsv @@ -0,0 +1,3 @@ +sample_id assembly read_layout reads_1 reads_2 mag_map +Gastrointestinal_5 inputs/assemblies/Gastrointestinal_5.fasta.gz paired inputs/reads/Gastrointestinal_5_R1.fastq.gz inputs/reads/Gastrointestinal_5_R2.fastq.gz inputs/mag_maps/Gastrointestinal_5.tsv +Skin_28 inputs/assemblies/Skin_28.fasta.gz paired inputs/reads/Skin_28_R1.fastq.gz inputs/reads/Skin_28_R2.fastq.gz inputs/mag_maps/Skin_28.tsv diff --git a/metapathways/regtests/cami_test/provenance.json b/metapathways/regtests/cami_test/provenance.json new file mode 100644 index 0000000..91aae59 --- /dev/null +++ b/metapathways/regtests/cami_test/provenance.json @@ -0,0 +1,170 @@ +{ + "source_dataset": "CAMI II human-associated short-read gold-standard assemblies and simulated reads", + "source_doi": "10.4126/FRL01-006425518", + "selection": "Prepare tiny real CAMI inputs; requires minimap2, not MetaPathways or Pathway Tools.\n\nSelect three abundant genomes per sample, retain up to 50 kb of one contig each,\nthen retain intact original read pairs with a primary MAPQ >=20 alignment to\nthose regions from the first 250,000 pairs. This is deliberately coverage-biased\ninterface test data, never an abundance or accuracy benchmark.\n", + "minimap2_version": "2.31-r1302", + "samples": [ + { + "sample_id": "Urogenital_22", + "source_assembly": "/data/home/ryan/data/mock_2022/MetaGs/CAMI_II_Urogenital/short_read/2017.12.04_18.56.22_sample_22/contigs/anonymous_gsa.fasta", + "source_reads": "/data/home/ryan/data/mock_2022/MetaGs/CAMI_II_Urogenital/short_read/2017.12.04_18.56.22_sample_22/reads/anonymous_reads.fq.gz", + "source_mapping": "/data/home/ryan/data/mock_2022/MetaGs/CAMI_II_Urogenital/short_read/2017.12.04_18.56.22_sample_22/contigs/gsa_mapping.tsv", + "scanned_pairs": 250000, + "retained_pairs": 3000, + "assembly_bases": 150000, + "regions": [ + { + "#anonymous_contig_id": "S22C5358", + "genome_id": "OTU_97.19349.0", + "tax_id": "1578", + "contig_id": "CP002341.1", + "number_reads": "2550346", + "start_position": "2", + "end_position": "2125751", + "retained_contig_start_1based": 1, + "retained_contig_end_1based": 50000, + "aligned_pairs": 1460 + }, + { + "#anonymous_contig_id": "S22C8829", + "genome_id": "OTU_97.578.0", + "tax_id": "1763", + "contig_id": "CP018363.1", + "number_reads": "5712090", + "start_position": "1", + "end_position": "5626623", + "retained_contig_start_1based": 1, + "retained_contig_end_1based": 50000, + "aligned_pairs": 299 + }, + { + "#anonymous_contig_id": "S22C9504", + "genome_id": "OTU_97.39620.0", + "tax_id": "1578", + "contig_id": "CP011403.1", + "number_reads": "10184206", + "start_position": "1", + "end_position": "1751564", + "retained_contig_start_1based": 1, + "retained_contig_end_1based": 50000, + "aligned_pairs": 1241 + } + ] + }, + { + "sample_id": "Gastrointestinal_5", + "source_assembly": "/data/home/ryan/data/mock_2022/MetaGs/CAMI_II_Gastrointestinal/short_read/2017.12.04_18.45.54_sample_5/contigs/anonymous_gsa.fasta", + "source_reads": "/data/home/ryan/data/mock_2022/MetaGs/CAMI_II_Gastrointestinal/short_read/2017.12.04_18.45.54_sample_5/reads/anonymous_reads.fq.gz", + "source_mapping": "/data/home/ryan/data/mock_2022/MetaGs/CAMI_II_Gastrointestinal/short_read/2017.12.04_18.45.54_sample_5/contigs/gsa_mapping.tsv", + "scanned_pairs": 250000, + "retained_pairs": 2326, + "assembly_bases": 150000, + "regions": [ + { + "#anonymous_contig_id": "S5C2519", + "genome_id": "OTU_97.1154.0", + "tax_id": "506", + "contig_id": "HE965803.1", + "number_reads": "5485844", + "start_position": "2", + "end_position": "4887378", + "retained_contig_start_1based": 1, + "retained_contig_end_1based": 50000, + "aligned_pairs": 951 + }, + { + "#anonymous_contig_id": "S5C5928", + "genome_id": "OTU_97.68.0", + "tax_id": "841", + "contig_id": "CP003040.1", + "number_reads": "3359972", + "start_position": "6", + "end_position": "3592124", + "retained_contig_start_1based": 1, + "retained_contig_end_1based": 50000, + "aligned_pairs": 731 + }, + { + "#anonymous_contig_id": "S5C5950", + "genome_id": "OTU_97.44483.0", + "tax_id": "541000", + "contig_id": "CP004044.1", + "number_reads": "1411892", + "start_position": "2", + "end_position": "905459", + "retained_contig_start_1based": 1, + "retained_contig_end_1based": 50000, + "aligned_pairs": 644 + } + ] + }, + { + "sample_id": "Skin_28", + "source_assembly": "/data/home/ryan/data/mock_2022/MetaGs/CAMI_II_Skin/short_read/2017.12.04_18.56.22_sample_28/contigs/anonymous_gsa.fasta", + "source_reads": "/data/home/ryan/data/mock_2022/MetaGs/CAMI_II_Skin/short_read/2017.12.04_18.56.22_sample_28/reads/anonymous_reads.fq.gz", + "source_mapping": "/data/home/ryan/data/mock_2022/MetaGs/CAMI_II_Skin/short_read/2017.12.04_18.56.22_sample_28/contigs/gsa_mapping.tsv", + "scanned_pairs": 250000, + "retained_pairs": 3000, + "assembly_bases": 150000, + "regions": [ + { + "#anonymous_contig_id": "S28C1233", + "genome_id": "OTU_97.44585.0", + "tax_id": "1279", + "contig_id": "BA000018.3", + "number_reads": "5213080", + "start_position": "1", + "end_position": "2814815", + "retained_contig_start_1based": 1, + "retained_contig_end_1based": 50000, + "aligned_pairs": 764 + }, + { + "#anonymous_contig_id": "S28C1662", + "genome_id": "OTU_97.34494.0", + "tax_id": "1279", + "contig_id": "AP009351.1", + "number_reads": "2665852", + "start_position": "2", + "end_position": "2878897", + "retained_contig_start_1based": 1, + "retained_contig_end_1based": 50000, + "aligned_pairs": 565 + }, + { + "#anonymous_contig_id": "S28C1722", + "genome_id": "OTU_97.37297.0", + "tax_id": "1279", + "contig_id": "CP012978.1", + "number_reads": "446618", + "start_position": "1722189", + "end_position": "2678868", + "retained_contig_start_1based": 1, + "retained_contig_end_1based": 50000, + "aligned_pairs": 1673 + } + ] + } + ], + "sha256": { + "Gastrointestinal_5.selection.log": "733392ec9acbe4badfb2b9b7fe2170091a6aa0ef99ce796a8dc2308b8af5e619", + "Skin_28.selection.log": "3cc86f1d32e15923615670bf5393003e36e668b61a0398e44992182f95e01e93", + "Urogenital_22.selection.log": "84d33e12363026ba9ea8568ab3ee204ab735b8020d057e5c56a8ace101ca174b", + "all.tsv": "fa581491d87c5e2c9bd08edc6965e1c7febc064c14555c9b4df79661c4ca5f73", + "inputs/assemblies/Gastrointestinal_5.fasta.gz": "dc0fad0125cfc3670ed488a63af71b76289aae92089c29843a5263b96d714e80", + "inputs/assemblies/Skin_28.fasta.gz": "978efa7285eeda349ce1ad4e1b5155a635b4d9de89f2565105212a35f9a7d8fb", + "inputs/assemblies/Urogenital_22.fasta.gz": "c62ccf9a276bd1252cc5c4e3bbccb922b502653f53d204a8ea865c2b633fa3c9", + "inputs/mag_maps/Gastrointestinal_5.tsv": "5f41303fab5804bee022edfaa56b06ba6d0b51c4148b41f985b943ef2cbbb72a", + "inputs/mag_maps/Skin_28.tsv": "5eb47bf6ab2981b0001e5a4653cf60669c32c461f25964ac4837b0223901cdde", + "inputs/mag_maps/Urogenital_22.tsv": "65c77d5a48b13448b8c71d51aa63b6f6260da9ff5cb69385c3eaa9cd182977fe", + "inputs/reads/Gastrointestinal_5_R1.fastq.gz": "adafa12f02a0d89da29340df2797d6f23c7ff0e02f78ab55b5fb952a06f92ae6", + "inputs/reads/Gastrointestinal_5_R2.fastq.gz": "1a7ccdcb55985d473523ea3c8cf3462c8054d9ceb2d85a76feac8cf1d9c67b00", + "inputs/reads/Skin_28_R1.fastq.gz": "edf96bd449b83ca39e09b45876d5db51a7d93fb37848ec21cb2e5696c16b045f", + "inputs/reads/Skin_28_R2.fastq.gz": "46a1340c46ec1de1e18cdcb8ca355d1f6eff4ee481bf0d47b1bbe503b018df91", + "inputs/reads/Urogenital_22_R1.fastq.gz": "963a2ef3bb53aa8e38a171f4550b88677c8a1650e0b7efc3e964e7cca161b827", + "inputs/reads/Urogenital_22_R2.fastq.gz": "4a4d8dad32aff87ae14d3faccf43c47cb40ba323bae697660f0b8f9a6f9ef7dc", + "pair.tsv": "e788c33a7887f68ffd0dcfdab843200bbe87f783082ff0c0854a976683add49a", + "single.tsv": "e883b3f23e280841da0ce02be61287611c9e34e2d6b81ffc168611dcd8621f6d" + }, + "preparation_script_sha256": "874a64ae1c53978e9eb087108ae1dd25ac1b02fc9add137d95962f85eee0ff16" +} diff --git a/metapathways/regtests/cami_test/single.tsv b/metapathways/regtests/cami_test/single.tsv new file mode 100644 index 0000000..290954f --- /dev/null +++ b/metapathways/regtests/cami_test/single.tsv @@ -0,0 +1,2 @@ +sample_id assembly read_layout reads_1 reads_2 mag_map +Urogenital_22 inputs/assemblies/Urogenital_22.fasta.gz paired inputs/reads/Urogenital_22_R1.fastq.gz inputs/reads/Urogenital_22_R2.fastq.gz inputs/mag_maps/Urogenital_22.tsv diff --git a/metapathways/regtests/cami_test/validation.json b/metapathways/regtests/cami_test/validation.json new file mode 100644 index 0000000..a1b5437 --- /dev/null +++ b/metapathways/regtests/cami_test/validation.json @@ -0,0 +1,51 @@ +{ + "checks": [ + "payload SHA256", + "single/pair/all manifest validation", + "automatic input discovery", + "paired FASTQ IDs/lengths", + "complete contig-to-genome assignment", + "Prodigal meta gene predictions on every contig" + ], + "workflow_executed": false, + "samples": [ + { + "sample_id": "Gastrointestinal_5", + "assembly_bases": 150000, + "pairs": 2326, + "predicted_genes_per_contig": { + "S5C2519": 59, + "S5C5928": 42, + "S5C5950": 41 + } + }, + { + "sample_id": "Skin_28", + "assembly_bases": 150000, + "pairs": 3000, + "predicted_genes_per_contig": { + "S28C1233": 42, + "S28C1662": 40, + "S28C1722": 40 + } + }, + { + "sample_id": "Urogenital_22", + "assembly_bases": 150000, + "pairs": 3000, + "predicted_genes_per_contig": { + "S22C5358": 50, + "S22C8829": 45, + "S22C9504": 47 + } + } + ], + "small_reference_check": { + "method": "Prodigal meta-mode proteins searched with BLASTP against bundled swissprot_test at E <= 1e-6; not full MP annotation/BSR filtering", + "ORFs_with_hits": { + "Gastrointestinal_5": 3, + "Skin_28": 3, + "Urogenital_22": 4 + } + } +} diff --git a/metapathways/regtests/test_db/functional/swissprot_test b/metapathways/regtests/test_db/functional/swissprot_test index dfc9b9d..4b52f35 100644 --- a/metapathways/regtests/test_db/functional/swissprot_test +++ b/metapathways/regtests/test_db/functional/swissprot_test @@ -273,3 +273,1339 @@ WDLEEEAENLIQEQSEDDQGWVWLP MAKSALFTVRNNESCPKCGAELVIRSGKHGPFLGCSQYPACDYVRPLKSSADGHIVKVLE GQVCPACGANLVLRQGRFGMFIGCINYPECEHTELIDKPDETAITCPQCRTGHLVQRRSR YGKTFHSCDRYPECQFAINFKPIAGECPECHYPLLIEKKTAQGVKHFCASKQCGKPVSAE +>sp|A0A086F3E3|TM175_CHRP1 Potassium channel HX13_20290 OS=Chryseobacterium sp. (strain P1-3) OX=1517683 GN=HX13_20290 PE=1 SV=1 +MTKGRLEAFSDGVLAIIITIMVLELKVPEGSSWASLQPILPRFLAYIFSFIYVGIYWNNH +HHLFQTVKKVNGSILWANLHLLFWLSLMPIATEWIGTSHFAQNPVATYGIGLIMSAIAYT +ILENVIIRCEGENSKLKEAIHSKFKEYISIIFYVLGIATSFFYPYIAIGFYYLVALIWLI +PDKRIEKSLKEN +>sp|A0A0H2ZQT7|WALJ_STRP2 Exodeoxyribonuclease WalJ OS=Streptococcus pneumoniae serotype 2 (strain D39 / NCTC 7466) OX=373153 GN=walJ PE=1 SV=1 +MSEIGFKYSILASGSSGNSFYLETSKKKLLVDAGLSGKKITSLLAEINRKPEDLDAILIT +HEHSDHIHGVGVLARKYGMDLYANEKTWQAMENSKYLGKVDSSQKHIFEMGKTKTFGDID +IESFGVSHDAVAPQFYRFMKDDKSFVLLTDTGYVSDRMAGIVENADGYLIEANHDVEILR +SGSYAWRLKQRILSDLGHLSNEDGAEAMIRTLGNRTKKIYLGHLSKENNIKELAHMTMVN +QLAQADLGVGVDFKVYDTSPDTATPLTEI +>sp|A0A0H3GCG4|PDEA_LISM4 Cyclic-di-AMP phosphodiesterase PdeA OS=Listeria monocytogenes serotype 1/2a (strain 10403S) OX=393133 GN=pdeA PE=3 SV=1 +MSGYFQKRMLKYPLYGLIAATIILSVITFFFSWWLSALVVVGGIILTVAMFYFEYRLNED +VQLYVSNLTYRIKRSEEEALVEMPMGILLYDEHYKIEWVNPFMSKYFDKAELIGESLEEV +GPEFLDVITGNDEKGIMSIAWRDHRFDTIVKRKERILYLYDRTEYYDLNKKFQANKSVFA +VIFLDNYDEWAQGMDDRRRSALNNLVTSMLTNWAREHRIYLKRISTDRFMAFLTEEMLKR +LEEEKFQILDRIRERTSKQNIPLTLSIGIGYKEDDLIQLADLAQSSLDLALGRGGDQVVI +KQPEGKVRFYGGKTNPMEKRTRVRARVISQALQELITQSDQVFVMGHRYPDMDVIGSSLG +VMRIAEMNDRNAYVVVEPGKMSPDVKRLMNEIEEYPNVIKNIVTPQVALENITEKSLLVV +VDTHKPSMVINKELLDSATNVVVVDHHRRSEEFVGSPVLVYIEPYASSTAELITELFEYQ +PDLEQVGKIEATALLSGIVVDTKNFTLRTGSRTFDAASYLRSLGADTILVQQFLKEDITT +FTQRSRLVESLEIYHDGMAIATGHEDEEFGTVIAAQAADTMLSMEGVQASFVITLRPDKL +IGISARSLGQINVQVIMEKLGGGGHLSNAATQLKDVTIAEAEKQLISAIDAYWKGET +>sp|A0A4Y1WBN6|YYCJ_BACAN Exodeoxyribonuclease YycJ OS=Bacillus anthracis OX=1392 GN=yycJ PE=1 SV=1 +MGLHFSVLASGSTGNMLYVGTDEKKLLVDAGLSGKATEALFKQAELNINDVSGILVTHEH +SDHIKGLGVLARKYDLPVYANEKTWNAMEHLIGNIPTDQKFIFSVGDVKTFGDIEVESFG +VSHDAAEPMFYAFHNNNRKLALITDTGYVSDRMKGVIKGANAFVFESNHDVEMLRMGRYP +WSIKRRILSDVGHVCNEDAALAMADVITDETKHIYLAHLSLDNNMKELARMSVSQVLEEK +GFGVGEAFEIHDTDPKMPTKIQYV +>sp|A0PXQ3|PANB_CLONN 3-methyl-2-oxobutanoate hydroxymethyltransferase OS=Clostridium novyi (strain NT) OX=386415 GN=panB PE=3 SV=1 +MKNTVTTFQKAKNNGEKLTMLTAYDYSTAKLIDESGINGILVGDSLGMVCLGYEDTLSVT +MEDMIHHTRAVSRGVKNTLVVGDMPFMSYQSSVYDAVVNAGRLIKEGGATAVKLEGGATV +IEQIKAIVNAQIPVMAHIGLTPQSINVFGGFKVQGKDEEKAQKLIEDAKKIEEAGAFAIV +LECVPAKLAELITKAVSIPTIGIGAGAGCDGQILVYQDMLGMFSDMSPKFVKKFADVGEL +MKDGFKAYIKEVQEGTFPSKEHCFKIDESVLDKLY +>sp|A0Q8S8|CRGA_MYCA1 Cell division protein CrgA OS=Mycobacterium avium (strain 104) OX=243243 GN=crgA PE=3 SV=1 +MPKSKVRKKNDFTVSAVSRTPVKVKVGPSSVWFVALFIGLMLIGLVWLMVFQLAAVGSQA +PTALNWMAQLGPWNYAIAFAFMITGLLLTMRWH +>sp|A3DHZ4|DNAA_ACET2 Chromosomal replication initiator protein DnaA OS=Acetivibrio thermocellus (strain ATCC 27405 / DSM 1237 / JCM 9322 / NBRC 103400 / NCIMB 10682 / NRRL B-4536 / VPI 7372) OX=203119 GN=dnaA PE=3 SV=1 +MNTQLNEIWQKTLGLLKNELTEISFNTWIKTIDPLSLTGNTINLAVPAEFNKGILESRYQ +TLIKNAIKQVTFKEYEIAFIVPSQENLNKLTKQTESAGNEDSPLSVLNPKYTFDTFVIGN +SNRFAHAAALAVAEAPGKAYNPLFIYGGVGLGKTHLMHAIGHYILEQNSSQKVLYVSSEK +FTNELINAIKDNRNEEFRSKYRNIDVLLIDDIQFIAGKERTEEEFFHTFNALYEANKQII +LSSDKPPKEISLEDRLRSRFEWGLIADMQAPDLETRIAILRKKAQLENLTVPNEVIVFIA +DKIASNIRELEGALNRVIAYSSLTENEITVELASEALKDILSANKAKVLNCTTIQEAVAR +YFDIRPEEFKSKKRTRDIAFPRQIAMYLCRELTEMSLPKIGEEFGGRDHTTVIHACEKIS +EEIESNSETRRAVSEIKRNLLGK +>sp|A5INP9|HUTH_STAA9 Histidine ammonia-lyase OS=Staphylococcus aureus (strain JH9) OX=359786 GN=hutH PE=3 SV=1 +MTLYLDGETLTIEDIKSFLQQQSKIEIIDDALERVKKSRAVVERIIENEETVYGITTGFG +LFSDVRIDPTQYNELQVNLIRSHACGLGEPFSKEVALVMMILRLNTLLKGHSGATLELVR +QLQFFINERIIPIIPQQGSLGASGDLAPLSHLALALIGEGKVLYRGEEKDSDDVLRELNR +QPLNLQAKEGLALINGTQAMTAQGVISYIEAEDLGYQSEWIAALTHQSLNGIIDAYRHDV +HSVRNFQEQINVAARMRDWLEGSTLTTRQAEIRVQDAYTLRCIPQIHGASFQVFNYVKQQ +LEFEMNAANDNPLIFEEANETFVISGGNFHGQPIAFALDHLKLGVSELANVSERRLERLV +NPQLNGDLPAFLSPEPGLQSGAMIMQYAAASLVSENKTLAHPASVDSITSSANQEDHVSM +GTTAARHGYQIIENARRVLAIECVIALQAAELKGVEGLSPKTRRKYEEFRSIVPSITHDR +QFHKDIEAVAQYLKQSIYQTTACH +>sp|A5TY85|PKNA_MYCTA Serine/threonine-protein kinase PknA OS=Mycobacterium tuberculosis (strain ATCC 25177 / H37Ra) OX=419947 GN=pknA PE=1 SV=1 +MSPRVGVTLSGRYRLQRLIATGGMGQVWEAVDNRLGRRVAVKVLKSEFSSDPEFIERFRA +EARTTAMLNHPGIASVHDYGESQMNGEGRTAYLVMELVNGEPLNSVLKRTGRLSLRHALD +MLEQTGRALQIAHAAGLVHRDVKPGNILITPTGQVKITDFGIAKAVDAAPVTQTGMVMGT +AQYIAPEQALGHDASPASDVYSLGVVGYEAVSGKRPFAGDGALTVAMKHIKEPPPPLPPD +LPPNVRELIEITLVKNPAMRYRSGGPFADAVAAVRAGRRPPRPSQTPPPGRAAPAAIPSG +TTARVAANSAGRTAASRRSRPATGGHRPPRRTFSSGQRALLWAAGVLGALAIIIAVLLVI +KAPGDNSPQQAPTPTVTTTGNPPASNTGGTDASPRLNWTERGETRHSGLQSWVVPPTPHS +RASLARYEIAQ +>sp|A6QD58|WALK_STAAE Sensor protein kinase WalK OS=Staphylococcus aureus (strain Newman) OX=426430 GN=walK PE=3 SV=1 +MKWLKQLQSLHTKLVIVYVLLIIIGMQIIGLYFTNNLEKELLDNFKKNITQYAKQLEISI +EKVYDEKGSVNAQKDIQNLLSEYANRQEIGEIRFIDKDQIIIATTKQSNRSLINQKANDS +SVQKALSLGQSNDHLILKDYGGGKDRVWVYNIPVKVDKKVIGNIYIESKINDVYNQLNNI +NQIFIVGTAISLLITVILGFFIARTITKPITDMRNQTVEMSRGNYTQRVKIYGNDEIGEL +ALAFNNLSKRVQEAQANTESEKRRLDSVITHMSDGIIATDRRGRIRIVNDMALKMFGMAK +EDIIGYYMLSVLSLEDEFKLEEIQENNDSFLLDLNEEEGLIARVNFSTIVQETGFVTGYI +AVLHDVTEQQQVERERREFVANVSHELRTPLTSMNSYIEALEEGAWKDEELAPQFLSVTR +EETERMIRLVNDLLQLSKMDNESDQINKEIIDFNMFINKIINRHEMSAKDTTFIRDIPKK +TIFTEFDPDKMTQVFDNVITNAMKYSRGDKRVEFHVKQNPLYNRMTIRIKDNGIGIPINK +VDKIFDRFYRVDKARTRKMGGTGLGLAISKEIVEAHNGRIWANSVEGQGTSIFITLPCEV +IEDGDWDE +>sp|A6U2M4|SYL_STAA2 Leucine--tRNA ligase OS=Staphylococcus aureus (strain JH1) OX=359787 GN=leuS PE=3 SV=1 +MLNYNHNQIEKKWQDYWDENKTFKTNDNLGQKKFYALDMFPYPSGAGLHVGHPEGYTATD +IISRYKRMQGYNVLHPMGWDAFGLPAEQYALDTGNDPREFTKKNIQTFKRQIKELGFSYD +WDREVNTTDPEYYKWTQWIFIQLYNKGLAYVDEVAVNWCPALGTVLSNEEVIDGVSERGG +HPVYRKPMKQWVLKITEYADQLLADLDDLDWPESLKDMQRNWIGRSEGAKVSFDVDNTEG +KVEVFTTRPDTIYGASFLVLSPEHALVNSITTDEYKEKVKAYQTEASKKSDLERTDLAKD +KSGVFTGAYAINPLSGEKVQIWIADYVLSTYGTGAIMAVPAHDDRDYEFAKKFDLPIIEV +IEGGNVEEAAYTGEGKHINSGELDGLENEAAITKAIQLLEQKGAGEKKVNYKLRDWLFSR +QRYWGEPIPVIHWEDGTMTTVPEEELPLLLPETDEIKPSGTGESPLANIDSFVNVVDEKT +GMKGRRETNTMPQWAGSCWYYLRYIDPKNENMLADPEKLKHWLPVDLYIGGVEHAVLHLL +YARFWHKVLYDLGIVPTKEPFQKLFNQGMILGEGNEKMSKSKGNVINPDDIVQSHGADTL +RLYEMFMGPLDAAIAWSEKGLDGSRRFLDRVWRLMVNEDGTLSSKIVTTNNKSLDKVYNQ +TVKKVTEDFETLGFNTAISQLMVFINECYKVDEVYKPYIEGFVKMLAPIAPHIGEELWSK +LGHEESITYQPWPTYDEALLVDDEVEIVVQVNGKLRAKIKIAKDTSKEEMQEIALSNDNV +KASIEGKDIMKVIAVPQKLVNIVAK +>sp|A8YW47|RS6_LACH4 Small ribosomal subunit protein bS6 OS=Lactobacillus helveticus (strain DPC 4571) OX=405566 GN=rpsF PE=3 SV=1 +MATTKYEVTYIIKPDVDEESKKALVENYDKVIADNGGTMVESKDWGKRRFAYEIDKYREG +TYHIMTFTADNADAVNEFGRLSKIDNMILRSMTVKLDK +>sp|A8YYT2|SYS_STAAT Serine--tRNA ligase OS=Staphylococcus aureus (strain USA300 / TCH1516) OX=451516 GN=serS PE=3 SV=1 +MLDIRLFRNEPDTVKSKIELRGDDPKVVDEILELDEQRRKLISATEEMKARRNKVSEEIA +LKKRNKENADDVIAEMRTLGDDIKEKDSQLNEIDNKMTGILCRIPNLISDDVPQGESDED +NVEVKKWGTPREFSFEPKAHWDIVEELKMADFDRAAKVSGARFVYLTNEGAQLERALMNY +MITKHTTQHGYTEMMVPQLVNADTMYGTGQLPKFEEDLFKVEKEGLYTIPTAEVPLTNFY +RNEIIQPGVLPEKFTGQSACFRSEAGSAGRDTRGLIRLHQFDKVEMVRFEQPEDSWNALE +EMTTNAEAILEELGLPYRRVILCTGDIGFSASKTYDLEVWLPSYNDYKEISSCSNCTDFQ +ARRANIRFKRDKAAKPELAHTLNGSGLAVGRTFAAIVENYQNEDGTVTIPEALVPFMGGK +TQISKPVK +>sp|A8Z4J3|RISB_STAAT 6,7-dimethyl-8-ribityllumazine synthase OS=Staphylococcus aureus (strain USA300 / TCH1516) OX=451516 GN=ribH PE=3 SV=1 +MNFEGKLIGKDLKVAIVVSRFNDFITGRLLEGAKDTLIRHDVNEDNIDVAFVPGAFEIPL +VAKKLASSGNYDAVITLGCVIRGATSHYDYVCNEVAKGVSKVNDQTNVPVIFGILTTESI +EQAVERAGTKAGNKGAEAAVSAIEMANLLKSIKA +>sp|A9KHK4|GLRP_LACP7 D-galactosyl-beta-1->4-L-rhamnose phosphorylase OS=Lachnoclostridium phytofermentans (strain ATCC 700394 / DSM 18823 / ISDg) OX=357809 GN=Cphy_1920 PE=1 SV=1 +MEQQKEITKGGFTLPGEAGFEKLTLELANRWGADVIRDSDGTELSDDILNAGYGIYSTIC +LIRDHNAWAKANIDKLQQTFLVTSPVVANSETLTIDLMEGFFKEQFLVNDSEEALEYWQV +YDRTTETLLQKESWSYHPQNQTVVLTGICPWHKYTVSFMAYRIWEEISMYNHTTNNWNKE +HLMQIDPIHKETQEYLLTWMDDWCKKHEQTTVVRFTSMFYNFVWMWGSNEKNRYLFSDWA +SYDFTVSPHALKLFEEEYGYVLTAEDFIHQGKFHVTHMPADKHKLDWMEFINNFVVDFGK +KLIDIVHNYGKLAYVFYDDSWVGVEPYHKNFEKFGFDGLIKCVFSGFEVRLCAGVKVNTH +ELRLHPYLFPVGLGGAPTFMEGGNPTLDAKNYWISVRRALLREPIDRIGLGGYLHLVEDF +PDFTDYIEKIANEFRRIKELHNAGKPMALKPRIAVLHSWGSLRSWTLSGHFHETYMHDLI +HINESLSGLPFDVKFINFEDINQGALEEVDVVINAGIMGSAWTGGQAWEDQEIIERLTRF +VYEGKAFIGVNEPSALTGYDTLYRMAHVLGVDMDLGDRVSHGRYSFTEEPVEELEFAECG +PKAKRNIYLTDGLAKVLKEENGIPVMTSYEFGRGRGIYLASYEHSIKNARTLLNIILYAA +GESFHQEGITNNVYTECAYYEKDKILVMINNSNTLQESSVTIKGRTYTKDIPAFDTVILP +LE +>sp|A9KPP1|DNAA_LACP7 Chromosomal replication initiator protein DnaA OS=Lachnoclostridium phytofermentans (strain ATCC 700394 / DSM 18823 / ISDg) OX=357809 GN=dnaA PE=3 SV=1 +MKSLIQEKWNEILEFLKIEYNVTEVSYKTWLLPLKVYDVKDNVIKLSVDDTKIGANSLDF +IKNKYSQFLKTAIAEVINQDFEIEFVLLSQTKAEEKVQTQAPNKIKNESLSYLNPRYTFD +TFVVGANNNLAHAASLAVAESPAEIYNPLFIYGGVGLGKTHLMHSIAHYILEQNPNSKVL +YVTSEKFTNELIESIRNADTTPTEFREKYRNIDVLLIDDIQFIIGKERTQEEFFHTFNTL +HESKKQIIISSDKPPKDILTLEERLRSRFEWGLTVDIQSPDYETRMAILKKKEELDCLTI +DDEVMKYIASNIKSNIRELEGALTKIVALSRLKKKEVDVILAEEALKDLISPDNKKTVTL +DLIIEVVSEHFTTSTSEIYSDNRSRNIAYPRQIAMYLCRKLTSLSLTDIGKMMGNRDHST +VLHGCNKVEKDIKKDPSFQNTIDVLIKKINPTP +>sp|B2UYM8|SYY_CLOBA Tyrosine--tRNA ligase OS=Clostridium botulinum (strain Alaska E43 / Type E3) OX=508767 GN=tyrS PE=3 SV=1 +MANVLDELLDRGYIKQFTHEEETRKLLENEKVTFYIGFDPTADSLHVGHFIAMMFMAHMQ +RAGHRPIALLGGGTAMVGDPSGKTDMRKMLTKEQIQHNVDSIKKQMERFIDFSDDKALIV +NNADWLLDLNYVDFLREVGVHFSVNRMLSAECFKQRLEKGLSFLEFNYMLMQGYDFYVLN +QKYGCKMELGGDDQWSNMIAGVELVRRKAQGDAMAMTCTLLTNSQGQKMGKTVGGALWLD +ADKVSPFDFYQYWRNVDDADVEKCLALLTFLPMDEVRRLGALEGAEINGAKKILAFEVTK +LVHGEEEAKKAEEAANALFSGGADMSNVPTVTIAKEDLGSTVLDIIAKTKIVPSKKEGRR +LIEQGGLSINGEKIIDLTRTLNEDDFEDGSALIKRGKKNYNKIEIQ +>sp|B5ELX2|RPOB_ACIF5 DNA-directed RNA polymerase subunit beta OS=Acidithiobacillus ferrooxidans (strain ATCC 53993 / BNL-5-31) OX=380394 GN=rpoB PE=3 SV=1 +MAYSFTEKKRIRKDFGKSQSILAVPYLLATQMDSYREFLQESVPPAARKETGLEAVFRSV +FPMESYSGNAMLDYVSYRLEKPVFDVVECRQRGLTYCAGLRVRLRLAVMEKDDTTGAKRV +KDVKEQDVYMGELPLMTEHGSFVINGTERVIVSQLHRSPGVFFDHDRGKTHSSGKLLFNA +RIIPYRGSWLDFEFDPKDHVYARIDRRRKLPATTLLRALGYSTQDILEMFFDMETFRLID +GDLRYVLIPQRLQGEVAAFDIVSPETGDVLVQAGKRITVRQTKALSEVQGLHEIPVPDSF +LLGKVVARDIVHPETGEVVVQANEPVSGELLDALRKMPSLTLHTLYLNELDRGPYISETL +RIDTSRDAHDAQMEIYRLMRPGEPPTKDAAQNLFQGLFFSPDRYDLSAVGRMKFNRRVGR +DEITGPGVLNDADIIAVLKVLVALRNGVGEIDDIDHLGNRRVRSVGELMENQFRLGLVRV +ERAVKDRLALAESEGLTPQDLINAKPIAAVVNEFFGSSQLSQFMDQTNPLSEVTHKRRVS +ALGPGGLTRERAGFEVRDVHPTHYGRICPIETPEGPNIGLINSLSCYARTNSYGFLETPY +RRVVDRRTTDEVEYLSAIEEGNYMIAQANSTVDEHGVLTDELVSCRFKNEFTLANPDQVQ +FMDISPRQIVSVAASMIPFLEHDDANRALMGSNMQRQAVPTVRSDAPMVGTGMERVVAID +SGAAVVARRGGVVDIVDGARIVIRVSDEETLPEEPGVDIYNLVKYARSNQNTTLNQRPVV +KVGDVVSRGDVLADGPSTEMGELALGQNILVAFMPWNGYNFEDSILISERVVAEDRYTTI +HIEEFSVFARDTKLGPEEITRDIPNVAEGALRHLDESGIVVIGAELAPGDILVGKVTPKG +ETQLTPEEKLLRAIFGEKASDVKDNSMRMPAGMYGTVIDVQVFTRDGIEKDARAKSIEEH +ELARIRKDLNDQYRIVEEDSYQRIERQLIGKVAEGGPNGLAAGNKVTKAYLKDLPRAKWF +EIRLRTEESNESLEQIRAQLEEQRQRLDAVLEEKRRKLTQGDDLSPGVLKMVKVHVAIKR +HLQPGDKMAGRHGNKGVVSKIVPVEDMPYLADGTAVDIVLNPLGVPSRMNVGQILETHLG +WAAKGLGKKIGAMLDSEVAMAEMRAFLAEIYNRSGKKEDLDSLTDQEIRELSANLRGGVP +MATPVFDGASEEEIGDMLELAGLPRSGQVTLYDGRSGDAFDRPVTVGYLYMLKLHHLVDD +KMHARSTGPYSLVTQQPLGGKAQFGGQRFGEMEVWALEAYGAAYTLQEMLTVKSDDVSGR +SKMYESIVKGDFRMDAGMPESFNVLLKELRSLGIDIELEQSK +>sp|B8I2Z3|PANC_RUMCH Pantothenate synthetase OS=Ruminiclostridium cellulolyticum (strain ATCC 35319 / DSM 5812 / JCM 6584 / H10) OX=394503 GN=panC PE=3 SV=1 +MKSVNTIHEVKNIVKDWKKQGLSVGLVPTMGYLHEGHGSLILKARENSKVVVSIFVNPMQ +FGPTEDLEKYPRDLEKDLKYCEELGADLIFSPEPGEMYPEGFCTSVDMSVLTEELCGLSR +PSHFKGVCTIVNKLLNIVNPDRAYFGEKDAQQLVIVKRMVRDLNMDIEIVGCPIIREKDG +LAKSSRNTYLNSEERRAALVLSKSIFTGKDMVKKGCRETDVLLKKMKAIIDQEPLARIDY +LKAVDTMTMQQVDTIDRPVLIAMAVYIGNVRLIDNFSFIPD +>sp|C0SPC1|CCRZ_BACSU Cell cycle regulator CcrZ OS=Bacillus subtilis (strain 168) OX=224308 GN=ccrZ PE=1 SV=2 +MNIDMNWLGQLLGSDWEIFPAGGATGDAYYAKHNGQQLFLKRNSSPFLAVLSAEGIVPKL +VWTKRMENGDVITAQHWMTGRELKPKDMSGRPVAELLRKIHTSKALLDMLKRLGKEPLNP +GALLSQLKQAVFAVQQSSPLIQEGIKYLEEHLHEVHFGEKVVCHCDVNHNNWLLSEDNQL +YLIDWDGAMIADPAMDLGPLLYHYVEKPAWESWLSMYGIELTESLRLRMAWYVLSETITF +IAWHKAKGNDKEFHDAMEELHILMKRIVD +>sp|C4Z940|RECF_AGARV DNA replication and repair protein RecF OS=Agathobacter rectalis (strain ATCC 33656 / DSM 3377 / JCM 17463 / KCTC 5835 / VPI 0990) OX=515619 GN=recF PE=3 SV=1 +MIIKSIQLSNFRNYEKLDISFDTETNIIYGDNAQGKTNILEAAYLSGTTKSHKGSKDKEM +IRFGEDEAHIRTIVEKNDKEYRIDMHLRKNGAKGVAINKMPIKKASELFGILNIVFFSPE +DLNIIKNGPAERRRFIDLELCQLDKIYLSNLSKYNKTLVQRNRLLKDIAYRPDLIDTLQV +WDMQLLEYGRHVIKKRREFVNELNEIIQDIHSNISGGREKLILKYEPSIDDIFFEDELLK +ARSRDLKLCQTTVGPHRDDMLFSVDGVDIRKYGSQGQQRTSALSLKLSEISLVKKNINST +PVLLLDDVLSELDGNRQNYLLNSLSDTQTIITCTGLDEFVKNRFQVDKVFHVVKGQVEVI +DE +>sp|E3GC98|MDTG_ENTLS Multidrug resistance protein MdtG OS=Enterobacter lignolyticus (strain SCF1) OX=701347 GN=mdtG PE=3 SV=1 +MYSSDAPINWKRNLTITWIGCFLTGAAFSLVMPFLPLYVESLGVTGHSALNMWSGLVFSI +TFLFSAIASPFWGGLADRKGRKIMLLRSALGMAIVMLLMGLAQNIWQFLILRALLGLLGG +FIPNANALIATQIPRHKSGWALGTLSTGGVSGALLGPLVGGVLADSYGLRPVFFMTASVL +FLCFLLTLLFIREQFQPVAKKEMLHAREVIASLKSPKLVLSLFVTTLIIQVATGSIAPIL +TLYVRDLAGNVANIAFISGMIASVPGVAALISAPRLGKLGDRIGPEKILIAALVISVLLL +IPMSFVQTPWQLGLLRFLLGAADGALLPAVQTLLVYNATNQIAGRVFSYNQSFRDIGNVT +GPLIGASVSANYGFRAVFLVTAMVVLFNAVYTGLSLRRTGATALPAQDVEKDDTSLVR +>sp|F9UMX3|GLPF4_LACPL Glycerol uptake facilitator protein 4 OS=Lactiplantibacillus plantarum (strain ATCC BAA-793 / NCIMB 8826 / WCFS1) OX=220668 GN=glpF4 PE=3 SV=1 +MIHQLLAEFMGTALMIIFGVGVHCSEVLKGTKYRGSGHIFAITTWGFGITIALFIFGNVC +INPAMVLAQCILGNLSWSLFIPYSVAEVLGGVVGAVIVWIMYADHFAASADEISPITIRN +LFSTAPAVRNLPRNFFVEFFDTFIFISGILAISEVKTPGIVPIGVGLLVWAIGMGLGGPT +GFAMNLARDMGPRIAHAILPIKNKADSDWQYGIIVPGIAPFVGAACAALFMHGFFGIG +>sp|H8L901|RACD_ENTFU Aspartate racemase OS=Enterococcus faecium (strain Aus0004) OX=1155766 GN=EFAU004_01689 PE=1 SV=1 +MENFFSILGGMGTMATESFVRLINHRTKATKDQEYLNYVLFNHATVPDRTAYILDRSEEN +PMPFLLDDIEKQNLLRPNFIVLTCNTAHYFFEELQAATDIPILHMPREAANELVRQHTTG +RVAILGTEGSMKAGIYEREVKNLGFETVIPDTALQEKINYLIYHEIKESDYLNQELYYEI +LEEAVERLNCEKVILGCTELSLMHEFAEDNHYPVIDAQSILADRTIERALAERSEALDTA +SEK +>sp|H8L902|ASL_ENTFU D-aspartate ligase OS=Enterococcus faecium (strain Aus0004) OX=1155766 GN=EFAU004_01690 PE=1 SV=1 +MMNSIENEEFIPILLGSDMNVYGMARSFNEAYGKICQAYASDQLAPTRYSKIVNVEVIPG +FDKDPVFIETMLRLAKERYSDKSKKYLLIACGDGYAELISQHKQELSEYFICPYIDYSLF +ERLINKVSFYEVCEEYDLPYPKTLIVREEMLVNGHLEQELPFEFPVALKPANSVEYLSVQ +FEGRKKAFILETREEFDLILGRIYEAGYKSEMIVQDFIPGDDSNMRVLNAYVDEDHQVRM +MCLGHPLLEDPTPASIGNYVVIMPDYNEKIYQTIKAFLEKIEYTGFANFDMKYDPRDGEY +KLFEINLRQGRSSFFVTLNGLNLARFVTEDRVFNKPFVETTYGTNQSDKARLWMGVPKKI +FLEYARENEDKKLAEQMIKENRYGTTVFYEKDRSIKRWLLMKYMFHNYIPRFKKYFHVKE +G +>sp|O06672|DPO3B_STRPN Beta sliding clamp OS=Streptococcus pneumoniae serotype 4 (strain ATCC BAA-334 / TIGR4) OX=170187 GN=dnaN PE=1 SV=2 +MIHFSINKNLFLQALNTTKRAISSKNAIPILSTVKIDVTNEGITLIGSNGQISIENFISQ +KNEDAGLLITSLGSILLEASFFINVVSSLPDVTLDFKEIEQNQIVLTSGKSEITLKGKDS +EQYPRIQEISASTPLILETKLLKKIINETAFAASTQESRPILTGVHFVLSQHKELKTVAT +DSHRLSQKKLTLEKNSDDFDVVIPSRSLREFSAVFTDDIETVEIFFANNQILFRSENISF +YTRLLEGNYPDTDRLIPTDFNTTITFNVVNLRQSMERARLLSSATQNGTVKLEIKDGVVS +AHVHSPEVGKVNEEIDTDQVTGEDLTISFNPTYLIDSLKALNSEKVTISFISAVRPFTLV +PADTDEDFMQLITPVRTN +>sp|O32068|YTZG_BACSU Uncharacterized RNA pseudouridine synthase YtzG OS=Bacillus subtilis (strain 168) OX=224308 GN=ytzG PE=3 SV=2 +MRLDKLLANSGYGSRKEVKAVVKAGAVMIDGKPAKDVKEHVDPDTQEVTVYGEPVDYREF +IYLMMNKPQGVLSATEDSRQQTVVDLLTPEEMRFEPFPAGRLDKDTEGFLLLTNDGQLAH +RLLSPKKHVPKTYEVHLKSQISREDISDLETGVYIEGGYKTKPAKAEIKTNDSGNTVIYL +TITEGKYHQVKQMAKAVGNEVVYLKRLSMGRVSLDPALAPGEYRELTEEELHLLNEPQA +>sp|O34357|YTPP_BACSU Thioredoxin-like protein YtpP OS=Bacillus subtilis (strain 168) OX=224308 GN=ytpP PE=2 SV=1 +MKKIESTQELEKAVKDDWSVFMFSADWCPDCRFVEPFLPELEANFPEFTYYYVDRDKFID +TCAEWEIYGIPSFVVFNEGKEVNRFVSKDRKTKEEIEQFLTDSLAKA +>sp|O34546|YTTB_BACSU Uncharacterized MFS-type transporter YttB OS=Bacillus subtilis (strain 168) OX=224308 GN=yttB PE=3 SV=1 +MPRALKILVIGMFINVTGASFLWPLNTIYIHNHLGKSLTVAGLVLMLNSGASVAGNLCGG +FLFDKIGGFKSIMLGIAITLASLMGLVFFHDWPAYIVLLTIVGFGSGVVFPASYAMAGSV +WPEGGRKAFNAIYVAQNAGVAVGSALGGVVASFSFSYVFLANAVLYLIFFFIVYFGFRNI +QTGDASQTSVLDYDAVNSKAKFAALIILSGGYVLGWLAYSQWSTTIASYTQSIGISLSLY +SVLWTVNGILIVLGQPLVSFVVKKWAESLKAQMVIGFIIFIVSFSMLLTAKQFPMFLAAM +VILTIGEMLVWPAVPTIANQLAPKGKEGFYQGFVNSAATGGRMIGPLFGGVLVDHYGIRA +LVLSLLVLLLISIATTLLYDKRIKSAKETNKQASISS +>sp|O34760|YTNP_BACSU Probable quorum-quenching lactonase YtnP OS=Bacillus subtilis (strain 168) OX=224308 GN=ytnP PE=1 SV=2 +METMKIGNITLTWLDGGVTHMDGGAMFGVVPKPLWSKKYPVNEKNQIELRTDPILIQKDG +LNIIIDAGIGYGKLTDKQKRNYGVTQESNVKPSLAALGLTVADIDVIAMTHLHFDHACGL +TEYEGERLVSVFPNAVIYTSAVEWDEMRHPNIRSKNTYWKENWEAVAGQVKTFEDTLTIT +EGITMHHTGGHSDGHSVLICEDAGETAVHMADLMPTHAHRNPLWVLAYDDYPMTSIPQKQ +KWQAFAAEKDAWFIFYHDAEYRALQWEEDGSIKKSVKRMKR +>sp|O34924|YTOP_BACSU Putative aminopeptidase YtoP OS=Bacillus subtilis (strain 168) OX=224308 GN=ytoP PE=3 SV=1 +MNQETKALFQTLTQLPGAPGNEHQVRAFMKQELAKYADDIVQDRLGSVFGVRRGAEDAPR +IMVAGHMDEVGFMVTSITDNGLLRFQTLGGWWSQVLLAQRVEIQTDNGPVPGVISSIPPH +LLTDAQRNRPMDIKNMMIDIGADDKEDAIKIGIRPGQQIVPVCPFTTMANEKKILSKAWD +NRYGCGLSIELLKELHGKELPNTLYAGATVQEEVGLRGAQTASHMIKPDLFFALDASPAN +DMSGDKNEFGQLGKGFLLRILDRTTVMHRGMREFVLDMAETHDIPYQYFVSGGGTDAGKV +HISNSGVPSAVIGICSRYIHTNATIIHIDDYAAAKEMLIKLVTACDKQTVDAIKENM +>sp|O34943|YTPR_BACSU Putative tRNA-binding protein YtpR OS=Bacillus subtilis (strain 168) OX=224308 GN=ytpR PE=4 SV=1 +MNAFYNKEGVGDTLLISLQDVTREQLGYEKHGDVVKIFNNETKETTGFNIFNASSYLTID +ENGPVALSETFVQDVNEILNRNGVEETLVVDLSPKFVVGYVESKEKHPNADKLSVCKVNV +GEETLQIVCGAPNVDQGQKVVVAKVGAVMPSGLVIKDAELRGVPSSGMICSAKELDLPDA +PAEKGILVLEGDYEAGDAFQF +>sp|O34948|YKWC_BACSU Uncharacterized oxidoreductase YkwC OS=Bacillus subtilis (strain 168) OX=224308 GN=ykwC PE=1 SV=1 +MKKTIGFIGLGVMGKSMASHILNDGHPVLVYTRTKEKAESILQKGAIWKDTVKDLSKEAD +VIITMVGYPSDVEEVYFGSNGIIENAKEGAYLIDMTTSKPSLAKKIAEAAKEKALFALDA +PVSGGDIGAQNGTLAIMVGGEKEAFEACMPIFSLMGENIQYQGPAGSGQHTKMCNQIAIA +AGMIGVAEAMAYAQKSGLEPENVLKSITTGAAGSWSLSNLAPRMLQGNFEPGFYVKHFIK +DMGIALEEAELMGEEMPGLSLAKSLYDKLAAQGEENSGTQSIYKLWVK +>sp|O35008|YTQA_BACSU Uncharacterized protein YtqA OS=Bacillus subtilis (strain 168) OX=224308 GN=ytqA PE=3 SV=1 +MMQNNPFPYSNTEKRYHTLNYHLREHFGHKVFKVALDGGFDCPNRDGTVAHGGCTFCSAA +GSGDFAGNRTDDLITQFHDIKNRMHEKWKDGKYIAYFQAFTNTHAPVEVLREKFESVLAL +DDVVGISIATRPDCLPDDVVDYLAELNERTYLWVELGLQTVHERTALLINRAHDFNCYVE +GVNKLRKHGIRVCSHIINGLPLEDRDMMMETAKAVADLDVQGIKIHLLHLLKGTPMVKQY +EKGKLEFLSQDDYVQLVCDQLEIIPPEMIVHRITGDGPIELMIGPMWSVNKWEVLGAINK +ELENRGSYQGKFFQRLEEESAL +>sp|O50628|GYRA_HALH5 DNA gyrase subunit A OS=Halalkalibacterium halodurans (strain ATCC BAA-125 / DSM 18197 / FERM 7344 / JCM 9153 / C-125) OX=272558 GN=gyrA PE=3 SV=1 +MAEQDQSRVKEINISQEMKTSFMDYAMSVIVSRALPDVRDGMKPVHRRILYAMNELGMTS +DKAYKKSARIVGEVIGKYHPHGDSAVYETMVRMAQDFSYRYMLVDGHGNFGSIDGDAAAA +MRYTEARMSKISMELVRDINKDTIDYQDNYDGSEKEPVVMPSRFPNLLVNGASGIAVGMA +TNIPPHQLGEVIDGVLALSKNPDISVPELMEHIPGPDFPTGAEILGRSGIRKAYQTGRGS +ITLRAKTEIEEHHGKQRIIVHEIPYQVNKAKLIEKIAELVRDKKIDGITDLRDESDRNGM +RIVIEVRKDANANVLLNNLYKQTALQTSFGINLLALVEGQPKVLNLKECLEHYLAHQVIV +IRRRTAFELRKAEARAHILEGLRIALDHLDEVISLIRSSQTTEIARNGLMERFELSYEQA +QAILDMRLQRLTGLERDKIEAEYKELIERIAELKAILADHEKVLDIIREELLELKEKYND +ERKTAISASEDMFEDEDLIPRQNVVITLTHHGYIKRLPISTYRSQKRGGRGIQGMGTNED +DFVQHLFTTNSHHTILFFTNKGKVYRLKGYEIPELGRTAKGIPIINLLQIEQDEYISTII +PIEEFTEDHYLFFMTKDGIAKRTQLSSFANIRRGGLFAINLREGDELHGVRLTNGDKEVI +VGTRQGMAIRFHETDVRLMGRTATGVKGISLTGDDHVVGMDIVEDGQDVLIVTEKGFGKR +TPIADYRIQTRGGKGIKTCNITEKNGPLVSLKVVSVDHDLMIITASGIIIRLHVKDISVT +GRITQGVTLIRVAEGEEVATVARVDIEDDELDEDESIEEERDDRSEVEQGENE +>sp|P05057|KANU_STAAU Aminoglycoside nucleotidyltransferase (4') OS=Staphylococcus aureus OX=1280 GN=knt PE=1 SV=1 +MNGPIIMTREERMKIVHEIKERILDKYGDDVKAIGVYGSLGRQTDGPYSDIEMMCVMSTE +EAEFSHEWTTGEWKVEVNFDSEEILLDYASQVESDWPLTHGQFFSILPIYDSGGYLEKVY +QTAKSVEAQTFHDAICALIVEELFEYAGKWRNIRVQGPTTFLPSLTVQVAMAGAMLIGLH +HRICYTTSASVLTEAVKQSDLPSGYDHLCQFVMSGQLSDSEKLLESLENFWNGIQEWTER +HGYIVDVSKRIPF +>sp|P05649|DPO3B_BACSU Beta sliding clamp OS=Bacillus subtilis (strain 168) OX=224308 GN=dnaN PE=1 SV=1 +MKFTIQKDRLVESVQDVLKAVSSRTTIPILTGIKIVASDDGVSFTGSDSDISIESFIPKE +EGDKEIVTIEQPGSIVLQARFFSEIVKKLPMATVEIEVQNQYLTIIRSGKAEFNLNGLDA +DEYPHLPQIEEHHAIQIPTDLLKNLIRQTVFAVSTSETRPILTGVNWKVEQSELLCTATD +SHRLALRKAKLDIPEDRSYNVVIPGKSLTELSKILDDNQELVDIVITETQVLFKAKNVLF +FSRLLDGNYPDTTSLIPQDSKTEIIVNTKEFLQAIDRASLLAREGRNNVVKLSAKPAESI +EISSNSPEIGKVVEAIVADQIEGEELNISFSPKYMLDALKVLEGAEIRVSFTGAMRPFLI +RTPNDETIVQLILPVRTY +>sp|P05650|RLBA_BACSU Probable ribosome maturation protein RlbA OS=Bacillus subtilis (strain 168) OX=224308 GN=rlbA PE=4 SV=1 +MANPISIDTEMITLGQFLKLADVIQSGGMAKWFLSEHEVLVNDEPDNRRGRKLYVGDVVE +IEGFGSFQVVN +>sp|P05653|GYRA_BACSU DNA gyrase subunit A OS=Bacillus subtilis (strain 168) OX=224308 GN=gyrA PE=1 SV=1 +MSEQNTPQVREINISQEMRTSFLDYAMSVIVSRALPDVRDGLKPVHRRILYAMNDLGMTS +DKPYKKSARIVGEVIGKYHPHGDSAVYESMVRMAQDFNYRYMLVDGHGNFGSVDGDSAAA +MRYTEARMSKISMEILRDITKDTIDYQDNYDGSEREPVVMPSRFPNLLVNGAAGIAVGMA +TNIPPHQLGEIIDGVLAVSENPDITIPELMEVIPGPDFPTAGQILGRSGIRKAYESGRGS +ITIRAKAEIEQTSSGKERIIVTELPYQVNKAKLIEKIADLVRDKKIEGITDLRDESDRTG +MRIVIEIRRDANANVILNNLYKQTALQTSFGINLLALVDGQPKVLTLKQCLEHYLDHQKV +VIRRRTAYELRKAEARAHILEGLRVALDHLDAVISLIRNSQTAEIARTGLIEQFSLTEKQ +AQAILDMRLQRLTGLEREKIEEEYQSLVKLIAELKDILANEYKVLEIIREELTEIKERFN +DERRTEIVTSGLETIEDEDLIERENIVVTLTHNGYVKRLPASTYRSQKRGGKGVQGMGTN +EDDFVEHLISTSTHDTILFFSNKGKVYRAKGYEIPEYGRTAKGIPIINLLEVEKGEWINA +IIPVTEFNAELYLFFTTKHGVSKRTSLSQFANIRNNGLIALSLREDDELMGVRLTDGTKQ +IIIGTKNGLLIRFPETDVREMGRTAAGVKGITLTDDDVVVGMEILEEESHVLIVTEKGYG +KRTPAEEYRTQSRGGKGLKTAKITENNGQLVAVKATKGEEDLMIITASGVLIRMDINDIS +ITGRVTQGVRLIRMAEEEHVATVALVEKNEEDENEEEQEEV +>sp|P07944|PBP_STAAU Beta-lactam-inducible penicillin-binding protein OS=Staphylococcus aureus OX=1280 GN=pbp PE=2 SV=1 +MKKIKIVPLILIVVVVGFGIYFYASKDKEINNTIDAIEDKNFKQVYKDSSYISKSDNGEV +EMTERPIKIYNSLGVKDINIQDRKIKKVSKNKKRVDAQYKIKTNYGNIDRNVQFNFVKED +GMWKLDWDHSVIIPGMQKDQSIHIENLKSERGKILDRNNVELANTGTHMRLGIVPKNVSK +KDYKAIAKELSISEDYINNKWIKIGYKMIPSFHFKTVKKMDEYLSDFAKKFHLTTNETES +RNYPLGKATSHLLGYVGPINSEELKQKEYKGYKDDAVIGKKGLEKLYDKKLQHEDGYRVT +IVRVDDNSNTIAHTLIEKKKKDGKDIQLTIDAKVQKSIYNNMKNDYGSGTAIHPQTGELL +ALVSTPSYDVYPFMYGMSNEEYNKLTEDKKEPLLNKFQITTSPGSTQKILTAMIGLNNKT +LDDKTSYKIDGKGWQKDKSWGGYNVTRYEVVNGNIDLKQAIESSDNIFFARVALELGSKK +FEKGMKKLGVGEDIPSDYPFYNAQISNKNLDNEILLADSGYGQGEILINPVQILSIYSAL +ENNGNINAPHLLKDTKNKVWKKNIISKENINLLNDGMQQVVNKTHKEDIYRSYANLIGKS +GTAELKMKQGETGRQIGWFISYDKDNPNMMMAINVKDVQDKGMASYNAKISGKVYDELYE +NGNKKYDIDE +>sp|P0A0B2|MECR_STAEP Methicillin resistance mecR1 protein OS=Staphylococcus epidermidis OX=1282 GN=mecR1 PE=3 SV=1 +MLSSFLMLSIISSLLTICVIFLVRMLYIKYTQNIMSHKIWLLVLVSTLIPLIPFYKISNF +TFSKDMMNRNVSDTTSSVSHMLDGQQSSVTKDLAINVNQFETSNITYMILLIWVFGSLLC +LFYMIKAFRQIDVIKSSSLESSYLNERLKVCQSKMQFYKKHITISYSSNIDNPMVFGLVK +SQIVLPTVVVETMNDKEIEYIILHELSHVKSHDLIFNQLYVVFKMIFWFNPALYISKTMM +DNDCEKVCDRNVLKILNRHEHIRYGESILKCSILKSQHINNVAAQYLLGFNSNIKERVKY +IALYDSMPKPNRNKRIVAYIVCSISLLIQAPLLSAHVQQDKYETNVSYKKLNQLAPYFKG +FDGSFVLYNEREQAYSIYNEPESKQRYSPNSTYKIYLALMAFDQNLLSLNHTEQQWDKHQ +YPFKEWNQDQNLNSSMKYSVNWYYENLNKHLRQDEVKSYLDLIEYGNEEISGNENYWNES +SLKISAIEQVNLLKNMKQHNMHFDNKAIEKVENSMTLKQKDTYKYVGKTGTGIVNHKEAN +GWFVGYVETKDNTYYFATHLKGEDNANGEKAQQISERILKEMELI +>sp|P0A0C3|REPB_STAAU Replication protein OS=Staphylococcus aureus OX=1280 GN=repB PE=3 SV=1 +MKHGIQSQKVVAEVIKQKPTVRWLFLTLTVKNVYDGEELNKSLSDMAQGFRRMMQYKKIN +KNLVGFMRATEVTINNKDNSYNQHMHVLVCVEPTYFKNTENYVNQKQWIQFWKKAMKLDY +DPNVKVQMIRPKNKYKSDIQSAIDETAKYPVKDTDFMTDDEEKNLKRLSDLEEGLHRKRL +ISYGGLLKEIHKKLNLDDTEEGDLIHTDDDEKADEDGFSIIAMWNWERKNYFIKE +>sp|P0A0C4|REPB_BACSP Replication protein OS=Bacillus sp. OX=1409 GN=repB PE=3 SV=1 +MKHGIQSQKVVAEVIKQKPTVRWLFLTLTVKNVYDGEELNKSLSDMAQGFRRMMQYKKIN +KNLVGFMRATEVTINNKDNSYNQHMHVLVCVEPTYFKNTENYVNQKQWIQFWKKAMKLDY +DPNVKVQMIRPKNKYKSDIQSAIDETAKYPVKDTDFMTDDEEKNLKRLSDLEEGLHRKRL +ISYGGLLKEIHKKLNLDDTEEGDLIHTDDDEKADEDGFSIIAMWNWERKNYFIKE +>sp|P0A0C7|PRE2_STAAU Plasmid recombination enzyme type 2 OS=Staphylococcus aureus OX=1280 GN=pre PE=3 SV=1 +MSYAVCRMQKVKSAGLKGMQFHNQRERKSRTNDDIDHERTRENYDLKNDKNIDYNERVKE +IIESQKTGTRKTRKDAVLVNELLVTSDRDFFEQLDPGEQKRFFEESYKLFSERYGKQNIA +YATVHNDEQTPHMHLGVVPMRDGKLQGKNVFNRQELLWLQDKFPEHMKKQGFELKRGERG +SDRKHIETAKFKKQTLEKEIDFLEKNLAVKKDEWTAYSDKVKSDLEVPAKRHMKSVEVPT +GEKSMFGLGKEIMKTEKKPTKNVVISERDYKNLVTAARDNDRLKQHVRNLMSTDMAREYK +KLSKEHGQVKEKYSGLVERFNENVNDYNELLEENKSLKSKISDLKRDVSLIYESTKEFLK +ERTDGLKAFKNVFKGFVDKVKDKTAQFQEKHDLEPKKNEFELTHNREVKKERSRDQGMSL +>sp|P0A150|YGIDB_PSEPU Uncharacterized protein in gidB 3'region OS=Pseudomonas putida OX=303 PE=4 SV=1 +MAKVFAIANQKGGVGKTTTCINLAASLAATKRRVLLIDLDPQGNATMGSGVDKHELEHSV +YDLLIGECDLAQAMHYSEHGGFQLLPANRDLTAAEVVLLEMQVKESRLRNALAPIRDNYD +YILIDCPPSLSMLTLNALVASDGVIIPMQCEYYALEGLSDLVDNIKRIAARLNPELKIEG +LLRTMYDPRLSLNNDVSAQLKEHFGPQLYDTVIPRNIRLAEAPSFGMPALAYDKQSRGAL +AYLALAGELVRRQRRPSRTAQTT +>sp|P0A4C6|RS4_BORBR Small ribosomal subunit protein uS4 OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rpsD PE=3 SV=1 +MARYIGPKCKLSRREGTDLFLKSARRSLDSKCKLDSKPGQHGRTSGARTSDYGLQLREKQ +KLKRMYGVLEKQFRKYFVEAERRRGNTGETLIQLLESRLDNVVYRMGFGSTRAEARQLVS +HRAIELNGHTADIASMLVKAGDVISIREKAKKQGRIRESLDLAASIGLPQWVEVDASKMT +GTFKSAPDRADVARDVNESMVVELYSR +>sp|P0A4E5|RPOA_BORPE DNA-directed RNA polymerase subunit alpha OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rpoA PE=3 SV=1 +MSTQGFLKPRSIEVEPVGAHHAKIVMEPFERGYGHTLGNALRRILLSSMTGYAPTEVQMT +GVVHEYSTIAGVREDVVDILLNLKGVVFKLHNRDEVTLVLRKNGAGAVVASDIELPHDVE +IINPDHLICNLTDAGKIEMQVKVEKGRGYVPGNVRALSEDRTHTIGRIVLDASFSPVRRV +SYAVESARVEQRTDLDKLVLDIETNGVISPEEAVRQAARILMDQISVFAALEGAGDAYEP +PVRGTPQIDPVLLRPVDDLELTVRSANCLKAENIYYIGDLIQRTENELLKTPNLGRKSLN +EIKEVLAARGLTLGMKLENWPPLGLERP +>sp|P0A4X9|MBTM_MYCBO Medium/long-chain-fatty-acid--[acyl-carrier-protein] ligase MbtM OS=Mycobacterium bovis (strain ATCC BAA-935 / AF2122/97) OX=233413 GN=mbtM PE=3 SV=1 +MSELAAVLTRSMQASAGDLMVLDRETSLWCRHPWPEVHGLAESVAAWLLDHDRPAAVGLV +GEPTVELVAAIQGAWLAGAAVSILPGPVRGANDQRWADATLTRFLGIGVRTVLSQGSYLA +RLRSVDTAGVTIGDLSTAAHTNRSATPVASEGPAVLQGTAGSTGAPRTAILSPGAVLSNL +RGLNQRVGTDAATDVGCSWLPLYHDMGLAFVLSAALAGAPLWLAPTTAFTASPFRWLSWL +SDSGATMTAAPNFAYNLIGKYARRVSEVDLGALRVTLNGGEPVDCDGLTRFAEAMAPFGF +DAGAVLPSYGLAESTCAVTVPVPGIGLLADRVIDGSGAHKHAVLGNPIPGMEVRISCGDQ +AAGNASREIGEIEIRGASMMAGYLGQQPIDPDDWFATGDLGYLGAGGLVVCGRAKEVISI +AGRNIFPTEVELVAAQVRGVREGAVVALGTGDRSTRPGLVVAAEFRGPDEANARAELIQR +VASECGIVPSDVVFVSPGSLPRTSSGKLRRLAVRRSLEMAD +>sp|P0AGA4|SECY_ECO57 Protein translocase subunit SecY OS=Escherichia coli O157:H7 OX=83334 GN=secY PE=3 SV=1 +MAKQPGLDFQSAKGGLGELKRRLLFVIGALIVFRIGSFIPIPGIDAAVLAKLLEQQRGTI +IEMFNMFSGGALSRASIFALGIMPYISASIIIQLLTVVHPTLAEIKKEGESGRRKISQYT +RYGTLVLAIFQSIGIATGLPNMPGMQGLVINPGFAFYFTAVVSLVTGTMFLMWLGEQITE +RGIGNGISIIIFAGIVAGLPPAIAHTIEQARQGDLHFLVLLLVAVLVFAVTFFVVFVERG +QRRIVVNYAKRQQGRRVYAAQSTHLPLKVNMAGVIPAIFASSIILFPATIASWFGGGTGW +NWLTTISLYLQPGQPLYVLLYASAIIFFCFFYTALVFNPRETADNLKKSGAFVPGIRPGE +QTAKYIDKVMTRLTLVGALYITFICLIPEFMRDAMKVPFYFGGTSLLIVVVVIMDFMAQV +QTLMMSSQYESALKKANLKGYGR +>sp|P0C1L0|T431_STAA8 Transposase for insertion sequence-like element IS431mec OS=Staphylococcus aureus (strain NCTC 8325 / PS 47) OX=93061 GN=tnp PE=4 SV=1 +MNYFRYKQFNKDVITVAVGYYLRYTLSYRDISEILRERGVNVHHSTVYRWVQEYAPILYQ +IWKKKHKKAYYKWRIDETYIKIKGKWSYLYRAIDAEGHTLDIWLRKQRDNHSAYAFIKRL +IKQFGKPQKVITDQAPSTKVAMAKVIKAFKLKPDCHCTSKYLNNLIEQDHRHIKVRKTRY +QSINTAKNTLKGIECIYALYKKNRRSLQIYGFSPCHEISIMLAS +>sp|P0C2T9|METC_LACLC Cystathionine beta-lyase OS=Lactococcus lactis subsp. cremoris OX=1359 GN=metC PE=1 SV=1 +MTSLKTKVIHGGISTDRTTGAVSVPIYQTSTYKQNGLGQPKEYEYSRSGNPTRHALEELI +ADLEGGVQGFAFSSGLAGIHAVLSLFSAGDHIILADDVYGGTFRLVDKVLTKTGIIYDLV +DLSNLEDLKAAFKAETKAVYFETPSNPLLKVLDIKEISSIAKAHNALTLVDNTFATPYLQ +QPIALGADIVLHSATKYLGGHSDVVAGLVTTNSNELAIEIGFLQNSIGAVLGPQDSWLVQ +RGIKTLAPRMEAHSANAQKIAEFLEASQAVSKVYYPGLVNHEGHEIAKKQMTAFGGMISF +ELTDENAVKNFVENLRYFTLAESLGGVESLIEVPAVMTHASIPKELREEIGIKDGLIRLS +VGVEALEDLLTDLKEALEKE +>sp|P0DG03|GYRA_STRPQ DNA gyrase subunit A OS=Streptococcus pyogenes serotype M3 (strain SSI-1) OX=193567 GN=gyrA PE=3 SV=1 +MQDRNLIDVNLTSEMKTSFIDYAMSVIVARALPDVRDGLKPVHRRILYGMNELGVTPDKP +HKKSARITGDVMGKYHPHGDSSIYEAMVRMAQWWSYRHMLVDGHGNFGSMDGDGAAAQRY +TEARMSKIALELLRDINKNTVNFQDNYDGSEREPVVLPARFPNLLVNGATGIAVGMATNI +PPHNLAESIDAVKMVMEHPDCTTRELMEVIPGPDFPTGALVMGRSGIHRAYDTGKGSIVL +RSRTEIETTQTGRERIVVTEFPYGVNKTKVHEHIVRLAQEKRLEGITAVRDESSREGVRF +VIEIRREASATVILNNLFKLTSLQTNFSFNMLAIENGVPKILSLRQIIDNYISHQKEVII +RRTRFDKDKAEARAHILEGLLIALDHLDEVIAIIRNSETDVIAQTELMSRFDLSERQSQA +ILDMRLRRLTGLERDKIQSEYDDLLALIADLSDILAKPERIITIIKEEMDEIKRKYANPR +RTELMVGEVLSLEDEDLIEEEDVLITLSNKGYIKRLAQDEFRAQKRGGRGVQGTGVNNDD +FVRELVSTSTHDTLLFFTNFGRVYRLKAYEIPEYGRTAKGLPIVNLLKLEDGETIQTIIN +ARKEETAGKSFFFTTKQGIVKRTEVSEFNNIRQNGLRALKLKEGDQLINVLLTSGQDDII +IGTHSGYSVRFNEASIRNMGRSATGVRGVKLREDDRVVGASRIQDNQEVLVITENGFGKR +TSATDYPTKGRGGKGIKTANITPKNGQLAGLVTVDGTEDIMVITNKGVIIRTNVANISQT +GRATLGVKIMKLDADAKIVTFTLVQPEDSSIAEINTDRENSISKNKDN +>sp|P0DKR7|WHB5B_MYCTO Transcriptional regulator WhiB5 OS=Mycobacterium tuberculosis (strain CDC 1551 / Oshkosh) OX=83331 GN=whiB5 PE=2 SV=1 +MAHPCATDPELWFGYPDDDGSDGAAKARAYERSATQARIQCLRRCPLLQQRRCAQHAVEH +RVEYGVWAGIKLPGGQYRKREQLAAAHDVLRRIAGGEINSRQLPDNAALLARNEGLEVTP +VPGVVVHLPIAQVGPQPAA +>sp|P10249|SUCP_STRMU Sucrose phosphorylase OS=Streptococcus mutans serotype c (strain ATCC 700610 / UA159) OX=210007 GN=gtfA PE=1 SV=4 +MPIINKTMLITYADSLGKNLKELNENIENYFGDAVGGVHLLPFFPSTGDRGFAPIDYHEV +DSAFGDWDDVKCLGEKYYLMFDFMINHISRQSKYYKDYQEKHEASAYKDLFLNWDKFWPK +NRPTQEDVDLIYKRKDRAPKQEIQFADGSVEHLWNTFGEEQIDLDVTKEVTMDFIRSTIE +NLAANGCDLIRLDAFAYAVKKLDTNDFFVEPEIWTLLDKVRDIAAVSGAEILPEIHEHYT +IQFKIADHDYYVYDFALPMVTLYSLYSSKVDRLAKWLKMSPMKQFTTLDTHDGIGVVDVK +DILTDEEITYTSNELYKVGANVNRKYSTAEYNNLDIYQINSTYYSALGDDDQKYFLARLI +QAFAPGIPQVYYVGFLAGKNDLELLESTKEGRNINRHYYSSEEIAKEVKRPVVKALLNLF +TYRNQSAAFDLDGRIEVETPNEATIVIERQNKDGSHIAKAEINLQDMTYRVTENDQTISF +E +>sp|P19834|YI11_STRCL Insertion element IS116 uncharacterized 44.8 kDa protein OS=Streptomyces clavuligerus OX=1901 PE=3 SV=1 +MSTRHDRIWVGIDAGKGHHWAVAVDADGETLFSTKVINDEAQVLTLIETAREREEVRWAV +DISGRASTLLLALLVAHGQNVVYVPGRTVNRMSGAYKGEGKTDAKDARVIADQARMRRDF +APLDRPPELVTTLRLLTNHRADLIADRVRLINRLRDLLTGICPALERAFDYSAAKGPVVM +LTEYQTPAALRRTGVKRLTTWLGRRKVRDADTVAAKAIEAARTQQVVLPGEKRATKLVCD +LAHQLLALDERIKDNDREIRETFRTDDRAEIIESMPGMGPVLGAEFVAIVGDLSGYKDAG +RLASHAGLAPVPRDSGRRTGNYHRPQRYNRRLRWLFYMSAQTAMMRPGPSRDYYLKKRGE +GLLHTQALLSLARRRVDVLWAMLRDKRLFTPAPPVTQTA +>sp|P26497|SP0J_BACSU Stage 0 sporulation protein J OS=Bacillus subtilis (strain 168) OX=224308 GN=spo0J PE=1 SV=2 +MAKGLGKGINALFNQVDLSEETVEEIKIADLRPNPYQPRKHFDDEALAELKESVLQHGIL +QPLIVRKSLKGYDIVAGERRFRAAKLAGLDTVPAIVRELSEALMREIALLENLQREDLSP +LEEAQAYDSLLKHLDLTQEQLAKRLGKSRPHIANHLRLLTLPENIQQLIAEGTLSMGHGR +TLLGLKNKNKLEPLVQKVIAEQLNVRQLEQLIQQLNQNVPRETKKKEPVKDAVLKERESY +LQNYFGTTVNIKRQKKKGKIEIEFFSNEDLDRILELLSERES +>sp|P27159|XYLR_STAXY Xylose repressor OS=Staphylococcus xylosus OX=1288 GN=xylR PE=3 SV=1 +MENNFIVNENEKRVLKQIFNNSNISRTQISKNLELNKATISNILNNLKHKSLVNEVGEGN +STKSGGRKPILLEINQKYGYYISMDLTYDSVELMYNYFDATILKQDSYELNDKNVSSILQ +ILKSNINVSEKYDTLYGLLGISISIHGIVDDEQNIINLPFHKNEKRTFTDELKSFTNVPV +VIENEANLSALYEKSLYINSNINNLITLSIHKGIGAGIIINKKLYRGSNGEAGEIGKTLV +LESINNNDNKYYKIEDICSQDALIQKINNRLGVTLTFTELIQYYNEGNSIVAHEIKQFIN +KMTVLIHNLNTQFNPDAIYINCPLINELPNILNEIKEQFSCFSQGSPIQLHLTTNVKQAT +LLGGTLAIMQKTLNINNIQMNIK +>sp|P37478|WALR_BACSU Transcriptional regulatory protein WalR OS=Bacillus subtilis (strain 168) OX=224308 GN=walR PE=1 SV=1 +MDKKILVVDDEKPIADILEFNLRKEGYEVHCAHDGNEAVEMVEELQPDLILLDIMLPNKD +GVEVCREVRKKYDMPIIMLTAKDSEIDKVIGLEIGADDYVTKPFSTRELLARVKANLRRQ +LTTAPAEEEPSSNEIHIGSLVIFPDAYVVSKRDETIELTHREFELLHYLAKHIGQVMTRE +HLLQTVWGYDYFGDVRTVDVTVRRLREKIEDNPSHPNWIVTRRGVGYYLRNPEQD +>sp|P37484|GDPP_BACSU Cyclic-di-AMP phosphodiesterase GdpP OS=Bacillus subtilis (strain 168) OX=224308 GN=gdpP PE=1 SV=1 +MPSFYEKPLFRYPIYALIALSIITILISFYFNWILGTVEVLLLAVILFFIKRADSLIRQE +IDAYISTLSYRLKKVGEEALMEMPIGIMLFNDQYYIEWANPFLSSCFNESTLVGRSLYDT +CESVVPLIKQEVESETVTLNDRKFRVVIKRDERLLYFFDVTEQIQIEKLYENERTVLAYI +FLDNYDDVTQGLDDQTRSTMNSQVTSLLNAWAQEYGIFLKRTSSERFIAVLNEHILTELE +NSKFSILDEVREKTSFDGVALTLSVGVGASVSSLKELGDLAQSSLDLALGRGGDQVAIKL +PNGKVKFYGGKTNPMEKRTRVRARVISHALKEIVTESSNVIIMGHKFPDMDSIGAAIGIL +KVAQANNKDGFIVIDPNQIGSSVQRLIGEIKKYEELWSRFITPEEAMEISNDDTLLVIVD +THKPSLVMEERLVNKIEHIVVIDHHRRGEEFIRDPLLVYMEPYASSTAELVTELLEYQPK +RLKINMIEATALLAGIIVDTKSFSLRTGSRTFDAASYLRAKGADTVLVQKFLKETVDSYI +KRAKLIQHTVLYKDNIAIASLPENEEEYFDQVLIAQAADSLLSMSEVEASFAVARRDEQT +VCISARSLGEVNVQIIMEALEGGGHLTNAATQLSGISVSEALERLKHAIDEYFEGGVQR +>sp|P37522|SOJ_BACSU Sporulation initiation inhibitor protein Soj OS=Bacillus subtilis (strain 168) OX=224308 GN=soj PE=1 SV=1 +MGKIIAITNQKGGVGKTTTSVNLGACLAYIGKRVLLVDIDPQGNATSGLGIEKADVEQCV +YDILVDDADVIDIIKATTVENLDVIPATIQLAGAEIELVPTISREVRLKRALEAVKQNYD +YIIIDCPPSLGLLTINALTASDSVVIPVQCEYYALEGLSQLLNTVRLVQKHLNTDLMIEG +VLLTMLDARTNLGIQVIEEVKKYFRDKVYKTVIPRNVRLSEAPSHGKPIILYDPRSRGAE +VYLDLAKEVAANG +>sp|P39062|ACSA_BACSU Acetyl-coenzyme A synthetase OS=Bacillus subtilis (strain 168) OX=224308 GN=acsA PE=1 SV=1 +MNLKALPAIEGDHNLKNYEETYRHFDWAEAEKHFSWHETGKLNAAYEAIDRHAESFRKNK +VALYYKDAKRDEKYTFKEMKEESNRAGNVLRRYGNVEKGDRVFIFMPRSPELYFIMLGAI +KIGAIAGPLFEAFMEGAVKDRLENSEAKVVVTTPELLERIPVDKLPHLQHVFVVGGEAES +GTNIINYDEAAKQESTRLDIEWMDKKDGFLLHYTSGSTGTPKGVLHVHEAMIQQYQTGKW +VLDLKEEDIYWCTADPGWVTGTVYGIFAPWLNGATNVIVGGRFSPESWYGTIEQLGVNVW +YSAPTAFRMLMGAGDEMAAKYDLTSLRHVLSVGEPLNPEVIRWGHKVFNKRIHDTWWMTE +TGSQLICNYPCMDIKPGSMGKPIPGVEAAIVDNQGNELPPYRMGNLAIKKGWPSMMHTIW +NNPEKYESYFMPGGWYVSGDSAYMDEEGYFWFQGRVDDVIMTSGERVGPFEVESKLVEHP +AIAEAGVIGKPDPVRGEIIKAFIALREGFEPSDKLKEEIRLFVKQGLAAHAAPREIEFKD +KLPKTRSGKIMRRVLKAWELNLPAGDLSTMED +>sp|P39912|AROG_BACSU Protein AroA(G) OS=Bacillus subtilis (strain 168) OX=224308 GN=aroA PE=1 SV=1 +MSNTELELLRQKADELNLQILKLINERGNVVKEIGKAKEAQGVNRFDPVRERTMLNNIIE +NNDGPFENSTIQHIFKEIFKAGLELQEEDHSKALLVSRKKKPEDTIVDIKGEKIGDGQQR +FIVGPCAVESYEQVAEVAAAAKKQGIKILRGGAFKPRTSPYDFQGLGVEGLQILKRVADE +FDLAVISEIVTPAHIEEALDYIDVIQIGARNMQNFELLKAAGAVKKPVLLKRGLAATISE +FINAAEYIMSQGNDQIILCERGIRTYETATRNTLDISAVPILKQETHLPVFVDVTHSTGR +RDLLLPTAKAALAIGADGVMAEVHPDPSVALSDSAQQMAIPEFEKWLNELKPMVKVNA +>sp|P50854|RISA_ACTPL Riboflavin synthase OS=Actinobacillus pleuropneumoniae OX=715 GN=ribE PE=3 SV=1 +MFTGIIEEVGKIAQIHKQGEFAVVTINATKVLQDVHLGDTIAVNGVCLTVTSFSSNQFTA +DVMSETLKRTSLGELKSNSPVNLERAMAANGRFGGHIVSGHIDGTGEIAEITPAHNSTWY +RIKTSPKLMRYIIEKGSITIDGISLTVVDTDDESFRVSIIPHTIKETNLGSKKIGSIVNL +ENDIVGKYIEQFLLKKPADEPKSNLSLDFLKQAGF +>sp|P54744|PKNB_MYCLE Serine/threonine-protein kinase PknB OS=Mycobacterium leprae (strain TN) OX=272631 GN=pknB PE=3 SV=3 +MTTPPHLSDRYELGDILGFGGMSEVHLARDIRLHRDVAVKVLRADLARDPSFYLRFRREA +QNAAALNHPSIVAVYDTGEAETSAGPLPYIVMEYVDGATLRDIVHTDGPMPPQQAIEIVA +DACQALNFSHQNGIIHRDVKPANIMISATNAVKVMDFGIARAIADSTSVTQTAAVIGTAQ +YLSPEQARGDSVDARSDVYSLGCVLYEILTGEPPFIGDSPVSVAYQHVREDPIPPSQRHE +GISVDLDAVVLKALAKNPENRYQTAAEMRADLIRVHSGQPPEAPKVLTDADRSCLLSSGA +GNFGVPRTDALSRQSLDETESDGSIGRWVAVVAVLAVLTIAIVAAFNTFGGNTRDVQVPD +VRGQVSADAISALQNRGFKTRTLQKPDSTIPPDHVISTEPGANASVGAGDEITINVSTGP +EQREVPDVSSLNYTDAVKKLTSSGFKSFKQANSPSTPELLGKVIGTNPSANQTSAITNVI +TIIVGSGPETKQIPDVTGQIVEIAQKNLNVYGFTKFSQASVDSPRPTGEVIGTNPPKDAT +VPVDSVIELQVSKGNQFVMPDLSGMFWADAEPRLRALGWTGVLDKGPDVDAGGSQHNRVA +YQNPPAGAGVNRDGIITLKFGQ +>sp|P56067|CYSM_HELPY Cysteine synthase OS=Helicobacter pylori (strain ATCC 700392 / 26695) OX=85962 GN=cysM PE=1 SV=1 +MMIITTMQDAIGRTPVFKFTNKDYPIPLNSAIYAKLEHLNPGGSVKDRLGQYLIGEGFKT +GKITSKTTIIEPTAGNTGIALALVAIKHHLKTIFVVPEKFSTEKQQIMRALGALVINTPT +SEGISGAIKKSKELAESIPDSYLPLQFENPDNPAAYYHTLAPEIVQELGTNLTSFVAGIG +SGGTFAGTARYLKERIPAIRLIGVEPEGSILNGGEPGPHEIEGIGVEFIPPFFENLDIDG +FETISDEEGFSYTRKLAKKNGLLVGSSSGAAFVAALKEAQRLPEGSQVLTIFPDVADRYL +SKGIYL +>sp|P63453|MBTL_MYCBO Acyl carrier protein MbtL OS=Mycobacterium bovis (strain ATCC BAA-935 / AF2122/97) OX=233413 GN=mbtL PE=3 SV=1 +MTSSPSTVSTTLLSILRDDLNIDLTRVTPDARLVDDVGLDSVAFAVGMVAIEERLGVALS +EEELLTCDTVGELEAAIAAKYRDE +>sp|P65476|MURC_STAAW UDP-N-acetylmuramate--L-alanine ligase OS=Staphylococcus aureus (strain MW2) OX=196620 GN=murC PE=3 SV=1 +MTHYHFVGIKGSGMSSLAQIMHDLGHEVQGSDIENYVFTEVALRNKGIKILPFDANNIKE +DMVVIQGNAFASSHEEIVRAHQLKLDVVSYNDFLGQIIDQYTSVAVTGAHGKTSTTGLLS +HVMNGDKKTSFLIGDGTGMGLPESDYFAFEACEYRRHFLSYKPDYAIMTNIDFDHPDYFK +DINDVFDAFQEMAHNVKKGIIAWGDDEHLRKIEADVPIYYYGFKDSDDIYAQNIQITDKG +TAFDVYVDGEFYDHFLSPQYGDHTVLNALAVIAISYLEKLDVTNIKEALETFGGVKRRFN +ETTIANQVIVDDYAHHPREISATIETARKKYPHKEVVAVFQPHTFSRTQAFLNEFAESLS +KADRVFLCEIFGSIRENTGALTIQDLIDKIEGASLINEDSINVLEQFDNAVVLFMGAGDI +QKLQNAYLDKLGMKNAF +>sp|P65592|NUSG_NEIMB Transcription termination/antitermination protein NusG OS=Neisseria meningitidis serogroup B (strain ATCC BAA-335 / MC58) OX=122586 GN=nusG PE=3 SV=1 +MSKKWYVVQAYSGFEKNVQRILEERIAREEMGDYFGQILVPVEKVVDIRNGRKTISERKS +YPGYVLVEMEMTDDSWHLVKSTPRVSGFIGGRANRPTPISQREAEIILQQVQTGIEKPKP +KVEFEVGQQVRVNEGPFADFNGVVEEVNYERNKLRVSVQIFGRETPVELEFSQVEKIN +>sp|P65763|PPIA_MYCBO Probable peptidyl-prolyl cis-trans isomerase A OS=Mycobacterium bovis (strain ATCC BAA-935 / AF2122/97) OX=233413 GN=ppiA PE=3 SV=1 +MADCDSVTNSPLATATATLHTNRGDIKIALFGNHAPKTVANFVGLAQGTKDYSTQNASGG +PSGPFYDGAVFHRVIQGFMIQGGDPTGTGRGGPGYKFADEFHPELQFDKPYLLAMANAGP +GTNGSQFFITVGKTPHLNRRHTIFGEVIDAESQRVVEAISKTATDGNDRPTDPVVIESIT +IS +>sp|P65884|PURA_STAAM Adenylosuccinate synthetase OS=Staphylococcus aureus (strain Mu50 / ATCC 700699) OX=158878 GN=purA PE=3 SV=1 +MSSIVVVGTQWGDEGKGKITDFLAEQSDVIARFSGGNNAGHTIQFGGETYKLHLVPSGIF +YKDKLAVIGNGVVVDPVALLKELDGLNERGIPTSNLRISNRAQVILPYHLAQDEYEERLR +GDNKIGTTKKGIGPAYVDKVQRIGIRMADLLEKETFERLLKSNIEYKQAYFKGMFNETCP +SFDDIFEEYYAAGQRLKEFVTDTSKILDDAFVADEKVLFEGAQGVMLDIDHGTYPFVTSS +NPIAGNVTVGTGVGPTFVSKVIGVCKAYTSRVGDGPFPTELFDEDGHHIREVGREYGTTT +GRPRRVGWFDSVVLRHSRRVSGITDLSINSIDVLTGLDTVKICTAYELDGKEITEYPANL +DQLKRCKPIFEELPGWTEDVTSVRTLEELPENARKYLERISELCNVQISIFSVGPDREQT +NLLKELW +>sp|P66936|GYRB_STAAM DNA gyrase subunit B OS=Staphylococcus aureus (strain Mu50 / ATCC 700699) OX=158878 GN=gyrB PE=3 SV=2 +MVTALSDVNNTDNYGAGQIQVLEGLEAVRKRPGMYIGSTSERGLHHLVWEIVDNSIDEAL +AGYANKIEVVIEKDNWIKVTDNGRGIPVDIQEKMGRPAVEVILTVLHAGGKFGGGGYKVS +GGLHGVGSSVVNALSQDLEVYVHRNETIYHQAYKKGVPQFDLKEVGTTDKTGTVIRFKAD +GEIFTETTVYNYETLQQRIRELAFLNKGIQITLRDERDEENVREDSYHYEGGIKSYVELL +NENKEPIHDEPIYIHQSKDDIEVEIAIQYNSGYATNLLTYANNIHTYEGGTHEDGFKRAL +TRVLNSYGLSSKIMKEEKDRLSGEDTREGMTAIISIKHGDPQFEGQTKTKLGNSEVRQVV +DKLFSEHFERFLYENPQVARTVVEKGIMAARARVAAKKAREVTRRKSALDVASLPGKLAD +CSSKSPEECEIFLVEGDSAGGSTKSGRDSRTQAILPLRGKILNVEKARLDRILNNNEIRQ +MITAFGTGIGGDFDLAKARYHKIVIMTDADVDGAHIRTLLLTFFYRFMRPLIEAGYVYIA +QPPLYKLTQGKQKYYVYNDRELDKLKSELNPTPKWSIARYKGLGEMNADQLWETTMNPEH +RALLQVKLEDAIEADQTFEMLMGDVVENRRQFIEDNAVYANLDF +>sp|P67705|Y023_MYCBO Uncharacterized HTH-type transcriptional regulator Mb0023 OS=Mycobacterium bovis (strain ATCC BAA-935 / AF2122/97) OX=233413 GN=BQ2027_MB0023 PE=4 SV=1 +MSRESAGAAIRALRESRDWSLADLAAATGVSTMGLSYLERGARKPHKSTVQKVENGLGLP +PGTYSRLLVAADPDAELARLIAAQPSNPTAVRRAGAVVVDRHSDTDVLEGYAEAQLDAIK +SVIDRLPATTSNEYETYILSVIAQCVKAEMLAASSWRVAVNAGADSTGRLMEHLRALEAT +RGALLERMPTSLSARFDRACAQSSLPEAVVAALIGVGADEMWDIRNRGVIPAGALPRVRA +FVDAIEASHDADEGQQ +>sp|P67925|BLE_GEOSE Bleomycin resistance protein OS=Geobacillus stearothermophilus OX=1422 GN=bleO PE=3 SV=1 +MLQSIPALPVGDIKKSIGFYCDKLGFTLVHHEDGFAVLMCNEVRIHLWEASDEGWRSRSN +DSPVCTGAESFIAGTASCRIEVEGIDELYQHIKPLGILHPNTSLKDQWWDERDFAVIDPD +NNLISFFQQIKS +>sp|P68263|MECI_STAEP Methicillin resistance regulatory protein MecI OS=Staphylococcus epidermidis OX=1282 GN=mecI PE=3 SV=1 +MDNKTYEISSAEWEVMNIIWMKKYASANNIIEEIQMQKDWSPKTIRTLITRLYKKGFIDR +KKDNKIFQYYSLVEESDIKYKTSKNFINKVYKGGFNSLVLNFVEKEDLSQDEIEELRNIL +NKK +>sp|P71590|FHAA_MYCTU FHA domain-containing protein FhaA OS=Mycobacterium tuberculosis (strain ATCC 25618 / H37Rv) OX=83332 GN=fhaA PE=1 SV=1 +MGSQKRLVQRVERKLEQTVGDAFARIFGGSIVPQEVEALLRREAADGIQSLQGNRLLAPN +EYIITLGVHDFEKLGADPELKSTGFARDLADYIQEQGWQTYGDVVVRFEQSSNLHTGQFR +ARGTVNPDVETHPPVIDCARPQSNHAFGAEPGVAPMSDNSSYRGGQGQGRPDEYYDDRYA +RPQEDPRGGPDPQGGSDPRGGYPPETGGYPPQPGYPRPRHPDQGDYPEQIGYPDQGGYPE +QRGYPEQRGYPDQRGYQDQGRGYPDQGQGGYPPPYEQRPPVSPGPAAGYGAPGYDQGYRQ +SGGYGPSPGGGQPGYGGYGEYGRGPARHEEGSYVPSGPPGPPEQRPAYPDQGGYDQGYQQ +GATTYGRQDYGGGADYTRYTESPRVPGYAPQGGGYAEPAGRDYDYGQSGAPDYGQPAPGG +YSGYGQGGYGSAGTSVTLQLDDGSGRTYQLREGSNIIGRGQDAQFRLPDTGVSRRHLEIR +WDGQVALLADLNSTNGTTVNNAPVQEWQLADGDVIRLGHSEIIVRMH +>sp|P77307|FETB_ECOLI Probable iron export permease protein FetB OS=Escherichia coli (strain K12) OX=83333 GN=fetB PE=1 SV=2 +MNSHNITNESLALALMLVVVAILISHKEKLALEKDILWSVGRAIIQLIIVGYVLKYIFSV +DDASLTLLMVLFICFNAAWNAQKRSKYIAKAFISSFIAITVGAGITLAVLILSGSIEFIP +MQVIPIAGMIAGNAMVAVGLCYNNLGQRVISEQQQIQEKLSLGATPKQASAILIRDSIRA +ALIPTVDSAKTVGLVSLPGMMSGLIFAGIDPVKAIKYQIMVTFMLLSTASLSTIIACYLT +YRKFYNSRHQLVVTQLKKK +>sp|P82371|SCRK_LACLC Fructokinase OS=Lactococcus lactis subsp. cremoris OX=1359 GN=scrK PE=3 SV=2 +MSVYYGSIEAGGTKFVLAIADEHFNIIKKFKFATTTPQETISKTIKYFKENRVSAIGLGS +FGPIDLNLSSKTYGYITSTPKVGWKNINLVGQLKEALDIPIYFTTDVNASAYGEMKNTGI +KNLVYLTIGTGIGGGAIQNGYFIGGIGHSEMGHQRINRHRDVLTFEGICPFHGDCLEGVA +AGPSLEARTGILGEKISSDDPIWDILSYYIAQAAINATLTLAPECIILGGGVMEKPNMIS +LIQKQFISMLNNYIDLPCSVEKYIRLPTVKENGSATLGNFYLAYSLFTKE +>sp|P94604|GYRB_CLOAB DNA gyrase subunit B OS=Clostridium acetobutylicum (strain ATCC 824 / DSM 792 / JCM 1419 / IAM 19013 / LMG 5710 / NBRC 13948 / NRRL B-527 / VKM B-1787 / 2291 / W) OX=272562 GN=gyrB PE=3 SV=2 +MSEKNEIYDESQIQVLEGLEAVRKRPGMYIGSTGTRGLHHLVYEIVDNSIDEALAGYCSH +IKVFIHKDNSVTVSDDGRGMPIGIHHKMKKPTVEVIMTILHAGGKFGGGAYRVSGGLHGV +GASVVNALSETCEVEVKTEGHIWKQTYHRGKVASPFEKIGDSDEHGTKIYFKPDPEIFED +TEYDYDTLSQRLRELAFLNKGIKIELTDERHDKNEIFHYEGGLKSFVSYLNRNKEVVFKE +PIYVEGSIDSNYSVEIALQYNDGYNENIFSFANNIHTIEGGTHLAGFKTALTRVINDYAK +KFGYLKENDKNLSGEDIREGLTAVVSVKLTEPQFEGQTKTKLGNTEVRGIVDSIISERVS +TYLEENPQIGKLVIDKALVASRAREAAKKAREITRRKSVLESTSLPGKLADCSSKDAEEC +EIYIVEGDSAGGSAKQGRNRRFQAILPLRGKIMNVEKQRIDKILNSEEIKAMATAFGGGI +GKDFDVSKLRYHKIIIMTDADVDGAHIRTLILTFFYRYMTELISEGHVFIAQPPLYKVTK +TRKEYYAYSDKELEDVLQDVGGKDKNTDIQRYKGLGEMNPEQLWETTMNPEQRTLIKVNI +EDAMAADEIFTILMGDKVDPRRKFIEENATKVVNLDV +>sp|P94605|GYRA_CLOAB DNA gyrase subunit A OS=Clostridium acetobutylicum (strain ATCC 824 / DSM 792 / JCM 1419 / IAM 19013 / LMG 5710 / NBRC 13948 / NRRL B-527 / VKM B-1787 / 2291 / W) OX=272562 GN=gyrA PE=3 SV=2 +MLNEGKVLPVDISSEMKKCYIDYAMSVIVSRALPDVRDGLKPVHRRILYSMHELGLTPEK +GYRKCARIVGDVLGKYHPHGDSSVYGALVRLAQDFNLRYTVVDGHGNFGSVDGDSAAAMR +YTEAKMNKIALEMVRDIGKNTVDFIPNFDGEEKEPVVLPSRFPNLLVNGSAGIAVGMATN +IPPHNLGEVIDGITMLIDNPEATILELMAQIKGPDFPTAGIIMGKSGIRAAYETGRGKIT +VRAKSEIEVEDNGKQKIIITEIPYQVNKARLVESIADLVKDKRIVGISDLRDESDRDGMR +IVIEIKKDANSNIILNQLYKHTRLQDTFGINMLALVDNRPEVLNLKQILQHYIKFQEQVI +RRRTEFDLEKASARAHILEGLKIALDHIDEVISLIRGSKTAQEAKLGLMDKFGLSEKQAQ +AILDLKLQRLTGLEREKIEDEYNELMKTIAELKSILADENKILAIIRDELNEIKAKYGDE +RKTAIERAENDIDIEDLIQEENVVITLTHAGYIKRINADTYTSQKRGGRGIQAMTTKEDD +FVENIFITSTHNNILFFTNKGRMYKLKAYQIPEAGRAAKGTNVVNLLQLDPNEKIQAVIS +IKEFDEESFLVMCTKKGIIKKTVVGMYKNIRKSGLIAINLNDDDELVSVRITKGDDDIII +VTNKGLAIRFNEVDVRPLGRNALGVKGITLKEDDFVVGMEVPNQESDVLVVSENGFGKRT +HVGEYKCQHRGGKGLITYKVSDKTGKLVGVRMVEDGDELMLINNLGIAIRINVSDISTTS +RNAMGVTLMRNNGDEKVLALAKINKDDSEQLEDSEEVSEVHDAEENNSEE +>sp|P96670|YDEM_BACSU Uncharacterized protein YdeM OS=Bacillus subtilis (strain 168) OX=224308 GN=ydeM PE=4 SV=1 +MKLDEFTIGQVFKTKSLKVSKDDIMRFAGEFDPQYMHVDEEKASKGRFNGIIASGIQTLA +ISFKLWIEEGFYGDDIIAGTEMNHMTFIKPVYPDDELFTIVEVLDKQPKRNELGILTVLL +STYNQKEVKVFEGELSVLIKR +>sp|P96726|YWQN_BACSU Putative NAD(P)H-dependent FMN-containing oxidoreductase YwqN OS=Bacillus subtilis (strain 168) OX=224308 GN=ywqN PE=1 SV=1 +MKIAVINGGTRSGGNTDVLAEKAVQGFDAEHIYLQKYPIQPIEDLRHAQGGFRPVQDDYD +SIIERILQCHILIFATPIYWFGMSGTLKLFIDRWSQTLRDPRFPDFKQQMSVKQAYVIAV +GGDNPKIKGLPLIQQFEHIFHFMGMSFKGYVLGEGNRPGDILRDHQALSAASRLLKRSDA +I +>sp|P99178|SYS_STAAN Serine--tRNA ligase OS=Staphylococcus aureus (strain N315) OX=158879 GN=serS PE=1 SV=1 +MLDIRLFRNEPDTVKSKIELRGDDPKVVDEILELDEQRRKLISATEEMKARRNKVSEEIA +LKKRNKENADDVIAEMRTLGDDIKEKDSQLNEIDNKMTGILCRIPNLISDDVPQGESDED +NVEVKKWGTPREFSFEPKAHWDIVEELKMADFDRAAKVSGARFVYLTNEGAQLERALMNY +MITKHTTQHGYTEMMVPQLVNADTMYGTGQLPKFEEDLFKVEKEGLYTIPTAEVPLTNFY +RNEIIQPGVLPEKFTGQSACFRSEAGSAGRDTRGLIRLHQFDKVEMVRFEQPEDSWNALE +EMTTNAEAILEELGLPYRRVILCTGDIGFSASKTYDIEVWLPSYNDYKEISSCSNCTDFQ +ARRANIRFKRDKAAKPELAHTLNGSGLAVGRTFAAIVENYQNEDGTVTIPEALVPFMGGK +TQISKPVK +>sp|P9WG46|GYRA_MYCTO DNA gyrase subunit A OS=Mycobacterium tuberculosis (strain CDC 1551 / Oshkosh) OX=83331 GN=gyrA PE=3 SV=1 +MTDTTLPPDDSLDRIEPVDIQQEMQRSYIDYAMSVIVGRALPEVRDGLKPVHRRVLYAMF +DSGFRPDRSHAKSARSVAETMGNYHPHGDASIYDSLVRMAQPWSLRYPLVDGQGNFGSPG +NDPPAAMRYTEARLTPLAMEMLREIDEETVDFIPNYDGRVQEPTVLPSRFPNLLANGSGG +IAVGMATNIPPHNLRELADAVFWALENHDADEEETLAAVMGRVKGPDFPTAGLIVGSQGT +ADAYKTGRGSIRMRGVVEVEEDSRGRTSLVITELPYQVNHDNFITSIAEQVRDGKLAGIS +NIEDQSSDRVGLRIVIEIKRDAVAKVVINNLYKHTQLQTSFGANMLAIVDGVPRTLRLDQ +LIRYYVDHQLDVIVRRTTYRLRKANERAHILRGLVKALDALDEVIALIRASETVDIARAG +LIELLDIDEIQAQAILDMQLRRLAALERQRIIDDLAKIEAEIADLEDILAKPERQRGIVR +DELAEIVDRHGDDRRTRIIAADGDVSDEDLIAREDVVVTITETGYAKRTKTDLYRSQKRG +GKGVQGAGLKQDDIVAHFFVCSTHDLILFFTTQGRVYRAKAYDLPEASRTARGQHVANLL +AFQPEERIAQVIQIRGYTDAPYLVLATRNGLVKKSKLTDFDSNRSGGIVAVNLRDNDELV +GAVLCSADDDLLLVSANGQSIRFSATDEALRPMGRATSGVQGMRFNIDDRLLSLNVVREG +TYLLVATSGGYAKRTAIEEYPVQGRGGKGVLTVMYDRRRGRLVGALIVDDDSELYAVTSG +GGVIRTAARQVRKAGRQTKGVRLMNLGEGDTLLAIARNAEESGDDNAVDANGADQTGN +>sp|P9WHW4|PSTP_MYCTO Serine/threonine protein phosphatase PstP OS=Mycobacterium tuberculosis (strain CDC 1551 / Oshkosh) OX=83331 GN=pstP PE=3 SV=1 +MARVTLVLRYAARSDRGLVRANNEDSVYAGARLLALADGMGGHAAGEVASQLVIAALAHL +DDDEPGGDLLAKLDAAVRAGNSAIAAQVEMEPDLEGMGTTLTAILFAGNRLGLVHIGDSR +GYLLRDGELTQITKDDTFVQTLVDEGRITPEEAHSHPQRSLIMRALTGHEVEPTLTMREA +RAGDRYLLCSDGLSDPVSDETILEALQIPEVAESAHRLIELALRGGGPDNVTVVVADVVD +YDYGQTQPILAGAVSGDDDQLTLPNTAAGRASAISQRKEIVKRVPPQADTFSRPRWSGRR +LAFVVALVTVLMTAGLLIGRAIIRSNYYVADYAGSVSIMRGIQGSLLGMSLHQPYLMGCL +SPRNELSQISYGQSGGPLDCHLMKLEDLRPPERAQVRAGLPAGTLDDAIGQLRELAANSL +LPPCPAPRATSPPGRPAPPTTSETTEPNVTSSPASPSPTTSASAPTGTTPAIPTSASPAA +PASPPTPWPVTSSPTMAALPPPPPQPGIDCRAAA +>sp|P9WJB4|FHAB_MYCTO FHA domain-containing protein FhaB OS=Mycobacterium tuberculosis (strain CDC 1551 / Oshkosh) OX=83331 GN=fhaB PE=3 SV=1 +MQGLVLQLTRAGFLMLLWVFIWSVLRILKTDIYAPTGAVMMRRGLALRGTLLGARQRRHA +ARYLVVTEGALTGARITLSEQPVLIGRADDSTLVLTDDYASTRHARLSMRGSEWYVEDLG +STNGTYLDRAKVTTAVRVPIGTPVRIGKTAIELRP +>sp|P9WJF2|CWSA_MYCTO Cell wall synthesis protein CwsA OS=Mycobacterium tuberculosis (strain CDC 1551 / Oshkosh) OX=83331 GN=cwsA PE=3 SV=1 +MSEQVETRLTPRERLTRGLAYSAVGPVDVTRGLLELGVGLGLQSARSTAAGLRRRYREGR +LAREVAAAQETLAQELTAAQDVVANLPQALQDARTQRRSKHHLWIFAGIAAAILAGGAVA +FSIVRRSSRPEPSPRPPSVEVQPRP +>sp|P9WKD0|PBPA_MYCTO Peptidoglycan D,D-transpeptidase PbpA OS=Mycobacterium tuberculosis (strain CDC 1551 / Oshkosh) OX=83331 GN=pbpA PE=3 SV=1 +MNASLRRISVTVMALIVLLLLNATMTQVFTADGLRADPRNQRVLLDEYSRQRGQITAGGQ +LLAYSVATDGRFRFLRVYPNPEVYAPVTGFYSLRYSSTALERAEDPILNGSDRRLFGRRL +ADFFTGRDPRGGNVDTTINPRIQQAGWDAMQQGCYGPCKGAVVALEPSTGKILALVSSPS +YDPNLLASHNPEVQAQAWQRLGDNPASPLTNRAISETYPPGSTFKVITTAAALAAGATET +EQLTAAPTIPLPGSTAQLENYGGAPCGDEPTVSLREAFVKSCNTAFVQLGIRTGADALRS +MARAFGLDSPPRPTPLQVAESTVGPIPDSAALGMTSIGQKDVALTPLANAEIAATIANGG +ITMRPYLVGSLKGPDLANISTTVGYQQRRAVSPQVAAKLTELMVGAEKVAQQKGAIPGVQ +IASKTGTAEHGTDPRHTPPHAWYIAFAPAQAPKVAVAVLVENGADRLSATGGALAAPIGR +AVIEAALQGEP +>sp|P9WMA1|Y025_MYCTU Uncharacterized protein Rv0025 OS=Mycobacterium tuberculosis (strain ATCC 25618 / H37Rv) OX=83332 GN=Rv0025 PE=1 SV=1 +MSEQAGSSVAVIQERQALLARQHDAVAEADRELADVLASAHAAMRESVRRLDAIAAELDR +AVPDQDQLAVDTPMGAREFQTFLVAKQREIVAVVAAAHELDRAKSAVLKRLRAQYTEPAR +>sp|P9WMA2|Y010_MYCTO Uncharacterized protein MT0013 OS=Mycobacterium tuberculosis (strain CDC 1551 / Oshkosh) OX=83331 GN=MT0013 PE=4 SV=1 +MQQTAWAPRTSGIAGCGAGGVVMAIASVTLVTDTPGRVLTGVAALGLILFASATWRARPR +LAITPDGLAIRGWFRTQLLRHSNIKIIRIDEFRRYGRLVRLLEIETVSGGLLILSRWDLG +TDPVEVLDALTAAGYAGRGQR +>sp|P9WMA7|Y007_MYCTU Uncharacterized protein Rv0007 OS=Mycobacterium tuberculosis (strain ATCC 25618 / H37Rv) OX=83332 GN=Rv0007 PE=1 SV=1 +MTAPNEPGALSKGDGPNADGLVDRGGAHRAATGPGRIPDAGDPPPWQRAATRQSQAGHRQ +PPPVSHPEGRPTNPPAAADARLNRFISGASAPVTGPAAAVRTPQPDPDASLGCGDGSPAE +AYASELPDLSGPTPRAPQRNPAPARPAEGGAGSRGDSAAGSSGGRSITAESRDARVQLSA +RRSRGPVRASMQIRRIDPWSTLKVSLLLSVALFFVWMITVAFLYLVLGGMGVWAKLNSNV +GDLLNNASGSSAELVSSGTIFGGAFLIGLVNIVLMTALATIGAFVYNLITDLIGGIEVTL +ADRD +>sp|P9WMB1|Y0026_MYCTU Uncharacterized protein Rv0026 OS=Mycobacterium tuberculosis (strain ATCC 25618 / H37Rv) OX=83332 GN=Rv0026 PE=4 SV=1 +MAFDAAMSTHEDLLATIRYVRDRTGDPNAWQTGLTPTEVTAVVTSTTRSEQLDAILRKIR +QRHSNLYYPAPPDREQGDAARAIADAEAALAHQNSATAQLDLQVVSAILNAHLKTVEGGE +SLHELQQEIEAAVRIRSDLDTPAGARDFQRFLIGKLKDIREVVATASLDAASKSALMAAW +TSLYDASKGDRGDADDRGPASVGSGGAPARGAGQQPELPTRAEPDCLLDSLLLEDPGLLA +DDLQVPGGTSAAIPSASSTPSLPNLGGATMPGGGATPALVPGVSAPGGLPLSGLLRGVGD +EPELTDFDERGQEVRDPADYEHSNEPDERRADDREGADEDAGLGKSESPPQAPTTVTLPN +GETVTAASPQLAAAIKAAASGTPIADAFQQQGIAIPLPGTAVANPVDPARISAGDVGVFT +ATPLPLALAKLFWTARFNTSQPCEGQTF +>sp|P9WN34|TRPG_MYCTO Anthranilate synthase component 2 OS=Mycobacterium tuberculosis (strain CDC 1551 / Oshkosh) OX=83331 GN=trpG PE=3 SV=1 +MRILVVDNYDSFVFNLVQYLGQLGIEAEVWRNDDHRLSDEAAVAGQFDGVLLSPGPGTPE +RAGASVSIVHACAAAHTPLLGVCLGHQAIGVAFGATVDRAPELLHGKTSSVFHTNVGVLQ +GLPDPFTATRYHSLTILPKSLPAVLRVTARTSSGVIMAVQHTGLPIHGVQFHPESILTEG +GHRILANWLTCCGWTQDDTLVRRLENEVLTAISPHFPTSTASAGEATGRTSA +>sp|P9WN99|RODA_MYCTU Peptidoglycan glycosyltransferase RodA OS=Mycobacterium tuberculosis (strain ATCC 25618 / H37Rv) OX=83332 GN=rodA PE=1 SV=1 +MTTRLQAPVAVTPPLPTRRNAELLLLCFAAVITFAALLVVQANQDQGVPWDLTSYGLAFL +TLFGSAHLAIRRFAPYTDPLLLPVVALLNGLGLVMIHRLDLVDNEIGEHRHPSANQQMLW +TLVGVAAFALVVTFLKDHRQLARYGYICGLAGLVFLAVPALLPAALSEQNGAKIWIRLPG +FSIQPAEFSKILLLIFFSAVLVAKRGLFTSAGKHLLGMTLPRPRDLAPLLAAWVISVGVM +VFEKDLGASLLLYTSFLVVVYLATQRFSWVVIGLTLFAAGTLVAYFIFEHVRLRVQTWLD +PFADPDGTGYQIVQSLFSFATGGIFGTGLGNGQPDTVPAASTDFIIAAFGEELGLVGLTA +ILMLYTIVIIRGLRTAIATRDSFGKLLAAGLSSTLAIQLFIVVGGVTRLIPLTGLTTPWM +SYGGSSLLANYILLAILARISHGARRPLRTRPRNKSPITAAGTEVIERV +>sp|P9WPL1|CP144_MYCTU Cytochrome P450 144 OS=Mycobacterium tuberculosis (strain ATCC 25618 / H37Rv) OX=83332 GN=cyp144 PE=1 SV=1 +MRRSPKGSPGAVLDLQRRVDQAVSADHAELMTIAKDANTFFGAESVQDPYPLYERMRAAG +SVHRIANSDFYAVCGWDAVNEAIGRPEDFSSNLTATMTYTAEGTAKPFEMDPLGGPTHVL +ATADDPAHAVHRKLVLRHLAAKRIRVMEQFTVQAADRLWVDGMQDGCIEWMGAMANRLPM +MVVAELIGLPDPDIAQLVKWGYAATQLLEGLVENDQLVAAGVALMELSGYIFEQFDRAAA +DPRDNLLGELATACASGELDTLTAQVMMVTLFAAGGESTAALLGSAVWILATRPDIQQQV +RANPELLGAFIEETLRYEPPFRGHYRHVRNATTLDGTELPADSHLLLLWGAANRDPAQFE +APGEFRLDRAGGKGHISFGKGAHFCVGAALARLEARIVLRLLLDRTSVIEAADVGGWLPS +ILVRRIERLELAVQ +>sp|Q03928|HSP18_CLOAB 18 kDa heat shock protein OS=Clostridium acetobutylicum (strain ATCC 824 / DSM 792 / JCM 1419 / IAM 19013 / LMG 5710 / NBRC 13948 / NRRL B-527 / VKM B-1787 / 2291 / W) OX=272562 GN=hsp18 PE=2 SV=1 +MFGMVPFRRNNNGLMRREDFFDKMFDNFFSDDFFPTTTFNGNAGFKVDIKEDDDKYTVAA +DLPGVKKDNIELQYENNYLTINAKRDDIVETKDDNNNFVRRERSYGELRRSFYVDNIDDS +KIDASFLDGVLRITLPKKVKGKDNGRRIDIH +>sp|Q03UD8|RS6_LEVBA Small ribosomal subunit protein bS6 OS=Levilactobacillus brevis (strain ATCC 367 / BCRC 12310 / CIP 105137 / JCM 1170 / LMG 11437 / NCIMB 947 / NCTC 947) OX=387344 GN=rpsF PE=3 SV=1 +METTKYEITYIIRPDLDDAAKTALVERFDKILTDNGAELINSKDWSKRRFAYEIGGFNEG +IYHVITLNATDDKGLNEFDRLAKINDSILRHMIVKRED +>sp|Q03UE4|DNAA_LEVBA Chromosomal replication initiator protein DnaA OS=Levilactobacillus brevis (strain ATCC 367 / BCRC 12310 / CIP 105137 / JCM 1170 / LMG 11437 / NCIMB 947 / NCTC 947) OX=387344 GN=dnaA PE=3 SV=1 +MPDMLTLWTDIKALFEENNSKTAYATWIETAKPIALDGNKLTLELPSPLHRDYWTHQHLD +QQLVEYAYQAAHEDIQPVLILENERQQQATLKAKTAPVAAGEPVEPTPTFMKETALNSRY +TFDTFVIGKGNQMAHAAALVVSEEPGVMYNPLFFYGGVGLGKTHLMHAIGNKMLEDRPDT +KVKYVTSEAFTNDFINAIQTRTQEQFRQEYRNVDLLLVDDIQFFANKEGTQEEFFHTFNA +LYDDGKQIVLTSDRLPNEIPKLQDRLVSRFAWGLSVDITPPDLETRIAILRNKADADQID +IPDDTLSYIAGQIDSNVRELEGALARVQAYSQLMHQPIATDLAAEALKSLNLANASDAVT +IPVIQDRVAKYFDVSLKDLKGKKRKKAIVMPRQIAMYLSRELTEASLPRIGNEFGGKDHT +TVIHAYDKITESLKTDPQLQKDIDSLKDDLRR +>sp|Q04CW7|RS18_LACDB Small ribosomal subunit protein bS18 OS=Lactobacillus delbrueckii subsp. bulgaricus (strain ATCC BAA-365 / Lb-18) OX=321956 GN=rpsR PE=3 SV=1 +MPQQRKGGRRRRKVDLIAANHIDYVDYKDVDLLKHFISERGKILPRRVTGTSAKNQRKVA +NAIKRARIMGLLPFVAED +>sp|Q04CX5|DNAA_LACDB Chromosomal replication initiator protein DnaA OS=Lactobacillus delbrueckii subsp. bulgaricus (strain ATCC BAA-365 / Lb-18) OX=321956 GN=dnaA PE=3 SV=1 +MFDLEKFWDSFNAEMRSEFNEVSYNAWFKNTKPVSFNKDTHELVISVQTPVAKGYWEQNI +SANLIQSAYAYAGIDIYPVFVVKNGPTPSSERMLEPQPQAKPEKARPQGREFTKDLRLNE +KYTFENFIQGEGNKLAAGAALAVADNPGTFYNPLFIFGGVGLGKTHLMQAIGHQMLAERP +DAKVVYIQSETFVNDFINSIKNKTQDKFREKYRTADLLLVDDIQFFAKKEGIQEEFFHTF +ETLYNDQKQIVMTSDRLPTEIPDLSERLVSRFAWGLQVEITPPDLETRIAILRKKAESEG +LEIDESTLDYVASQVDTNIRELEGALVKVQAQATIQKQDINIGLARSALADLKLVQKSRG +LQISKIQEVVANYFQTSVPDLKGKKRVRQIVIPRQIAMYLSRELTDASLPKIGQEFGGKD +HTTVMHACDKIARQIKTDTEIKSAVSDLRQMLER +>sp|Q0AU85|METN_SYNWW Methionine import ATP-binding protein MetN OS=Syntrophomonas wolfei subsp. wolfei (strain DSM 2245B / Goettingen) OX=335541 GN=metN PE=3 SV=1 +MIEISNLTKIYGSGPQEVMALKEVSLSIRKGEIFGIIGLSGAGKSTLIRCINMLERPTQG +TIMVDGQDVGSLNSLELRRLRQKIGMIFQHFNLLSSRTVFENVMFPLEIAAVPKNEAAQK +VNALLELVGLTDKAQVYPEQLSGGQKQRVGIARALANDPKVLLSDEATSALDPQTTRSIL +GLLKDINRKLGLTIVLITHDMNVIKDACDRVAVIDDSTIVEVGDVLNTFSNPGTPTSRSF +INSIINREIPAEILHRQVIDHGSNSSRLIRVSFIGPSAGEPIISSMVQKYAIAANILYGN +IDQVKDIPFGNLTLELIGPVNTINEALDFLRKCGLEIEVLNHAGN +>sp|Q0TTK8|FTSH_CLOP1 ATP-dependent zinc metalloprotease FtsH OS=Clostridium perfringens (strain ATCC 13124 / DSM 756 / JCM 1290 / NCIMB 6125 / NCTC 8237 / Type A) OX=195103 GN=ftsH PE=3 SV=1 +MFKDKKMLKYIVIYSIIAFGILLTFNMVKDEMLYEKVDYSTFMQMLDKKEVKSVNFSGNQ +IEITPSDSSNLKGKILYTTNPAVAGITQPELIKDLTVAGVEFNVTKPENYQLLGLLMSWV +FPLILIFFVGRMMFSKMNNKMGGGVMSFGKNNAKLYAENETGITFKDVAGQDEAKESLVE +IVDFLHDTRKYVEIGAKLPKGALLVGPPGTGKTLLAKAVAGEAKVPFFSMSGSDFVEMFV +GMGAARVRDLFKQAEEKAPCIVFIDEIDAIGKSRDGAIQGNDEREQTLNQLLTEMDGFDS +SKGVVILAATNRPEVLDKALLRPGRFDRRIIVDRPDLIGREEILKVHSRDVKLSDDVSLE +EIAKSTPGAVGADLANIVNEAALRAVKHGRKFVIQEDLDEAVEVIIAGQEKRDRILSPKE +KKIVAYHEVGHALVAALLNNTDPVHKITIVPRTMGALGYTMQLPEEEKYLVSKEEMIDQI +SVMLGGRAAEEVVFNSITTGASNDIERATQSARNMITIYGMSERFDMMALEAMSNRYLDG +RPVRNCSETTAAIADEEVLQVIKKAHEKSIKILIENRELLDEITGVLLDKETIMGDEFME +IVYGKYPEKREADEKAKKEIQSLREQALAKRKEKEEAIKKAREEALRLEEEQRKQDELKA +MIEAQEEAAKLARANNEANNDALDSSKENEEVKSNVNDGATEEKKDDSSTNNKVDGE +>sp|Q13SP0|MNMG_PARXL tRNA uridine 5-carboxymethylaminomethyl modification enzyme MnmG OS=Paraburkholderia xenovorans (strain LB400) OX=266265 GN=mnmG PE=3 SV=1 +MLFPTEFDVIVVGGGHAGTEAALASARMGNTTLLLTHNIETLGQMSCNPSIGGIGKGHLV +KEVDALGGAMAAATDEGGIQFRILNSSKGPAVRATRAQADRLLYKQAIRHRLENQPNLWL +FQQAVDDLMVEGDRVVGAVTQVGIRFRGRAVVLTAGTFLDGKIHVGLNNYTGGRAGDPAA +VSLSARLKELKLPQGRLKTGTPPRIDGRTIDFSQLEEQPGDLDPVPVFSFLGRVEQHPRQ +VPCWVTHTNARTHDIIRGGLDRSPMYTGVIEGVGPRYCPSIEDKIHRFASKESHQIFLEP +EGLTTNEFYPNGISTSLPFDVQLELVRSMRGLEHAHILRPGYAIEYDYFDPRGLKASLET +KVISGLFFAGQINGTTGYEEAAAQGLLAGINAGLYVQGKEAWCPRRDQAYLGVLVDDLVT +RGVSEPYRMFTSRAEYRLSLREDNADMRLTEIGRELGVVDDVRWDAFSRKRDAVSRETER +LRTTWVNPKTLSADEATALLGKPIDHEYSLADLLRRPGVSYDGVCALRAGACAAPETLAE +DDVLLAQIKEQIEIGIKYQGYIDRQAGEIERNEAHESTRLPEGLDYAEVRGLSFEARQKL +TQFRPETIGQASRISGITPAAISLLMVHLKRGLGRRPTKPAESGTDSAPVTQ +>sp|Q1GC33|RL9_LACDA Large ribosomal subunit protein bL9 OS=Lactobacillus delbrueckii subsp. bulgaricus (strain ATCC 11842 / DSM 20081 / BCRC 10696 / JCM 1002 / NBRC 13953 / NCIMB 11778 / NCTC 12712 / WDCM 00102 / Lb 14) OX=390333 GN=rplI PE=3 SV=1 +MKVIFMQDVKGRGKLGQVKDVPNGYAQNYLIKQGLAKEANKGNLNTLKRVEANEKAEYEA +QKAAAQEIKKQLEADETVVELKAKAGSDSRLFGSISSKKIIEGLDKQFGIKLDKHKLELR +EPIKVLGYTNVPVKLFKGVESKVRVHVTQEN +>sp|Q1GC40|RECF_LACDA DNA replication and repair protein RecF OS=Lactobacillus delbrueckii subsp. bulgaricus (strain ATCC 11842 / DSM 20081 / BCRC 10696 / JCM 1002 / NBRC 13953 / NCIMB 11778 / NCTC 12712 / WDCM 00102 / Lb 14) OX=390333 GN=recF PE=3 SV=1 +MYLSRFKQSGFRNLAPLNLEFDPHVNVFLGENAQGKTNLLEAIYFLAISRSHRTSNDREM +IAFGQDFASLAGRVHKRQLDLDLRIVISKKGKSAWVNRVEQARLSKYVGHLNAILFSPED +MELVKGAPSLRRRFMDLEFGQINPEYLYFASQYRQLLQQRNNYLKQLARRQASDQVLLGV +LTEQVATAASELIWRRYRYLADLNRYAAEAYRAISGQREELRVLYRPSAKEITAADQPAQ +IKQKLLDRFAEIADDELRRATTQLGPHRDDLEFQLDGKNAHLFASQGQQRTIALSLKLAE +IQLIKQLTGEEPILLLDDVMSELDQNRQAALLNFIHGQTQTFITTTDLDSISQEIVKQPR +IFYIHSGQIIEKEEGLNGRRR +>sp|Q1LI29|EFG1_CUPMC Elongation factor G 1 OS=Cupriavidus metallidurans (strain ATCC 43123 / DSM 2839 / NBRC 102507 / CH34) OX=266264 GN=fusA1 PE=3 SV=1 +MARKTPIERYRNIGISAHIDAGKTTTTERILFYTGVNHKIGEVHDGAATMDWMEQEQERG +ITITSAATTAFWKGMGGNYPEHRFNIIDTPGHVDFTIEVERSMRVLDGACMVYCAVGGVQ +PQSETVWRQANKYGVPRLAFVNKMDRTGANFFKVYDQLKTRLKANPVPVVVPIGAEDGFQ +GVVDLLEMKAIVWDEASQGVKFEYQDIPAELQATADEWREKMVESAAEASEELMEKYLGG +EELTRAEIVKALRDRTIACEIQPMLCGTAFKNKGVQRMLDAVIDFLPSPVDIPPVKGVDE +SDDEKKLERKADDNEKFSALAFKIMTDPFVGQLIFFRVYSGKINSGDTVYNPVKQKKERL +GRILQMHANQREEIKEVLAGDIAAAVGLKDATTGDTLCDPAAPIVLERMVFPEPVISQAV +EPKTKADQEKMGIALNRLAAEDPSFRVRTDEESGQTIISGMGELHLEILVDRMKREFGVE +ANIGAPQVAYRETIRKKAEDVEGKFVKQSGGRGQYGHAVITLEPQEPGKGFEFIDAIKGG +VIPREYIPAVEKGIVDTLPAGILAGFPVVDVKVTLTFGSYHDVDSNENAFRMAGSMAFKE +AMRKASPVLLEPMMAVEVETPEDYTGTVMGDLSSRRGIVQGMDDMVGGGKIIKAEVPLSE +MFGYSTALRSATQGRATYTMEFKHYAEAPKNIAEAVMTAKGKQ +>sp|Q1WVN7|RS18_LIGS1 Small ribosomal subunit protein bS18 OS=Ligilactobacillus salivarius (strain UCC118) OX=362948 GN=rpsR PE=3 SV=1 +MAQQRRGGRRRRKVDYIAANHIEYIDYKDTDLLRRFISERGKILPRRVTGTSAKNQRKLT +VAIKRARIMGLLPFVAED +>sp|Q1WVP2|RECF_LIGS1 DNA replication and repair protein RecF OS=Ligilactobacillus salivarius (strain UCC118) OX=362948 GN=recF PE=3 SV=1 +MYLEKLELKHFRNYEDVNVAFSPQVNVLIGKNAQGKTNLLESIYVLAMARSHRTSNDREM +VTFKKDAALIRGEVHQRLGNTKLELLISRKGKKAKVNHLEKARLSQYIGQLNVILFAPED +LALVKGAPSVRRRFIDMEFGQIDALYLHTLTEYRAVLRQRNKYLKELQTKKATDKVYLEI +LSEQLSESGSQIIFKRLEFLQELEKYADKLHNQITQGKEHLQFQYESTLKEYQGKSVLEL +KQSLIEQYKTMMDKEIFQGTTLLGPHRDDVRFMLNDKNVQVYGSQGQQRTAALSVKLAEI +DLMKEKTHEYPILLLDDVLSELDGARQTHLLKTIQNKVQTFLTTPGLSDVAQQLINKPKI +FRIDNGKITEENSFTIEEE +>sp|Q1WVQ8|MURE_LIGS1 UDP-N-acetylmuramyl-tripeptide synthetase OS=Ligilactobacillus salivarius (strain UCC118) OX=362948 GN=murE PE=3 SV=1 +MALEITPALSLLDEHHLLKEVVNEQKLTFENVTYDSRKVSDNTLFFCKGNFKPSYLTSAL +EKGATAYVSEQKYAEGMDATGIIVTNVQKAMALLGAAFYGFPQNELFIIAYTGTKGKTTS +AYFAEHILAKATDKKVALFSTIDRVLGNEPDQRFKSDLTTPESLDLFHDMRAAVNNGMTH +LVMEVSSQAYKKNRVYGLTFDVGVFLNISPDHIGRNEHPTFDDYLHCKEQLLVNSKRCVI +NGETAYLRDVYYTAKATTEPEDIYVFARKGAQISDNIPVDIEYQNDFEDLHKSVIEVKGL +SDKAQTLHVDGKYELSVPGDYNEGNATSAIIATLLAGAKGEDARKTLDRVHIPGRMEIIK +TKNNGTIYVDYAHNYASLKALFAFLKQQTHAGRVIAVLGSPGDKGISRRPGFGKAVSEEV +DHVILTTDDPGFEDPMKIAQEIDSYINHDNVKVEFELDREIAIEKAIKMSTNNDIVVLAG +KGEDPYQKIKGVDVPYPSDVNVAKKIVEEL +>sp|Q1WVR1|MSCL_LIGS1 Large-conductance mechanosensitive channel OS=Ligilactobacillus salivarius (strain UCC118) OX=362948 GN=mscL PE=3 SV=1 +MLKEFKEFISRGNVMDLAVGVIIGGAFTAIVNSLVKYIINPFLGLFVGAIDFSDLVFKIG +NATFRVGSFLNAVINFLIIAFVVFLMVKGINKVLRQDKKEEAPAPKDPQLEVLEEIRDSL +KKLDK +>sp|Q1WVT3|MSRB_LIGS1 Peptide methionine sulfoxide reductase MsrB OS=Ligilactobacillus salivarius (strain UCC118) OX=362948 GN=msrB PE=3 SV=1 +MKETKEELRQRIGEEAYQVTQNAATERAFTGKYDEFFEDGIYVDVVSGEPLFSSKDKYNS +GCGWPAFTQPINNRMVTNHEDNSFGMHRVEVRSRQAQSHLGHVFNDGPQDRGGLRYCINS +AALQFIPVAELDEKGYGEYKKLFD +>sp|Q24PG6|SFSA_DESHY Sugar fermentation stimulation protein homolog OS=Desulfitobacterium hafniense (strain Y51) OX=138119 GN=sfsA PE=3 SV=1 +MKYTNIREGQFLSRPNRFIAKVEIDGKEEICHVKNTGRCRELLIPGVTVFLQEADFEHRK +TKYDLIGVRKGNRLINMDSQVPNKVFCEWLEKGYFQELQHIKQEQTFRNSRFDFYLEAGQ +RKIFVEVKGVTLEEEGVALFPDAPTERGVKHLRELSQAVAAGYEAYVVFIIQMKDIHYFT +PNIKTHQAFGDALIQADKQGVKILALDCEVTEDSIEAGDFVTVKLVEG +>sp|Q2FFZ0|TRMB_STAA3 tRNA (guanine-N(7)-)-methyltransferase OS=Staphylococcus aureus (strain USA300) OX=367830 GN=trmB PE=3 SV=1 +MRVRYKPWAEDYLKDHPELVDMDGQHAGKMTEWFDKTQPIHIEIGSGMGQFITTLAAQNP +HINYISMEREKSIVYKVLDKVKEMGLTNLKIICNDAIELNEYFKDGEVSRIYLNFSDPWP +KNRHAKRRLTYHTFLALYQQILNDEGDLHFKTDNRGLFAYSLESMSQFGMYFTKINLNLH +QEDDGSNILTEYEKKFSDKGSRIYRMEAKFHSQK +>sp|Q2FXG9|MNMM_STAA8 tRNA (mnm(5)s(2)U34)-methyltransferase OS=Staphylococcus aureus (strain NCTC 8325 / PS 47) OX=93061 GN=mnmM PE=1 SV=1 +MKLERILPFSKTLIKQHITPESIVVDATCGNGNDTLFLAEQVPEGHVYGFDIQDLALENT +RDKVKDFNHVSLIKDGHENIEHHINDAHKGHIDAAIFNLGYLPKGDKSIVTKPDTTIQAI +NSLLSLMSIEGIIVLVIYHGHSEGQIEKHALLDYLSTLDQKHAQVLQYQFLNQRNHAPFI +CAIEKIS +>sp|Q2G2P8|NNRD_STAA8 ADP-dependent (S)-NAD(P)H-hydrate dehydratase OS=Staphylococcus aureus (strain NCTC 8325 / PS 47) OX=93061 GN=nnrD PE=3 SV=2 +MGGYITMETLNSINIPKRKEDSHKGDYGKILLIGGSANLGGAIMLAARACVFSGSGLITV +ATHPTNHSALHSRCPEAMVIDINDTKMLTKMIEMTDSILIGPGLGVDFKGNNAITFLLQN +IQPHQNLIVDGDAITIFSKLKPQLPTCRVIFTPHLKEWERLSGIPIEEQTYERNREAVDR +LGATVVLKKHGTEIFFKDEDFKLTIGSPAMATGGMGDTLAGMITSFVGQFDNLKEAVMSA +TYTHSFIGENLAKDMYVVPPSRLINEIPYAMKQLES +>sp|Q2L284|RL14_BORA1 Large ribosomal subunit protein uL14 OS=Bordetella avium (strain 197N) OX=360910 GN=rplN PE=3 SV=1 +MIQMQTTLDVADNTGARAVMCIKVLGGSKRRYAGIGDIIKVSVKDAAPRGRVKKGEIYNA +VVVRTAKGVRRKDGSLIRFGGNAAVLLNAKLEPIGTRIFGPVTRELRTEKFMKIVSLAPE +VL +>sp|Q2L2H8|RS12_BORA1 Small ribosomal subunit protein uS12 OS=Bordetella avium (strain 197N) OX=360910 GN=rpsL PE=3 SV=1 +MPTISQLVRKPREVSIIKSKSPALENCPQRRGVCTRVYTTTPKKPNSALRKVAKVRLTNG +YEVISYIGGEGHNLQEHSVVLVRGGRVKDLPGVRYHIVRGSLDLQGVKDRKQARSKYGAK +RPKKA +>sp|Q58481|Y1081_METJA UPF0047 protein MJ1081 OS=Methanocaldococcus jannaschii (strain ATCC 43067 / DSM 2661 / JAL-1 / JCM 10045 / NBRC 100440) OX=243232 GN=MJ1081 PE=3 SV=1 +MLVIKMLFKYQIKTNKREELVDITPYIISAISESKVKDGIAVIYVPHTTAGITINENADP +SVKHDIINFLSHLIPKNWNFTHLEGNSDAHIKSSLVGCSQTIIIKDGKPLLGTWQGIFFA +EFDGPRRREFYVKIIGDK +>sp|Q59495|SUCP_LEUME Sucrose phosphorylase OS=Leuconostoc mesenteroides OX=1245 PE=1 SV=1 +MEIQNKAMLITYADSLGKNLKDVHQVLKEDIGDAIGGVHLLPFFPSTGDRGFAPADYTRV +DAAFGDWADVEALGEEYYLMFDFMINHISRESVMYQDFKKNHDDSKYKDFFIRWEKFWAK +AGENRPTQADVDLIYKRKDKAPTQEITFDDGTTENLWNTFGEEQIDIDVNSAIAKEFIKT +TLEDMVKHGANLIRLDAFAYAVKKVDTNDFFVEPEIWDTLNEVREILTPLKAEILPEIHE +HYSIPKKINDHGYFTYDFALPMTTLYTLYSGKTNQLAKWLKMSPMKQFTTLDTHDGIGVV +DARDILTDDEIDYASEQLYKVGANVKKTYSSASYNNLDIYQINSTYYSALGNDDAAYLLS +RVFQVFAPGIPQIYYVGLLAGENDIALLESTKEGRNINRHYYTREEVKSEVKRPVVANLL +KLLSWRNESPAFDLAGSITVDTPTDTTIVVTRQDENGQNKAVLTADAANKTFEIVENGQT +VMSSDNLTQN +>sp|Q59935|MANA_STRMU Mannose-6-phosphate isomerase OS=Streptococcus mutans serotype c (strain ATCC 700610 / UA159) OX=210007 GN=pmi PE=3 SV=2 +MAEPLFLQSQMHKKIWGGNRLRKEFGYDIPSETTGEYWAISAHPNGVSVVKNGVYKGVPL +DELYAEHRELFGNSKSSVFPLLTKILDANDWLSVQVHPDNAYALEHEGELGKTECWYVIS +ADEGAEIIYGHEAKSKEELRQMIAAGDWDHLLTKIPVKAGDFFYVPSGTMHAIGKGIMIL +ETQQSSDTTYRVYDFDRKDDQGRKRALHIEQSIDVLTIGKPANATPAWLSLQGLETTVLV +SSPFFTVYKWQISGSVKMQQTAPYLLVSVLAGQGRITVGLEQYALRKGDHLILPNTIKSW +QFDGDLEIIASHSNEC +>sp|Q5HF24|DAAA_STAAC D-alanine aminotransferase OS=Staphylococcus aureus (strain COL) OX=93062 GN=dat PE=3 SV=1 +MEKIFLNGEFVSPSEAKVSYNDRGYVFGDGIYEYIRVYNGKLFTVTEHYERFLRSANEIG +LDLNYSVEELIELSRKLVDMNQIETGAIYIQATRGVAERNHSFPTPEVEPAIVAYTKSYD +RPYDHLENGVNGVTVEDIRWLRCDIKSLNLLGNVLAKEYAVKYNAVEAIQHRGETVTEGS +SSNAYAIKDGVIYTHPINNYILNGITRIVIKKIAEDYNIPFKEETFTVDFLKNADEVIVS +STSAEVTPVIKLDGEPVNDGKVGPITRQLQEGFEKYIESHSI +>sp|Q5HF39|ACUC_STAAC Acetoin utilization protein AcuC OS=Staphylococcus aureus (strain COL) OX=93062 GN=acuC PE=3 SV=1 +MQQHSSKTAYVYSDKLLQYRFHDQHPFNQMRLKLTTELLLNANLLSPEQIVQPRIATDDE +LMLIHKYDYVEAIKHASHGIISEDEAKKYGLNDEENGQFKHMHRHSATIVGGALTLADLI +MSGKVLNGCHLGGGLHHAQPGRASGFCIYNDIAITAQYLAKEYNQRVLIIDTDAHHGDGT +QWSFYADNHVTTYSIHETGKFLFPGSGHYTERGEDIGYGHTVNVPLEPYTEDASFLECFK +LTVEPVVKSFKPDIILSVNGVDIHYRDPLTHLNCTLHSLYEIPYFVKYLADSYTNGKVIM +FGGGGYNIWRVVPRAWSHVFLSLIDQPIQSGYLPLEWINKWKHYSSELLPKRWEDRLNDY +TYVPRTKEISEKNKKLALHIASWYESTRQ +>sp|Q5HJX8|PURA_STAAC Adenylosuccinate synthetase OS=Staphylococcus aureus (strain COL) OX=93062 GN=purA PE=3 SV=1 +MSSIVVVGTQWGDEGKGKITDFLAEQSDVIARFSGGNNAGHTIQFGGETYKLHLVPSGIF +YKDKLAVIGNGVVVDPVALLKELDGLNERGIPTSNLRISNRAQVILPYHLAQDEYEERLR +GDNKIGTTKKGIGPAYVDKVQRIGIRMADLLEKETFERLLKSNIEYKQAYFKGMFNETCP +SFDDIFEEYYAAGQRLKEFVTDTSKILDDAFVADEKVLFEGAQGVMLDIDHGTYPFVTSS +NPIAGNVTVGTGVGPTFVSKVIGVCKAYTSRVGDGPFPTELFDEDGHHIREVGREYGTTT +GRPRRVGWFDSVVLRHSRRVSGITDLSINSIDVLTGLDTVKICTAYELDGKEITEYPANL +DQLKRCKPIFEELPGWTEDVTNVRTLEELPENARKYLERISELCNVQISIFSVGPDREQT +NLLKELW +>sp|Q5HK03|GYRB_STAEQ DNA gyrase subunit B OS=Staphylococcus epidermidis (strain ATCC 35984 / DSM 28319 / BCRC 17069 / CCUG 31568 / BM 3577 / RP62A) OX=176279 GN=gyrB PE=3 SV=1 +MVNTLSDVNNTDNYGAGQIQVLEGLEAVRKRPGMYIGSTSERGLHHLVWEIVDNSIDEAL +AGYASHIEVVIEKDNWIKVTDNGRGIPVDIQEKMGRPAVEVILTVLHAGGKFGGGGYKVS +GGLHGVGSSVVNALSQDLEVYVHRNGTIYHQAYKQGVPQFDLKEIGDTDKTGTAIRFKAD +KEIFTETTVYNYETLQKRIRELAFLNKGIQITLKDEREEEVREDSYHYEGGIKSYVDLLN +ENKEPLHDEPIYIHQSKDDIEVEIALQYNSGYATNLLTYANNIHTYEGGTHEDGFKRALT +RVLNSYGTQSKIIKEDKDRLSGEDTREGLTAVVSIKHGDPQFEGQTKTKLGNSEVRQVVD +RLFSEHFERFLYENPSVGRIIVEKGIMASRARVAAKKAREVTRRKSALDVSSLPGKLADC +SSKNPEESEIFLVEGDSAGGSTKSGRDSRTQAILPLRGKILNVEKARLDRILNNNEIRQM +ITAFGTGIGGEFDISKARYHKIVIMTDADVDGAHIRTLLLTFFYRFMRPLIEAGYVYIAQ +PPLYKLTQGKQKYYVFNDRELDKLKQELNPSPKWSIARYKGLGEMNADQLWETTMNPEHR +SMLQVRLEDAIDADQTFEMLMGDVVENRRQFIEDNAVYANLDF +>sp|Q5XD45|Y533_STRP6 Putative phosphatase M6_Spy0533 OS=Streptococcus pyogenes serotype M6 (strain ATCC BAA-946 / MGAS10394) OX=286636 GN=M6_Spy0533 PE=1 SV=1 +MSIKLVAVDIDGTLLTDDRRITDDVFQAVQEAKAQGVHVVIATGRPIAGVISLLEQLELN +HKGNHVITFNGGLVQDAETGEEIVKELMTYDDYLETEFLSRKLGVHMHAITKEGIYTANR +NIGKYTVHESTLVNMPIFYRTPEEMTNKEIIKMMMIDEPDLLDAAIKQIPQHFFDKYTIV +KSTPFYLEFMPKTVSKGNAIKHLAKKLGLDMSQTMAIGDAENDRAMLEVVANPVVMENGV +PELKKIAKYITKSNNDSGVAHAIRKWVLN +>sp|Q65G33|ACUA_BACLD Acetoin utilization protein AcuA OS=Bacillus licheniformis (strain ATCC 14580 / DSM 13 / JCM 2505 / CCUG 7422 / NBRC 12200 / NCIMB 9375 / NCTC 10341 / NRRL NRS-1264 / Gibson 46) OX=279010 GN=acuA PE=1 SV=1 +MEHHKTYHAKELQTEKGSVLIEGPISPEKLAEYEFHDELTAFRPSQKQHEALIEIAGLPE +GRIIIARFRQTIVGYVTYVYPDPLERWSEGNMENLIELGAIEVIPAFRGHSVGKTLLAVS +MMDPQMEKYIIITTEYYWHWDLKGTNKDVWEYRKMMEKMMNAGGLVWFATDDPEISSHPA +NCLMARIGKEVSQESIERFDRLRFHNRFMY +>sp|Q6G8G1|RIBBA_STAAS Riboflavin biosynthesis protein RibBA OS=Staphylococcus aureus (strain MSSA476) OX=282459 GN=ribBA PE=3 SV=1 +MQFDNIDSALMALKNGETIIVVDDENRENEGDLVAVTEWMNDNTINFMAKEARGLICAPV +SKDIAQRLDLVQMVDDNSDIFGTQFTVSIDHVDTTTGISAYERTLTAKKLIDPSSEAKDF +NRPGHLFPLVAQDKGVLARNGHTEAAVDLAKLTGAKPAGVICEIMNDDGTMAKGQDLQKF +KEKHQLKMITIDDLIEYRKKLEPEIEFKAKVKMPTDFGTFDMYGFKATYTDEEIVVLTKG +AIRQHENVRLHSACLTGDIFHSQRCDCGAQLESSMKYINEHGGMIIYLPQEGRGIGLLNK +LRAYELIEQGYDTVTANLALGFDEDLRDYHIAAQILKYFNIEHINLLSNNPSKFEGLKQY +GIDIAERIEVIVPETVHNHDYMETKKIKMGHLI +>sp|Q6GD38|DUS_STAAS Probable tRNA-dihydrouridine synthase OS=Staphylococcus aureus (strain MSSA476) OX=282459 GN=dus PE=3 SV=1 +MKENFWSELPRPFFILAPMEDVTDIVFRHVVSEAARPDVFFTEFTNTESFCHPEGIHSVR +GRLTFSEDEQPMVAHIWGDKPEQFRETSIQLAKMGFKGIDLNMGCPVANVAKKGKGSGLI +LRPDVAAEIIQATKAGGLPVSVKTRLGYYEIDEWKDWLKHVFEQDIANLSIHLRTRKEMS +KVDAHWELIEAIKNLRDEIAPNTLLTINGDIPDRKTGLELAEKYGIDGVMIGRGIFHNPF +AFEKEPREHTSKELLDLLRLHLSLFNKYEKDEIRQFKSLRRFFKIYVRGIRGASELRHQL +MNTQSIAEARALLDEFEAQMDEDVKIEL +>sp|Q6GKU3|DPO3B_STAAR Beta sliding clamp OS=Staphylococcus aureus (strain MRSA252) OX=282458 GN=dnaN PE=3 SV=1 +MMEFTIKRDYFITQLNDTLKAISPRTTLPILTGIKIDAKEHEVILTGSDSEISIEITIPK +TVDGEDIVNISETGSVVLPGRFFVDIIKKLPGKDVKLSTNEQFQTLITSGHSEFNLSGLD +PDQYPLLPQVSRDDAIQLSVKVLKNVIAQTNFAVSTSETRPVLTGVNWLIQENELICTAT +DSHRLAVRKLQLEDVSENKNVIIPGKALAELNKIMSDNEEDIDIFFASNQVLFKVGNVNF +ISRLLEGHYPDTTRLFPENYEIKLSIDNGEFYHAIDRASLLAREGGNNVIKLSTGDDVVE +LSSTSPEIGTVKEEVDANDVEGGSLKISFNSKYMMDALKAIDNDEVEVEFFGTMKPFILK +PKGDDSVTQLILPIRTY +>sp|Q795R8|YTFP_BACSU Uncharacterized protein YtfP OS=Bacillus subtilis (strain 168) OX=224308 GN=ytfP PE=3 SV=2 +MKQYDVIVIGGGPSGLMAAIAAGEQGAGVLLIDKGNKLGRKLAISGGGRCNVTNRLPVEE +IIKHIPGNGRFLYSAFSEFNNEDIIKFFENLGIQLKEEDHGRMFPVTDKAQSVVDALLNR +LKQLRVTIRTNEKIKSVLYEDGQAAGIVTNNGEMIHSQAVIIAVGGKSVPHTGSTGDGYE +WAEAAGHTITELFPTEVPVTSGEPFIKQKTLQGLSLRDVAVSVLNKKGKPIITHKMDMLF +THFGLSGPAILRCSQFVVKELKKQPQVPIRIDLYPDINEETLFQKMYKELKEAPKKTIKN +VLKPWMQERYLLFLLEKNGISPNVSFSELPKDPFRQFVRDCKQFTVLANGTLSLDKAFVT +GGGVSVKEIDPKKMASKKMEGLYFCGEILDIHGYTGGYNITSALVTGRLAGLNAGQYARS +>sp|Q79GC6|EFTU_BORPA Elongation factor Tu OS=Bordetella parapertussis (strain 12822 / ATCC BAA-587 / NCTC 13253) OX=257311 GN=tuf1 PE=3 SV=1 +MAKGKFERTKPHVNVGTIGHVDHGKTTLTAAITTVLSNKFGGEARGYDQIDAAPEEKARG +ITINTSHVEYETETRHYAHVDCPGHADYVKNMITGAAQMDGAILVVSAADGPMPQTREHI +LLSRQVGVPYIIVFLNKADMVDDAELLELVEMEVRELLSKYDFPGDDTPIVKGSAKLALE +GDKGELGEQAILSLAQALDTYIPTPERAVDGAFLMPVEDVFSISGRGTVVTGRIERGVVK +VGEEIEIVGIKPTVKTTCTGVEMFRKLLDQGQAGDNVGILLRGTKREDVERGQVLAKPGS +INPHTDFTAEVYILSKEEGGRHTPFFNGYRPQFYFRTTDVTGTIDLPADKEMVLPGDNVS +MTVKLLAPIAMEEGLRFAIREGGRTVGAGVVAKIIK +>sp|Q7A0M4|Y1682_STAAW UPF0478 protein MW1682 OS=Staphylococcus aureus (strain MW2) OX=196620 GN=MW1682 PE=3 SV=1 +MDWILPIAGIIAAIAFLILCIGIVAVLNSVKKNLDYVAKTLDGVEGQVQGITRETTDLLH +KVNRLTEDIQGKVDRLNSVVDAVKGIGDSVQTLNSSVDRVTNSITHNISQNEDKISQVVQ +WSNVAMEIADKWQNRHYRRGSANYKANNVATDANHSYTSRVDK +>sp|Q7A514|ROT_STAAN HTH-type transcriptional regulator rot OS=Staphylococcus aureus (strain N315) OX=158879 GN=rot PE=1 SV=2 +MHKLAHTSFGIVGMFVNTCMVAKYVIINWEMFSMKKVNNDTVFGILQLETLLGDINSIFS +EIESEYKMSREEILILLTLWQKGSMTLKEMDRFVEVKPYKRTRTYNNLVELEWIYKERPV +DDERTVIIHFNEKLQQEKVELLNFISDAIASRATAMQNSLNAIIAV +>sp|Q7A522|PEPVL_STAAN Putative dipeptidase SA1572 OS=Staphylococcus aureus (strain N315) OX=158879 GN=SA1572 PE=1 SV=1 +MWKEKVQQYEDQIINDLKGLLAIESVRDDAKASEDAPVGPGPRKALDYMYEIAHRDGFTT +HDVDHIAGRIEAGKGNDVLGILCHVDVVPAGDGWDSNPFEPVVTEDAIIARGTLDDKGPT +IAAYYAIKILEDMNVDWKKRIHMIIGTDEESDWKCTDRYFKTEEMPTLGFAPDAEFPCIH +GEKGITTFDLVQNKLTEDQDEPDYELITFKSGERYNMVPDHAEARVLVKENMTDVIQDFE +YFLEQNHLQGDSTVDSGILVLTVEGKAVHGMDPSIGVNAGLYLLKFLASLNLDNNAQAFV +AFSNRYLFNSDFGEKMGMKFHTDVMGDVTTNIGVITYDNENAGLFGINLRYPEGFEFEKA +MDRFANEIQQYGFEVKLGKVQPPHYVDKNDPFVQKLVTAYRNQTNDMTEPYTIGGGTYAR +NLDKGVAFGAMFSDSEDLMHQKNEYITKKQLFNATSIYLEAIYSLCVEE +>sp|Q7A528|Y1564_STAAN UPF0354 protein SA1564 OS=Staphylococcus aureus (strain N315) OX=158879 GN=SA1564 PE=1 SV=1 +MNTFQMRDKLKERLSHLDVDFKFNREEETLRIYRTDNNKGITIKLNAIVAKYEDKKEKIV +DEIVYYVDEAIAQMADKTLESISSSQIMPVIRATSFDKKTKQGVPFIYDEHTAETAVYYA +VDLGKSYRLIDESMLEDLKLTEQQIREMSLFNVRKLSNSYTTDEVKGNIFYFINSNDGYD +ASRILNTAFLNEIEAQCQGEMLVAVPHQDVLIIADIRNKTGYDVMAHLTMEFFTKGLVPI +TSLSFGYKQGHLEPIFILGKNNKQKRDPNVIQRLEANRRKFNKDK +>sp|Q7TT91|EFTU_BORPE Elongation factor Tu OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=tuf1 PE=3 SV=1 +MAKGKFERTKPHVNVGTIGHVDHGKTTLTAAITTVLSNKFGGEARGYDQIDAAPEEKARG +ITINTSHVEYETETRHYAHVDCPGHADYVKNMITGAAQMDGAILVVSAADGPMPQTREHI +LLSRQVGVPYIIVFLNKADMVDDAELLELVEMEVRELLSKYDFPGDDTPIVKGSAKLALE +GDKGELGEQAILSLAQALDTYIPTPERAVDGAFLMPVEDVFSISGRGTVVTGRIERGVVK +VGEEIEIVGIKPTVKTTCTGVEMFRKLLDQGQAGDNVGILLRGTKREDVERGQVLAKPGS +INPHTDFTAEVYILSKEEGGRHTPFFNGYRPQFYFRTTDVTGTIDLPADKEMVLPGDNVS +MTVKLLAPIAMEEGLRFAIREGGRTVGAGVVAKIIK +>sp|Q7VTA7|RS11_BORPE Small ribosomal subunit protein uS11 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rpsK PE=3 SV=1 +MAKASTSGASRVRKKVKKNVSDGIAHVHASFNNTIITITDRQGNALSWATSGGAGFKGSR +KSTPFAAQVAAETAGRVAMEYGIKTLEVRIKGPGPGRESSVRALNALGIKISSIADITPV +PHNGCRPPKRRRI +>sp|Q7VTA8|RS13_BORPE Small ribosomal subunit protein uS13 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rpsM PE=3 SV=1 +MARIAGINIPPQQHAEIGLTAIFGIGRTRARKICEAAGVPVTKKVKDLTDAELERIREHI +GVFAVEGDLRREVQLSIKRLIDLGTYRGMRHKRGLPVRGQRTRTNARTRKGPRRAAASLK +K +>sp|Q7VTB0|IF12_BORPE Translation initiation factor IF-1 2 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=infA2 PE=3 SV=1 +MSKDDVIQMQGEVLENLPNATFRVKLENGHVVLGHISGKMRMHYIRILPGDKVTVELTPY +DLTRARIVFRSK +>sp|Q7VTB2|RL15_BORPE Large ribosomal subunit protein uL15 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rplO PE=3 SV=1 +MSDIQLNTLKPAEGSKHAKRRVGRGIGSGLGKTAGRGHKGQKSRSGGFHKVGFEGGQMPL +QRRLPKRGFTPLGQHLYAEVRLSELQLMEAEEIDVQALKAAGVVGQSVRYAKVIKSGELS +RKVVLRGITATAGARAAIEAAGGSLA +>sp|Q7VTB7|RS8_BORPE Small ribosomal subunit protein uS8 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rpsH PE=3 SV=1 +MSMSDPIADMLTRIRNAQQVDKTTVTMPASKLKVAIATVLKDEGYIDGYSVKGTQAKPEL +EITLKYYAGRPVIERIERVSRPGLRIYKGRTSIPQVMNGLGVAIVSTSRGVMTDRKARAN +GVGGEVLCYVA +>sp|Q7VTB8|RS14_BORPE Small ribosomal subunit protein uS14 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rpsN PE=3 SV=1 +MAKLSLINRDIKRAKLADKYAAKRAELKAIIDDQSKTDEERYQARLKLQQLPRNANPTRQ +RNRCVVTGRPRGVFRKFGLTRHKLREMAMKGEVPGMTKASW +>sp|Q7VTB9|RL5_BORPE Large ribosomal subunit protein uL5 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rplE PE=3 SV=1 +MSRLQEFYKSKVVADLQAKFGYKCVMEVPRITKITLNMGVSEAVADKKVIEHAVSDLTKI +SGQKPVVTKTRKAIAGFKIRENYPIGCMVTLRGQRMYEFLDRLVAVALPRVRDFRGISGR +AFDGRGNYNIGVKEQIIFPEIEYDKIDALRGLNISITTTAKTDDEAKALLTAFSFPFRN +>sp|Q7VTC0|RL24_BORPE Large ribosomal subunit protein uL24 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rplX PE=3 SV=1 +MNNIRKGDEVIVLTGRDKKRRGTVLARVDADHVLVEGVNVVKKHVKANPMANNPGGIVEK +TMPIHVSNVALFNPATGKGDRVGVQEVDGRKVRVFRSNGAVVGAKA +>sp|Q7VTC4|RS17_BORPE Small ribosomal subunit protein uS17 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rpsQ PE=3 SV=1 +MSETQNTQVAKRQRTLVGKVVSNKMDKTVVVLVERRVKHPIFGKIIMRSAKYKAHDESNQ +YNEGDTVEIAEGRPISRSKAWRVVRLVEAARVI +>sp|Q7VTC6|RL16_BORPE Large ribosomal subunit protein uL16 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rplP PE=3 SV=1 +MLQPSRRKYRKEQKGRNTGLASRGTHVSFGEFGLKATGRGRLTARQIEAARRAINRHIKR +GGRIWIRIFPDKPISQKPAEVRMGNGKGNPEYWVAEIQPGKVLYEMEGVSEELAREAFRL +AAAKLPISTTFVARHIGA +>sp|Q7VTD1|RL23_BORPE Large ribosomal subunit protein uL23 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rplW PE=3 SV=1 +MNAERLMQVILAPIVTEKATFVAEKNQQVAFRVVADATKPEIKAAVELLFKVQVESVQVL +NRKGKVKRFGRFVGRRRNERKAYVALKDGQEIDFAEVK +>sp|Q7VTD4|RS10_BORPE Small ribosomal subunit protein uS10 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rpsJ PE=3 SV=1 +MKNQKIRIRLKAFDYKLIDQSAAEIVDTAKRTGAVVRGPVPLPTRIRRYDVLRSPHVNKT +SRDQFEIRTHQRLMDIVDPTDKTVDALMRLDLPAGVDVEIALQ +>sp|Q7VTD6|RS7_BORPE Small ribosomal subunit protein uS7 OS=Bordetella pertussis (strain Tohama I / ATCC BAA-589 / NCTC 13251) OX=257313 GN=rpsG PE=3 SV=1 +MPRRREVPKREILPDPKFGSVELAKFMNVVMLDGKKAVAERIIYGALEQVQVKTGKDAIE +VFNLAINNIKPIVEVKSRRVGGANYQVPVEVRPVRRLALAMRWLREAAKKRGEKSMDLRL +AGELIDASEGRGAAMKKREDTHKMAEANKAFSHFRW +>sp|Q7W2E0|RL6_BORPA Large ribosomal subunit protein uL6 OS=Bordetella parapertussis (strain 12822 / ATCC BAA-587 / NCTC 13253) OX=257311 GN=rplF PE=3 SV=1 +MSRIAKYPVELPKGVEASIQPDQITVKGPLGTLVQSLTGDVNVAQEDGKLTFVVANDSRH +ANAMSGTLRALVANMVTGVSKGFERKLNLVGVGYRAQIQGDAVKLQLGFSHDVVHQLPAG +VKAECPTQTEIVIKGPNKQVVGQVAAEIRKYREPEPYKGKGVRYADERVVIKETKKK +>sp|Q7W2F1|RL22_BORPA Large ribosomal subunit protein uL22 OS=Bordetella parapertussis (strain 12822 / ATCC BAA-587 / NCTC 13253) OX=257311 GN=rplV PE=3 SV=1 +METTAIIRGVHISAQKTRLVADLIRGKSVAQALNILTFSPKKAAVILKKAVESAIANAEH +NDGADIDELKVTTIFVDKAQSMKRFSARAKGRGNRIEKQTCHITVKVGA +>sp|Q7W2F2|RS19_BORPA Small ribosomal subunit protein uS19 OS=Bordetella parapertussis (strain 12822 / ATCC BAA-587 / NCTC 13253) OX=257311 GN=rpsS PE=3 SV=1 +MSRSIKKGPFVDAHLIKKVDAAVAGKDKKPIKTWSRRSTVLPEFIGLTIAVHNGRQHVPV +YINENMVGHKLGEFALTRTFKGHAADKKSKR +>sp|Q7W2F5|RL4_BORPA Large ribosomal subunit protein uL4 OS=Bordetella parapertussis (strain 12822 / ATCC BAA-587 / NCTC 13253) OX=257311 GN=rplD PE=3 SV=1 +MDLKLLNDQGQAATFSAPDTIFGRDFNEALVHQIVVAFQANARSGNRAQKDRAEVKHSTK +KPWRQKGTGRARAGMTSSPLWRGGGRAFPNSPEENFSQKVNKKMYRAGIRSILSQLARED +RVAVVDTFTLESPKTKLAAAKLKSLGLDSVLIITDNVDENVYLATRNLPHVAVVEPRYAD +PLSLIHYKKVLITKPAIAQLEEMLG +>sp|Q7W2H3|RL11_BORPA Large ribosomal subunit protein uL11 OS=Bordetella parapertussis (strain 12822 / ATCC BAA-587 / NCTC 13253) OX=257311 GN=rplK PE=3 SV=1 +MAKKIVGFIKLQVPAGKANPSPPIGPALGQRGLNIMEFCKAFNAKTQGMEPGLPIPVVIT +AFADKSFTFIMKTPPATVLIKKASGVQKGSAKPHTDKVGTLTRAQAEEIAKTKQPDLTAA +DLDAAVRTIAGSARSMGITVEGV +>sp|Q7WRA5|RL30_BORBR Large ribosomal subunit protein uL30 OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rpmD PE=3 SV=1 +MAQKQIKVTLVRSVIGTKQSHRDTIRGLGLRRINSSRVLVDTPEVRGMIHKVGYLVSVSE +A +>sp|Q7WRA6|RS5_BORBR Small ribosomal subunit protein uS5 OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rpsE PE=3 SV=1 +MATKVQGKHAAEKENDDGLREKMIAVNRVSKVVKGGRTMSFAALSVVGDGDGRIGMGKGK +AREVPVSVQKAMEQARRGMFKVALKNGTLHHTVVGKHGASTVLISPAAEGTGVIAGGPMR +AIFEVMGVRNVVAKSLGSSNPYNLVRATLNGLRASLTPAEVAAKRGKSVEEILG +>sp|Q7WRA7|RL18_BORBR Large ribosomal subunit protein uL18 OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rplR PE=3 SV=1 +MDKKVSRLRRAVPTRRKIAQLRVHRLSVFRSNLHIYANIISPEGDRVLVSASTLEAEVRA +QLGGAGKGGNATAAALVGKRVAEKAKAAGIELVAFDRSGFRYHGRVKALAEAAREAGLKF +>sp|Q7WRB7|RL29_BORBR Large ribosomal subunit protein uL29 OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rpmC PE=3 SV=1 +MKASELRSKDAAELGKELESLLKAQFGLRMQKATQQLANTSQLRNVRRDIARVRTLLTEK +AGK +>sp|Q7WRB9|RS3_BORBR Small ribosomal subunit protein uS3 OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rpsC PE=3 SV=1 +MGQKIHPTGFRLAVTRNWTSRWFADDKAFGTMLAEDIRVREYLKKKLKSASVGRVIIERP +AKNARITVYSARPGVVIGKRGEDIENLKADLQRLMGVPVHVNIEEIRKPETDAQLIADSI +SQQLEKRIMFRRAMKRAMQNAMRLGAQGIKIMSSGRLNGIEIARTEWYREGRVPLHTLKA +NIDYGTSEAHTTYGVIGIKVWVYKGDMLANGELPPEAATPREEERRPRRAPRGDRPDGAR +TGRPGGRGRGPRKADAAPAPEGE +>sp|Q7WRC2|RL2_BORBR Large ribosomal subunit protein uL2 OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rplB PE=3 SV=1 +MALVKVKPTSAGRRGMVKVVSPKLHKGAPHAALLEKKTRGSGRNNNGHITVRHRGGGHKQ +HYRVVDFRRNKDGIPAKVERLEYDPNRTAHIALLCYADGERRYIIAPRGLEVGATLISGI +EAPIRAGNTLPIRNIPVGTTIHCIEMIPGKGAQMARSAGASAVLMAREGTYAQVRLRSGE +VRRVHIQCRATIGEVGNEEHSLRQIGKAGAMRWRGVRPTVRGVAMNPIDHPHGGGEGRTG +EAREPVSPWGTPAKGFKTRRNKRTNNMIVQRRKRK +>sp|Q7WRC5|RL3_BORBR Large ribosomal subunit protein uL3 OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rplC PE=3 SV=1 +MEKTMSNSTPTPAAHRLGLVGRKVGMTRIFTEDGESIPVTVLDVSNNRVTQVKSLESDGY +AAIQVTYGTRRATRVVKPQAGHYAKAGAEAGSILKEFRLDPARAAEFAAGAVIAVESVFE +AGQQVDVTGTSIGKGFAGTIKRHNFGSQRASHGNSRSHRVPGSIGQAQDPGRIFPGKRMS +GHMGDVTRTVQNLDVVRVDAERGLLMVKGAVPGHAGGDVIVRPAVKAPAKKGA +>sp|Q7WRD8|RPOC_BORBR DNA-directed RNA polymerase subunit beta' OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rpoC PE=3 SV=1 +MKALLDLFKQVSQDEQFDAIKIGIASPEKIRSWSFGEVRKPETINYRTFKPERDGLFCAK +IFGPIKDYECLCGKYKRLKHRGVICEKCGVEVTVAKVRRERMGHIELASPVAHIWFLKSL +PSRLGMVLDMTLRDIERVLYFEAWCVIEPGMTPLKRGQIMSDDDFLAKTEEYGDDFRALM +GAEAVRELLRTIDIDREVETLRGELKATSSEAKIKKISKRLKVLEGFQKSGIKAEWMVME +VLPVLPPDLRPLVPLDGGRFATSDLNDLYRRVINRNNRLKRLLELKAPEIILRNEKRMLQ +EAVDSLLDNGRRGKAMTGANKRQLKSLADMIKGKSGRFRQNLLGKRVDYSGRSVIVVGPQ +LKLHQCGLPKLMALELFKPFIFNRLEMMGLATTIKAAKKLVESQEPVVWDILEEVIREHP +VMLNRAPTLHRLGIQAFEPVLIEGKAIQLHPLVCAAFNADFDGDQMAVHVPLSLEAQLEA +RTLMLASNNVLFPANGEPSIVPSQDIVLGLYYTTRERINGKGEGIFFADVSEVQRAYDNG +EVELQTRITVRLTEYERDEQGEWQPVKHRHETTVGRALLSEILPKGLPFTVLNKALKKKE +ISRLINQSFRRCGLRDTVIFADKLMQSGFRLATRGGISIAMEDMLIPKAKEGILAEASRE +VKEIDKQYSSGLVTSQERYNNVVDIWGKAGDKVGKAMMEQLATEPVVNRHGEEVRQESFN +SIYMMADSGARGSAAQIRQLAGMRGLMAKPDGSIIETPITANFREGLNVLQYFISTHGAR +KGLADTALKTANSGYLTRRLVDVTQDLVITEDDCGTSHGYAMKALVEGGEVIEPLRDRIL +GRVAAIDVVNPDTQETAIAAGTLLDEDLVDLIDRLGVDEVKVRTPLTCETRHGLCAHCYG +RDLGRGSHVNVGEAVGVIAAQSIGEPGTQLTMRTFHIGGAASRSALASAVETKSNGTVGF +ASTMRYVTNAKGERVAISRSGELAIFDDNGRERERHKIPYGATVLVGDGEAVKAGTRLAS +WDPLTRPIVSEYSGAVRFENIEEGVTVAKQVDEVTGLSTLVVITPKTRGGKIVMRPQIKL +VNENGEDVKIAGTDHSVNISFPVGALITVRDGQQVAVGEVLARIPQESQKTRDITGGLPR +VAELFEARSPKDAGMLAEVTGTVSFGKDTKGKQRLVITDLEGVSHEFLILKEKQVLVHDG +QVVNKGEMIVDGPADPHDILRLQGIEKLATYIVDEVQDVYRLQGVKINDKHIEVIVRQML +RRVNIVDPGDTEFIPGEQVERSELLNENDRVVAEDKRPASYDNVLLGITKASLSTDSFIS +AASFQETTRVLTEAAIMGKRDDLRGLKENVIVGRLIPAGTGLAYHIARKDKEALEAAERE +AARQLANPFEDAPVTVGGEPEAPAADTPSDDSAE +>sp|Q7WRE0|RL7_BORBR Large ribosomal subunit protein bL12 OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rplL PE=3 SV=1 +MALSKAEILDAIAGMSVLELSELIKEMEEKFGVSAAAAAVAVAAPAAGGAGAAAAEEQTE +FTVVLLEAGANKVSVIKAVRELTGLGLKEAKDLVDGAPKPVKEALPKADAEAAKKKLEEA +GAKVEVK +>sp|Q7WRE1|RL10_BORBR Large ribosomal subunit protein uL10 OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rplJ PE=3 SV=1 +MSLNRQEKAVVIEEVSAQVAKAQSIVIAEYRGLDVASVTVLRKTARESGVYLRVLKNTLV +RRAVAGTAFEPLSEQLTGPLIYGISADPVAAAKVLAGFAKSNDKLVIKAGSLPNSLLTQD +GVKALATMPSREELLSKLLGTMQAPIAQFVRTLNEVPTKFARGLAAVRDQKAAA +>sp|Q7WRE2|RL1_BORBR Large ribosomal subunit protein uL1 OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rplA PE=3 SV=1 +MAKLSKRAAAIAQKIDRTKLYPVGEALNLVKETATAKFDESIDVAVQLGIDPKKSDQLVR +GSVVLPAGTGKTVRVAVFAQGEKADAARAAGADIVGLDDLAEQIKAGQMDFDVVIASPDT +MRVVGALGQVLGPRGLMPNPKVGTVTPDVATAVKNAKAGQIQYRTDKAGIIHATIGRASF +GVEQLQNNLAALVDALQKARPAAAKGIYLRKLAVSSTMGGGARVEIASLSAN +>sp|Q7WRF0|RSMG_BORBR Ribosomal RNA small subunit methyltransferase G OS=Bordetella bronchiseptica (strain ATCC BAA-588 / NCTC 13252 / RB50) OX=257310 GN=rsmG PE=3 SV=1 +MSAVPDIPGGPAQRLAQACDALRLPADAGQQQKLLRYIEQMQRWNRTYNLTAIRDPGQML +VQHLFDSLSVVAPLERGLPGVVLAIMRAHWDVTCVDAVEKKTAFVRQMAGALGLPNLQAA +HTRIEQLEPAQCDVVISRAFASLQDFAKLAGRHVREGGTLVAMKGKVPDDEIQALQQHGH +WTVERIEPLVVPALDAQRCLIWMRRSQGNI +>sp|Q830C9|Y2859_ENTFA Probable tautomerase EF_2859 OS=Enterococcus faecalis (strain ATCC 700802 / V583) OX=226185 GN=EF_2859 PE=3 SV=3 +MPFVHVELIEGRTEEQLTNMVKDITEAVSKNAGAPKENIHVIVNELKKDRYAQGGEWKK +>sp|Q833S7|PAND_ENTFA Aspartate 1-decarboxylase OS=Enterococcus faecalis (strain ATCC 700802 / V583) OX=226185 GN=panD PE=3 SV=1 +MKVTMLKGKIHRAKVEQAELDYVGSITVDMDLLEAAGIYEYEKVQIVDVNNGNRFETYTI +AGERGTGMICLNGAAARCVSTGDKIILMAYCELNENEVKDHRPKVVFVDDNNKVERVTSY +EKHGRLSDIIL +>sp|Q839Z1|GYRB_ENTFA DNA gyrase subunit B OS=Enterococcus faecalis (strain ATCC 700802 / V583) OX=226185 GN=gyrB PE=1 SV=1 +MRERAQEYDASQIQVLEGLEAVRKRPGMYIGSTSGEGLHHLVWEIVDNSIDEALAGFAKS +IQVIIEPDDSITVIDDGRGIPVGIQAKTGRPAVETVFTVLHAGGKFGGGGYKVSGGLHGV +GSSVVNALSTSLDVRVYKDGKVYYQEYRRGAVVDDLKVIEETDRHGTTVHFIPDPEIFTE +TTVYDFDKLATRVRELAFLNRGLHISIEDRREGQEDKKEYHYEGGIKSYVEHLNANKDVI +FPEPIFIEGEQQDITVEVSMQYTDGYHSNILSFANNIHTYEGGTHESGFKTSLTRVINDY +ARKQKLMKENDEKLTGEDVREGLTAVVSIKHPDPQFEGQTKTKLGNSEVRTVTDRLFSEY +FTKFLMENPTVGKQIVEKGMLASKARLAAKRAREVTRRKGALEISNLPGKLADCSSKDPE +KCELFIVEGDSAGGSAKQGRSREFQAILPIRGKILNVEKASMDKILANEEIRSLFTAMGT +GFGEDFDVSKARYHKLVIMTDADVDGAHIRTLLLTLFYRFMRPIVEAGYVYIAQPPLYGV +KQGKNITYVQPGKHAEEELAKVLEELPASPKPSVQRYKGLGEMDDHQLWETTMDPEKRLM +ARVSVDDAIEADQIFEMLMGDRVEPRRAFIEENAHYVKNLDI +>sp|Q83AH2|PARB_COXBU Probable chromosome-partitioning protein ParB OS=Coxiella burnetii (strain RSA 493 / Nine Mile phase I) OX=227377 GN=parB PE=1 SV=2 +MSMTQKRGLGRGLSDLGLNELLTEINDASLADSKTELKKLTIDVIQPGRYQPRRQMDKDA +LEELANSIRAQGIIQPIVVRPVGQRYEIIAGERRWRAAQLAGLKEVPAVIRPITDEAAIT +MSLIENIQRQNLNAIEEAAALQRLLDEFKMTHEEIAEAVGKSRTSVTNSLRLLKLNPDVK +ALLEQGHLDMGHARALLALEGFQQSEAANIIIKRALSVRETEKLIQHWQSEGKSSANRPS +MDPDVARLQHHLSDKLGAAVTIRHGAKGKGKLIIHYNSADELEGILDRIR +>sp|Q890K1|SSB_LACPL Single-stranded DNA-binding protein OS=Lactiplantibacillus plantarum (strain ATCC BAA-793 / NCIMB 8826 / WCFS1) OX=220668 GN=ssb PE=3 SV=1 +MINRTILVGRLTRDPELRYTNGGAAVATFTIAVNRQFTNQNGEREADFISCVIWRKAAEN +FANFTHKGSLVGIDGRIQTRNYENQQGVRVYVTEVVVENFSLLESRAESERHQAANGGSG +NNNYNNGNSNYNNNNNGYSNQGQNAAPQQSSANNNNPFGNGNTGNASSAAPSSSANNNNQ +ADPFANNGDQIDISDDDLPF +>sp|Q8NW33|CCPA_STAAW Catabolite control protein A OS=Staphylococcus aureus (strain MW2) OX=196620 GN=ccpA PE=3 SV=1 +MTVTIYDVAREARVSMATVSRVVNGNQNVKAETKNKVNEVIKRLNYRPNAVARGLASKKT +TTVGVIIPDISNIYYSQLARGLEDIATMYKYHSIISNSDNDPEKEKEIFNNLLSKQVDGI +IFLGGTITEEMKELINQSSVPVVVSGTNGKDAHIASVNIDFTEAAKEITGELIEKGAKSF +ALVGGEHSKKAQEDVLEGLTEVLNKNGLQLGDTLNCSGAESYKEGVKAFAKMKGNLPDAI +LCISDEEAIGIMHSAMDAGIKVPEELQIISFNNTRLVEMVRPQLSSVIQPLYDIGAVGMR +LLTKYMNDEKIEEPNVVLPHRIEYRGTTK +>sp|Q8NYV4|DUS_STAAW Probable tRNA-dihydrouridine synthase OS=Staphylococcus aureus (strain MW2) OX=196620 GN=dus PE=3 SV=1 +MKENFWSELPRPFFILAPMEDVTDIVFRHVVSEAARPDVFFTEFTNTESFCHPEGIHSVR +GRLTFSEDEQPMVAHIWGDKPEQFRETSIQLAKMGFKGIDLNMGCPVANVAKKGKGSGLI +LRPDVAAEIIQATKAGGLPVSVKTRLGYYEIDEWKDWLKHVFEQDIANLSIHLRTRKEMS +KVDAHWELIEAIKNLRDEIAPNTLLTINGDIPDRKTGLELAEKYGIDGVMIGRGIFHNPF +AFEKEPREHTSKELLDLLRLHLSLFNKYEKDEIRQFKSLRRFFKIYVRGIRGASELRHQL +MNTQSIAEARALLDEFEAQMDEDVKIEL +>sp|Q8R8N9|Y1956_CALS4 Protein TTE1956 OS=Caldanaerobacter subterraneus subsp. tengcongensis (strain DSM 15242 / JCM 11007 / NBRC 100824 / MB4) OX=273068 GN=TTE1956 PE=3 SV=1 +MGCYLMPHPPIIVHEVGRGEERKIQKTIDSLENISQEIKQKRPDTVIVITPHGYVFKDAV +SITTLPSLEGDLSQFGAREVKFKFENDLDLAREIMEETKKRNIPIAEVSEELIKKYVLPK +KLDHGTIVPLYFVTKHYPDFKLIHMSYGFLPFEKLYEFGVAIKDAINKSDRRVVFIASGD +LSHKLTPNSPNGYTPKGEVFDKTLLQLLSEMKVEEVIRMDKDLIEEAAECGFRSVCVMLG +VLDGYEVNAKVLSHEGPFGVGYGVAKFDVTGYKSTSLLERFSTIEEDPYVKLAKESLEYY +VKYRRPMPVPEGLPEEMYRRKAGVFVTLHKKGELRGCIGTVVPQKKNVAEEIIRNAISAG +FEDPRFPPVREEELPEIEYSVDVLMPTQPVKSKDELDPKRYGVIVRKGFRAGLLLPDIEG +VDTVEEQLSIALRKAGIRPDEDYTIEKFEVERHEQRVF +>sp|Q932M0|GYRA_STAAM DNA gyrase subunit A OS=Staphylococcus aureus (strain Mu50 / ATCC 700699) OX=158878 GN=gyrA PE=3 SV=1 +MAELPQSRINERNITSEMRESFLDYAMSVIVARALPDVRDGLKPVHRRILYGLNEQGMTP +DKSYKKSARIVGDVMGKYHPHGDLSIYEAMVRMAQDFSYRYPLVDGQGNFGSMDGDGAAA +MRYTEARMTKITLELLRDINKDTIDFIDNYDGNEREPSVLPARFPNLLANGASGIAVGMA +TNIPPHNLTELINGVLSLSKNPDISIAELMEDIEGPDFPTAGLILGKSGIRRAYETGRGS +IQMRSRAVIEERGGGRQRIVVTEIPFQVNKARMIEKIAELVRDKKIDGITDLRDETSLRT +GVRVVIDVRKDANASVILNNLYKQTPLQTSFGVNMIALVNGRPKLINLKEALVHYLEHQK +TVVRRRTQYNLRKAKDRAHILEGLRIALDHIDEIISTIRESDTDKVAMKSLQQRFKLSEK +QAQAILDMRLRRLTGLERDKIEAEYNELLNYISELETILADEEVLLQLVRDELTEIRDRF +GDDRRTEIQLGGFEDLEDEDLIPEEQIVITLSHNNYIKRLPVSTYRAQNRGGRGVQGMNT +LEEDFVSQLVTLSTHDHVLFFTNKGRVYKLKGYEVPELSRQSKGIPVVNAIELENDEVIS +TMIAVKDLESEDNFLVFATKRGVVKRSALSNFSRINRNGKIAISFREDDELIAVRLTSGQ +EDILIGTSHASLIRFPESTLRPLGRTATGVKGITLREGDEVVGLDVAHANSVDEVLVVTE +NGYGKRTPVNDYRLSNRGGKGIKTATITERNGNVVCITTVTGEEDLMIVTNAGVIIRLDV +ADISQNGRAAQGVRLIRLGDDQFVSTVAKVKEDAEDETNEDEQSTSTVSEDGTEQQREAV +VNDETPGNAIHTEVIDSEENDEDGRIEVRQDFMDRVEEDIQQSSDEDEE +>sp|Q97MB3|DPO4_CLOAB DNA polymerase IV OS=Clostridium acetobutylicum (strain ATCC 824 / DSM 792 / JCM 1419 / IAM 19013 / LMG 5710 / NBRC 13948 / NRRL B-527 / VKM B-1787 / 2291 / W) OX=272562 GN=dinB PE=3 SV=1 +MSKVIFHVDVNSAFLSWTAVEKLKNGEDIDIRKVPSVIGGDEKSRHGVVLAKSTPAKKYG +IVTGESLYHARKKCPNILVFPPRFDTYRKASESMMNLLKRFTPHIEKYSIDECFMDVTND +LRGMESVEFASIIKERIKNELGFTVNVGISTNRLLAKMASELNKPDKINTLYKHEIKDKM +WPLPVGELFMVGKSMKRKLNELHIKTIGELAKYDVNILKAKFKSHGNMVWEYANGIDNSD +ISRYRDEIKCISNETTLSMDLTDTEKIHKILVTLCENLGQRLRDANKYCTSISVNIRTSD +FRNYSHQKKLKNAVSSTRDIILYADKIFNEMWGREPIRLLGVQLSGLCSGSSVQISMFDE +KTDTRNEILDKTLDSIRKKYGDNVIMRSVLLEKEEK +>sp|Q97N21|SYS2_CLOAB Serine--tRNA ligase 2 OS=Clostridium acetobutylicum (strain ATCC 824 / DSM 792 / JCM 1419 / IAM 19013 / LMG 5710 / NBRC 13948 / NRRL B-527 / VKM B-1787 / 2291 / W) OX=272562 GN=serS2 PE=3 SV=1 +MLDLDLIRNDTEKVKKALLKKIDNVDFTELLKLDDERRKLIHEVEVLKNKKNEASKQISN +IKSQGGKVDESFFKDIKEISNKISELETSLEPIKGKMDTFLEALPNIPDEDVLPGGKENN +KVVHVYGEKPQFEFEPKDHVELSNIHDLIDYKRGTKLSGNGFWIYKGYGAILEWALLNYF +IEEHIKDGYEFILPPHILNYECGRTAGQFPKFKDEVFKVGSNGEGEGMQFILPTAETALV +NLHRDEILKEDELPKKYFAYTPCYRVEAGSYRASERGMIRGHQFNKIEMFQYTKPEDSDA +ALEELIGKAEKLVKGLGLHYRLSKLAAADCSASMAKTYDIEVWIPSMNEYKEVSSASNAR +DYQARRGKIRFRREETKKIEYVNTLNASGLATSRVLPAILEQMQDKDGSIVVPEVLRKWV +GKDKL +>sp|Q9KTJ6|METI_VIBCH Probable D-methionine transport system permease protein MetI OS=Vibrio cholerae serotype O1 (strain ATCC 39315 / El Tor Inaba N16961) OX=243277 GN=metI PE=3 SV=1 +MSFNTIAQWFALNSDLLLTATWQTLYMVAIAGAVGFALGIPLGVILHTTKKEGLLENLPL +NRALGAVVNIGRSVPFLVLMVAIIPVTKLIVGTFIGTTAAIVPLTIGAIPFVARLIESAL +LEVPTGLVEAAQSMGATPLQIIRKVLLPEALPTILNSVTITLVTLVSYSAMAGTVGGGGL +GDVAIRYGFHRYDITIMAVTVVMLIVLVQIIQSIGDALVRRVDHR +>sp|Q9L7L3|GYRB_MYCPA DNA gyrase subunit B OS=Mycolicibacterium paratuberculosis (strain ATCC BAA-968 / K-10) OX=262316 GN=gyrB PE=3 SV=1 +MAAQKKKAQDEYGASAITVLEGLEAVRKRPGMYIGSTGERGLHHLIWEVVDNSVDEAMAG +YADRVDVRILDDGSVEVADNGRGIPVAMHATGAPTVDVVMTQLHAGGKFGGENSGYNVSG +GLHGVGVSVVNALSTRLEVNIARDGYEWSQYYDHAVPGTLKQGEATKRTGTTIRFWADPD +IFETTEYDFETVARRLQEMAFLNKGLTINLTDERVTNEEVVDEVVSDTADAPKSAQEKAA +ESAAPHKVKHRTFHYPGGLVDFVKHINRTKNPIHQSIIDFGGKGPGHEVEIAMQWNGGYS +ESVHTFANTINTHEGGTHEEGFRSALTSVVNKYAKDKKLLKDKDPNLTGDDIREGLAAVI +SVKVSEPQFEGQTKTKLGNTEVKSFVQKVCNEQLTHWFEANPADAKVIVNKAVSSAQARI +AARKARELVRRKSATDLGGLPGKLADCRSTDPRKSELYVVEGDSAGGSAKSGRDSMFQAI +LPLRGKIINVEKARIDRVLKNTEVQAIITALGTGIHDEFDITKLRYHKIVLMADADVDGQ +HISTLLLTLLFRFMRPLIEHGHVFLAQPPLYKLKWQRSDPEFAYSDRERDGLLEAGLKAG +KKINKDDGIQRYKGLGEMDAKELWETTMDPTVRVLRQVTLDDAAAADELFSILMGEDVDA +RRSFITRNAKDVRFLDV +>sp|Q9L7L4|Y004_MYCPA UPF0232 protein MAP_0004 OS=Mycolicibacterium paratuberculosis (strain ATCC BAA-968 / K-10) OX=262316 GN=MAP_0004 PE=3 SV=2 +MSDDQSPSPSGEPTAMDLVRRTLEEARAAARAQGKDAGRGRAAAPTPRRVAGQRRSWSGP +GPDARDPQPLGRLARDLARKRGWSAQVAEGTVLGNWTAVVGHQIADHAVPTGLRDGVLSV +SAESTAWATQLRMMQAQLLAKIAAAVGNGVVTSLKITGPAAPSWRKGPRHIAGRGPRDTY +G +>sp|Q9L7L5|RECF_MYCPA DNA replication and repair protein RecF OS=Mycolicibacterium paratuberculosis (strain ATCC BAA-968 / K-10) OX=262316 GN=recF PE=3 SV=2 +MYVRHLGLRDFRSWAHADLELQPGRTVFIGSNGFGKTNLLEALWYSSTLGSHRVGTDAPL +IRAGADRAVVSTIVVNDGRECAVDLEIAAGRANKARLNRSPVRSTREVLGVLRAVLFAPE +DLALVRGDPSERRRYLDDLATLRRPAIAAVRADYDKVLRQRTALLKSLSGARHRGDRGAL +DTLDVWDSRLAEYGAQLMAARIDLVNQLAPEVEKAYQLLAPGSRAASIGYRSSLGAAASA +EVNAGDRDYLEAALLAGLAAHRDAELERGMCLVGPHRDDLELWLGEQVAKGFASHGESWS +LALSLRLAAFELLRADESDPVLLLDDVFAELDAARRRALAAVAESAEQVLVTAAVLEDIP +TGWQARRLFVELRDTDAGRVSELRP +>sp|Q9L7L6|DPO3B_MYCPA Beta sliding clamp OS=Mycolicibacterium paratuberculosis (strain ATCC BAA-968 / K-10) OX=262316 GN=dnaN PE=3 SV=1 +MDAATTTAGLSDLKFRLVRESFADAVSWVAKSLPSRPAVPVLSGVLLSGTDEGLTISGFD +YEVSAEAQVAAEIASPGSVLVSGRLLSDIVRALPNKPIDFYVDGNRVALNCGSARFSLPT +MAVEDYPTLPTLPEETGTLPADLFAEAIGQVAIAAGRDDTLPMLTGIRVEISGDTVVLAA +TDRFRLAVRELTWSAASPDIEAAVLVPAKTLAEAARTGIDGSDVRLSLGAGAGVGKDGLL +GISGNGKRSTTRLLDAEFPKFRQLLPAEHTAVATINVAELTEAIKLVALVADRGAQVRME +FSEGSLRLSAGADDVGRAEEDLAVDFAGEPLTIAFNPTYLTDGLGSVRSERVSFGFTTPG +KPALLRPASDDDSPPSGSGPFSALPTDYVYLLMPVRLPG +>sp|Q9L7L7|DNAA_MYCPA Chromosomal replication initiator protein DnaA OS=Mycolicibacterium paratuberculosis (strain ATCC BAA-968 / K-10) OX=262316 GN=dnaA PE=3 SV=1 +MADDPGSSFTTVWNAVVSELNGEPVADGGAANRTTLVTPLTPQQRAWLNLVRPLTIVEGF +ALLSVPSSFVQNEIERHLRAPITDALSRRLGQQIQLGVRIAPPPDDVEDAPIPPAEPFPD +TDAALSADDGADGEPVENGEPVTDTQPGWPNYFTERPHAIDPAVAAGTSLNRRYTFDTFV +IGASNRFAHAAALAIAEAPARAYNPLFIWGESGLGKTHLLHAAGNYAQRLFPGMRVKYVS +TEEFTNDFINSLRDDRKVAFKRSYRDVDVLLVDDIQFIEGKEGIQEEFFHTFNTLHNANK +QIVISSDRPPKQLATLEDRLRTRFEWGLITDVQPPELETRIAILRKKAQMERLAVPDDVL +ELIASSIERNIRELEGALIRVTAFASLNKTPIDKSLAEIVLRDLIADASTMQISAATIMA +ATAEYFDTTVEELRGPGKTRALAQSRQIAMYLCRELTDLSLPKIGQAFGRDHTTVMYAQR +KILSEMAERREVFDHVKELTTRIRQRSKR +>sp|Q9R2F4|CHNE_ACIJO 6-oxohexanoate dehydrogenase OS=Acinetobacter johnsonii OX=40214 GN=chnE PE=1 SV=1 +MNYPNIPLYINGEFLDHTNRDVKEVFNPVNHECIGLMACASQADLDYALESSQQAFLRWK +KTSPITRSEILRTFAKLAREKAAEIGRNITLDQGKPLKEAIAEVTVCAEHAEWHAEECRR +IYGRVIPPRNPNVQQLVVREPLGVCLAFSPWNFPFNQAIRKISAAIAAGCTIIVKGSGDT +PSAVYAIAQLFHEAGLPNGVLNVIWGDSNFISDYMIKSPIIQKISFTGSTPVGKKLASQA +SLYMKPCTMELGGHAPVIVCDDADIDAAVEHLVGYKFRNAGQVCVSPTRFYVQEGIYKEF +SEKVVLRAKQIKVGCGLDASSDMGPLAQARRMHAMQQIVEDAVHKGSKLLLGGNKISDKG +NFFEPTVLGDLCNDTQFMNDEPFGPIIGLIPFDTIDHVLEEANRLPFGLASYAFTTSSKN +AHQISYGLEAGMVSINHMGLALAETPFGGIKDSGFGSEGGIETFDGYLRTKFITQLN +>sp|Q9WXS3|UXUB_THEMA D-mannonate oxidoreductase OS=Thermotoga maritima (strain ATCC 43589 / DSM 3109 / JCM 10099 / NBRC 100826 / MSB8) OX=243274 GN=uxuB PE=1 SV=1 +MRLNRETIKDRAAWEKIGVRPPYFDLDEVEKNTKEQPKWVHFGGGNIFRGFVAAVLQNLL +EEGKEDTGINVIELFDYEVIDKVYKPYDNLSIAVTIKPDGDFEKRIIASVMEALKGDPSH +PDWERAKEIFRNPSLQLASLTITEKGYNIEDQAGNLFPQVMEDMKNGPVSPQTSMGKVAA +LLYERFKAGRLPIALLSLDNFSRNGEKLYSSVKRISEEWVKSGLVEKDFIDYLEKDVAFP +WSMIDKIVPGPSEFIKEHLEKLGIEGMEIFVTSKRTHIAPFVNMEWAQYLVIEDSFPNGR +PKLEGADRNVFLTDRETVEKAERMKVTTCLNPLHTALAIFGCLLGYKKIADEMKDPLLKK +LVEGVGEEGIKVVVDPGIINPREFLNEVINIRLPNPYLPDTPQRIATDTSQKMPIRFGET +IKAYHERPDLDPRNLKYIPLVIAGWCRYLMGIDDEGREMQLSPDPLLENLRSYVSKIKFG +DPESTDDHLKPILSSQQLFRVNLYEVGLGEKIEELFKKMITGPRAVRKTLEEVVGREDG +>sp|Q9X4C9|DNAB_GEOSE Replicative DNA helicase DnaB OS=Geobacillus stearothermophilus OX=1422 GN=dnaB PE=1 SV=1 +MSELFSERIPPQSIEAEQAVLGAVFLDPAALVPASEILIPEDFYRAAHQKIFHAMLRVAD +RGEPVDLVTVTAELAASEQLEEIGGVSYLSELADAVPTAANVEYYARIVEEKSVLRRLIR +TATSIAQDGYTREDEIDVLLDEADRKIMEVSQRKHSGAFKNIKDILVQTYDNIEMLHNRD +GEITGIPTGFTELDRMTSGFQRSDLIIVAARPSVGKTAFALNIAQNVATKTNENVAIFSL +EMSAQQLVMRMLCAEGNINAQNLRTGKLTPEDWGKLTMAMGSLSNAGIYIDDTPSIRVSD +IRAKCRRLKQESGLGMIVIDYLQLIQGSGRSKENRQQEVSEISRSLKALARELEVPVIAL +SQLSRSVEQRQDKRPMMSDIRESGSIEQDADIVAFLYRDDYYNKDSENKNIIEIIIAKQR +NGPVGTVQLAFIKEYNKFVNLERRFDEAQIPPGA +>sp|Q9Y6N5|SQOR_HUMAN Sulfide:quinone oxidoreductase, mitochondrial OS=Homo sapiens OX=9606 GN=SQOR PE=1 SV=1 +MVPLVAVVSGPRAQLFACLLRLGTQQVGPLQLHTGASHAARNHYEVLVLGGGSGGITMAA +RMKRKVGAENVAIVEPSERHFYQPIWTLVGAGAKQLSSSGRPTASVIPSGVEWIKARVTE +LNPDKNCIHTDDDEKISYRYLIIALGIQLDYEKIKGLPEGFAHPKIGSNYSVKTVEKTWK +ALQDFKEGNAIFTFPNTPVKCAGAPQKIMYLSEAYFRKTGKRSKANIIFNTSLGAIFGVK +KYADALQEIIQERNLTVNYKKNLIEVRADKQEAVFENLDKPGETQVISYEMLHVTPPMSP +PDVLKTSPVADAAGWVDVDKETLQHRRYPNVFGIGDCTNLPTSKTAAAVAAQSGILDRTI +SVIMKNQTPTKKYDGYTSCPLVTGYNRVILAEFDYKAEPLETFPFDQSKERLSMYLMKAD +LMPFLYWNMMLRGYWGGPAFLRKLFHLGMS +>sp|Q9ZF37|YSUP_LACHE Putative sugar uptake protein OS=Lactobacillus helveticus OX=1587 PE=3 SV=1 +MNILIALIPALGWGIFSLIAGKIKNSHPANELMGLGTGALIIGIITAIIHPASSNITIFS +LSLISGMFCALGQSGQFISMRNIGISKTMPLSTGFQLIGNTLIGAIIFGEWTSSSQYLIG +TLALILIIVGVSLTAISKDKSAKLKMRDIILLLFTSIGYWIYSSFPKAITANAQTLFLPQ +MIGIFIGSIIFLLVSRQTKVLKEKATWLNIFSGFSYGIAAFSYIFSAQLNGVITAFIYSQ +LCVIISTLGGIFFIGENKTKSELIATFVGLILIIIGAAIQ +>sp|Q9ZLB9|ASPG_HELPJ Probable L-asparaginase OS=Helicobacter pylori (strain J99 / ATCC 700824) OX=85963 GN=ansA PE=1 SV=1 +MAQNLPTIALLATGGTIAGSGVDASLGSYKSGELGVKELLKAIPSLNKIARIQGEQVSNI +GSQDMNEEIWFKLAQRAQELLDDSRIQGVVITHGTDTLEESAYFLNLVLHSTKPVVLVGA +MRNASSLSADGALNLYYAVSVAVNEKSANKGVLVVMDDTIFRVREVVKTHTTHISTFKAL +NSGAIGSVYYGKTRYYMQPLRKHTTESEFSLSQLKTPLPKVDIIYTHAGMTPDLFQASLN +SHAKGVVIAGVGNGNVSAGFLKAMQEASQMGVVIVRSSRVGSGGVTSGEIDDKAYGFITS +DNLNPQKARVLLQLALTKTNDKAKIQEMFEEY diff --git a/metapathways/regtests/test_db/test_reference.json b/metapathways/regtests/test_db/test_reference.json new file mode 100644 index 0000000..3caf645 --- /dev/null +++ b/metapathways/regtests/test_db/test_reference.json @@ -0,0 +1,1434 @@ +{ + "description": "Small interface-test reference: original K12 fixture plus genuine SwissProt best-hit targets from the three CAMI test samples. Selection is deliberately biased to exercise annotation, taxonomy and MAG splitting; not for accuracy assessment.", + "source": "MPDB_260131/functional/swissprot", + "source_sha256": "987e9d468c2691008b8c7d9d3eea79dec207b5c892312f8d63d1fe6f5b01cead", + "original_ids": [ + "sp|A7ZHS2|LPXB_ECO24", + "sp|B1X6D4|AROE_ECODH", + "sp|B1XD36|GLND_ECODH", + "sp|B5Z0I6|GLO2_ECO5E", + "sp|B7M1Y6|RNH2_ECO8A", + "sp|C4ZUE2|FMT_ECOBW", + "sp|C4ZUE3|RSMB_ECOBW", + "sp|P0A914|PAL_SHIFL", + "sp|P0A9W9|YRDA_ECOLI", + "sp|P0ABG1|CDSA_ECOLI", + "sp|P0ABV0|TOLQ_ECO57", + "sp|P0ABV9|TOLR_SHIFL", + "sp|P0AE20|MAP1_ECO57", + "sp|P0AEH1|RSEP_ECOLI", + "sp|P0AEZ8|MLTD_ECOL6", + "sp|P0AFP0|YADS_ECOLI", + "sp|P0AGJ0|TRKA_ECO57", + "sp|P0C0V0|DEGP_ECOLI", + "sp|P19934|TOLA_ECOLI", + "sp|P21645|LPXD_ECOLI", + "sp|P30852|SMF_ECOLI", + "sp|P30863|DKGB_ECOLI", + "sp|P30864|YAFC_ECOLI", + "sp|P30866|YAFE_ECOLI", + "sp|P37028|BTUF_ECOLI", + "sp|P37047|CDAR_ECOLI", + "sp|P37049|YAEI_ECOLI", + "sp|P45568|DXR_ECOLI", + "sp|P45748|TSAC_ECOLI", + "sp|P45768|YHDY_ECOLI", + "sp|P45769|YHDZ_ECOLI", + "sp|P45771|YRDD_ECOLI", + "sp|P45795|YRDB_ECOLI", + "sp|P45955|CPOB_ECOLI", + "sp|P60475|UPPS_SHIFL", + "sp|Q3YWX5|SMG_SHISS", + "sp|Q3Z5I0|SKP_SHISS", + "sp|Q59827|DGTP_SHIBO", + "sp|Q8X8X5|DPO3A_ECO57" + ], + "added_ids": [ + "sp|A0A086F3E3|TM175_CHRP1", + "sp|A0A0H2ZQT7|WALJ_STRP2", + "sp|A0A0H3GCG4|PDEA_LISM4", + "sp|A0A4Y1WBN6|YYCJ_BACAN", + "sp|A0PXQ3|PANB_CLONN", + "sp|A0Q8S8|CRGA_MYCA1", + "sp|A3DHZ4|DNAA_ACET2", + "sp|A5INP9|HUTH_STAA9", + "sp|A5TY85|PKNA_MYCTA", + "sp|A6QD58|WALK_STAAE", + "sp|A6U2M4|SYL_STAA2", + "sp|A8YW47|RS6_LACH4", + "sp|A8YYT2|SYS_STAAT", + "sp|A8Z4J3|RISB_STAAT", + "sp|A9KHK4|GLRP_LACP7", + "sp|A9KPP1|DNAA_LACP7", + "sp|B2UYM8|SYY_CLOBA", + "sp|B5ELX2|RPOB_ACIF5", + "sp|B8I2Z3|PANC_RUMCH", + "sp|C0SPC1|CCRZ_BACSU", + "sp|C4Z940|RECF_AGARV", + "sp|E3GC98|MDTG_ENTLS", + "sp|F9UMX3|GLPF4_LACPL", + "sp|H8L901|RACD_ENTFU", + "sp|H8L902|ASL_ENTFU", + "sp|O06672|DPO3B_STRPN", + "sp|O32068|YTZG_BACSU", + "sp|O34357|YTPP_BACSU", + "sp|O34546|YTTB_BACSU", + "sp|O34760|YTNP_BACSU", + "sp|O34924|YTOP_BACSU", + "sp|O34943|YTPR_BACSU", + "sp|O34948|YKWC_BACSU", + "sp|O35008|YTQA_BACSU", + "sp|O50628|GYRA_HALH5", + "sp|P05057|KANU_STAAU", + "sp|P05649|DPO3B_BACSU", + "sp|P05650|RLBA_BACSU", + "sp|P05653|GYRA_BACSU", + "sp|P07944|PBP_STAAU", + "sp|P0A0B2|MECR_STAEP", + "sp|P0A0C3|REPB_STAAU", + "sp|P0A0C4|REPB_BACSP", + "sp|P0A0C7|PRE2_STAAU", + "sp|P0A150|YGIDB_PSEPU", + "sp|P0A4C6|RS4_BORBR", + "sp|P0A4E5|RPOA_BORPE", + "sp|P0A4X9|MBTM_MYCBO", + "sp|P0AGA4|SECY_ECO57", + "sp|P0C1L0|T431_STAA8", + "sp|P0C2T9|METC_LACLC", + "sp|P0DG03|GYRA_STRPQ", + "sp|P0DKR7|WHB5B_MYCTO", + "sp|P10249|SUCP_STRMU", + "sp|P19834|YI11_STRCL", + "sp|P26497|SP0J_BACSU", + "sp|P27159|XYLR_STAXY", + "sp|P37478|WALR_BACSU", + "sp|P37484|GDPP_BACSU", + "sp|P37522|SOJ_BACSU", + "sp|P39062|ACSA_BACSU", + "sp|P39912|AROG_BACSU", + "sp|P50854|RISA_ACTPL", + "sp|P54744|PKNB_MYCLE", + "sp|P56067|CYSM_HELPY", + "sp|P63453|MBTL_MYCBO", + "sp|P65476|MURC_STAAW", + "sp|P65592|NUSG_NEIMB", + "sp|P65763|PPIA_MYCBO", + "sp|P65884|PURA_STAAM", + "sp|P66936|GYRB_STAAM", + "sp|P67705|Y023_MYCBO", + "sp|P67925|BLE_GEOSE", + "sp|P68263|MECI_STAEP", + "sp|P71590|FHAA_MYCTU", + "sp|P77307|FETB_ECOLI", + "sp|P82371|SCRK_LACLC", + "sp|P94604|GYRB_CLOAB", + "sp|P94605|GYRA_CLOAB", + "sp|P96670|YDEM_BACSU", + "sp|P96726|YWQN_BACSU", + "sp|P99178|SYS_STAAN", + "sp|P9WG46|GYRA_MYCTO", + "sp|P9WHW4|PSTP_MYCTO", + "sp|P9WJB4|FHAB_MYCTO", + "sp|P9WJF2|CWSA_MYCTO", + "sp|P9WKD0|PBPA_MYCTO", + "sp|P9WMA1|Y025_MYCTU", + "sp|P9WMA2|Y010_MYCTO", + "sp|P9WMA7|Y007_MYCTU", + "sp|P9WMB1|Y0026_MYCTU", + "sp|P9WN34|TRPG_MYCTO", + "sp|P9WN99|RODA_MYCTU", + "sp|P9WPL1|CP144_MYCTU", + "sp|Q03928|HSP18_CLOAB", + "sp|Q03UD8|RS6_LEVBA", + "sp|Q03UE4|DNAA_LEVBA", + "sp|Q04CW7|RS18_LACDB", + "sp|Q04CX5|DNAA_LACDB", + "sp|Q0AU85|METN_SYNWW", + "sp|Q0TTK8|FTSH_CLOP1", + "sp|Q13SP0|MNMG_PARXL", + "sp|Q1GC33|RL9_LACDA", + "sp|Q1GC40|RECF_LACDA", + "sp|Q1LI29|EFG1_CUPMC", + "sp|Q1WVN7|RS18_LIGS1", + "sp|Q1WVP2|RECF_LIGS1", + "sp|Q1WVQ8|MURE_LIGS1", + "sp|Q1WVR1|MSCL_LIGS1", + "sp|Q1WVT3|MSRB_LIGS1", + "sp|Q24PG6|SFSA_DESHY", + "sp|Q2FFZ0|TRMB_STAA3", + "sp|Q2FXG9|MNMM_STAA8", + "sp|Q2G2P8|NNRD_STAA8", + "sp|Q2L284|RL14_BORA1", + "sp|Q2L2H8|RS12_BORA1", + "sp|Q58481|Y1081_METJA", + "sp|Q59495|SUCP_LEUME", + "sp|Q59935|MANA_STRMU", + "sp|Q5HF24|DAAA_STAAC", + "sp|Q5HF39|ACUC_STAAC", + "sp|Q5HJX8|PURA_STAAC", + "sp|Q5HK03|GYRB_STAEQ", + "sp|Q5XD45|Y533_STRP6", + "sp|Q65G33|ACUA_BACLD", + "sp|Q6G8G1|RIBBA_STAAS", + "sp|Q6GD38|DUS_STAAS", + "sp|Q6GKU3|DPO3B_STAAR", + "sp|Q795R8|YTFP_BACSU", + "sp|Q79GC6|EFTU_BORPA", + "sp|Q7A0M4|Y1682_STAAW", + "sp|Q7A514|ROT_STAAN", + "sp|Q7A522|PEPVL_STAAN", + "sp|Q7A528|Y1564_STAAN", + "sp|Q7TT91|EFTU_BORPE", + "sp|Q7VTA7|RS11_BORPE", + "sp|Q7VTA8|RS13_BORPE", + "sp|Q7VTB0|IF12_BORPE", + "sp|Q7VTB2|RL15_BORPE", + "sp|Q7VTB7|RS8_BORPE", + "sp|Q7VTB8|RS14_BORPE", + "sp|Q7VTB9|RL5_BORPE", + "sp|Q7VTC0|RL24_BORPE", + "sp|Q7VTC4|RS17_BORPE", + "sp|Q7VTC6|RL16_BORPE", + "sp|Q7VTD1|RL23_BORPE", + "sp|Q7VTD4|RS10_BORPE", + "sp|Q7VTD6|RS7_BORPE", + "sp|Q7W2E0|RL6_BORPA", + "sp|Q7W2F1|RL22_BORPA", + "sp|Q7W2F2|RS19_BORPA", + "sp|Q7W2F5|RL4_BORPA", + "sp|Q7W2H3|RL11_BORPA", + "sp|Q7WRA5|RL30_BORBR", + "sp|Q7WRA6|RS5_BORBR", + "sp|Q7WRA7|RL18_BORBR", + "sp|Q7WRB7|RL29_BORBR", + "sp|Q7WRB9|RS3_BORBR", + "sp|Q7WRC2|RL2_BORBR", + "sp|Q7WRC5|RL3_BORBR", + "sp|Q7WRD8|RPOC_BORBR", + "sp|Q7WRE0|RL7_BORBR", + "sp|Q7WRE1|RL10_BORBR", + "sp|Q7WRE2|RL1_BORBR", + "sp|Q7WRF0|RSMG_BORBR", + "sp|Q830C9|Y2859_ENTFA", + "sp|Q833S7|PAND_ENTFA", + "sp|Q839Z1|GYRB_ENTFA", + "sp|Q83AH2|PARB_COXBU", + "sp|Q890K1|SSB_LACPL", + "sp|Q8NW33|CCPA_STAAW", + "sp|Q8NYV4|DUS_STAAW", + "sp|Q8R8N9|Y1956_CALS4", + "sp|Q932M0|GYRA_STAAM", + "sp|Q97MB3|DPO4_CLOAB", + "sp|Q97N21|SYS2_CLOAB", + "sp|Q9KTJ6|METI_VIBCH", + "sp|Q9L7L3|GYRB_MYCPA", + "sp|Q9L7L4|Y004_MYCPA", + "sp|Q9L7L5|RECF_MYCPA", + "sp|Q9L7L6|DPO3B_MYCPA", + "sp|Q9L7L7|DNAA_MYCPA", + "sp|Q9R2F4|CHNE_ACIJO", + "sp|Q9WXS3|UXUB_THEMA", + "sp|Q9X4C9|DNAB_GEOSE", + "sp|Q9Y6N5|SQOR_HUMAN", + "sp|Q9ZF37|YSUP_LACHE", + "sp|Q9ZLB9|ASPG_HELPJ" + ], + "supporting_hits": [ + { + "sample": "Gastrointestinal_5", + "orf": "C1-G1", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q13SP0|MNMG_PARXL" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G2", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7WRF0|RSMG_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G3", + "contig": "Gastrointestinal_5-C1", + "target": "sp|P0A150|YGIDB_PSEPU" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G5", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q83AH2|PARB_COXBU" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G6", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q9ZLB9|ASPG_HELPJ" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G7", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7TT91|EFTU_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G9", + "contig": "Gastrointestinal_5-C1", + "target": "sp|P65592|NUSG_NEIMB" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G10", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7W2H3|RL11_BORPA" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G11", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7WRE2|RL1_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G12", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7WRE1|RL10_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G13", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7WRE0|RL7_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G14", + "contig": "Gastrointestinal_5-C1", + "target": "sp|B5ELX2|RPOB_ACIF5" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G15", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7WRD8|RPOC_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G25", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q2L2H8|RS12_BORA1" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G26", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTD6|RS7_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G27", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q1LI29|EFG1_CUPMC" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G28", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q79GC6|EFTU_BORPA" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G29", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTD4|RS10_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G30", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7WRC5|RL3_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G31", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7W2F5|RL4_BORPA" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G32", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTD1|RL23_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G33", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7WRC2|RL2_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G34", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7W2F2|RS19_BORPA" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G35", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7W2F1|RL22_BORPA" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G36", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7WRB9|RS3_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G37", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTC6|RL16_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G38", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7WRB7|RL29_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G39", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTC4|RS17_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G41", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q9R2F4|CHNE_ACIJO" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G43", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q2L284|RL14_BORA1" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G44", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTC0|RL24_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G45", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTB9|RL5_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G46", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTB8|RS14_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G47", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTB7|RS8_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G48", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7W2E0|RL6_BORPA" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G49", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7WRA7|RL18_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G50", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7WRA6|RS5_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G51", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7WRA5|RL30_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G52", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTB2|RL15_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G53", + "contig": "Gastrointestinal_5-C1", + "target": "sp|P0AGA4|SECY_ECO57" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G54", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTB0|IF12_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G56", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTA8|RS13_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G57", + "contig": "Gastrointestinal_5-C1", + "target": "sp|Q7VTA7|RS11_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G58", + "contig": "Gastrointestinal_5-C1", + "target": "sp|P0A4C6|RS4_BORBR" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C1-G59", + "contig": "Gastrointestinal_5-C1", + "target": "sp|P0A4E5|RPOA_BORPE" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C2-G1", + "contig": "Gastrointestinal_5-C2", + "target": "sp|A9KPP1|DNAA_LACP7" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C2-G4", + "contig": "Gastrointestinal_5-C2", + "target": "sp|C4Z940|RECF_AGARV" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C2-G5", + "contig": "Gastrointestinal_5-C2", + "target": "sp|P94604|GYRB_CLOAB" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C2-G6", + "contig": "Gastrointestinal_5-C2", + "target": "sp|P94605|GYRA_CLOAB" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C2-G9", + "contig": "Gastrointestinal_5-C2", + "target": "sp|A0PXQ3|PANB_CLONN" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C2-G10", + "contig": "Gastrointestinal_5-C2", + "target": "sp|B8I2Z3|PANC_RUMCH" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C2-G11", + "contig": "Gastrointestinal_5-C2", + "target": "sp|Q833S7|PAND_ENTFA" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C2-G26", + "contig": "Gastrointestinal_5-C2", + "target": "sp|Q0TTK8|FTSH_CLOP1" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C2-G36", + "contig": "Gastrointestinal_5-C2", + "target": "sp|B2UYM8|SYY_CLOBA" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G1", + "contig": "Gastrointestinal_5-C3", + "target": "sp|A3DHZ4|DNAA_ACET2" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G5", + "contig": "Gastrointestinal_5-C3", + "target": "sp|P94604|GYRB_CLOAB" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G8", + "contig": "Gastrointestinal_5-C3", + "target": "sp|P37522|SOJ_BACSU" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G9", + "contig": "Gastrointestinal_5-C3", + "target": "sp|P26497|SP0J_BACSU" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G11", + "contig": "Gastrointestinal_5-C3", + "target": "sp|Q97N21|SYS2_CLOAB" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G12", + "contig": "Gastrointestinal_5-C3", + "target": "sp|Q8R8N9|Y1956_CALS4" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G15", + "contig": "Gastrointestinal_5-C3", + "target": "sp|Q58481|Y1081_METJA" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G16", + "contig": "Gastrointestinal_5-C3", + "target": "sp|Q24PG6|SFSA_DESHY" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G17", + "contig": "Gastrointestinal_5-C3", + "target": "sp|Q9WXS3|UXUB_THEMA" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G19", + "contig": "Gastrointestinal_5-C3", + "target": "sp|A9KHK4|GLRP_LACP7" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G27", + "contig": "Gastrointestinal_5-C3", + "target": "sp|Q03928|HSP18_CLOAB" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G31", + "contig": "Gastrointestinal_5-C3", + "target": "sp|Q97MB3|DPO4_CLOAB" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G39", + "contig": "Gastrointestinal_5-C3", + "target": "sp|Q0AU85|METN_SYNWW" + }, + { + "sample": "Gastrointestinal_5", + "orf": "C3-G40", + "contig": "Gastrointestinal_5-C3", + "target": "sp|Q9KTJ6|METI_VIBCH" + }, + { + "sample": "Skin_28", + "orf": "C1-G2", + "contig": "Skin_28-C1", + "target": "sp|Q6GKU3|DPO3B_STAAR" + }, + { + "sample": "Skin_28", + "orf": "C1-G5", + "contig": "Skin_28-C1", + "target": "sp|P66936|GYRB_STAAM" + }, + { + "sample": "Skin_28", + "orf": "C1-G6", + "contig": "Skin_28-C1", + "target": "sp|Q932M0|GYRA_STAAM" + }, + { + "sample": "Skin_28", + "orf": "C1-G7", + "contig": "Skin_28-C1", + "target": "sp|Q2G2P8|NNRD_STAA8" + }, + { + "sample": "Skin_28", + "orf": "C1-G8", + "contig": "Skin_28-C1", + "target": "sp|A5INP9|HUTH_STAA9" + }, + { + "sample": "Skin_28", + "orf": "C1-G9", + "contig": "Skin_28-C1", + "target": "sp|P99178|SYS_STAAN" + }, + { + "sample": "Skin_28", + "orf": "C1-G14", + "contig": "Skin_28-C1", + "target": "sp|P37484|GDPP_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C1-G16", + "contig": "Skin_28-C1", + "target": "sp|Q9X4C9|DNAB_GEOSE" + }, + { + "sample": "Skin_28", + "orf": "C1-G17", + "contig": "Skin_28-C1", + "target": "sp|P65884|PURA_STAAM" + }, + { + "sample": "Skin_28", + "orf": "C1-G22", + "contig": "Skin_28-C1", + "target": "sp|A0A4Y1WBN6|YYCJ_BACAN" + }, + { + "sample": "Skin_28", + "orf": "C1-G28", + "contig": "Skin_28-C1", + "target": "sp|P0C1L0|T431_STAA8" + }, + { + "sample": "Skin_28", + "orf": "C1-G29", + "contig": "Skin_28-C1", + "target": "sp|P0A0C3|REPB_STAAU" + }, + { + "sample": "Skin_28", + "orf": "C1-G30", + "contig": "Skin_28-C1", + "target": "sp|P0A0C4|REPB_BACSP" + }, + { + "sample": "Skin_28", + "orf": "C1-G31", + "contig": "Skin_28-C1", + "target": "sp|P0A0C7|PRE2_STAAU" + }, + { + "sample": "Skin_28", + "orf": "C1-G32", + "contig": "Skin_28-C1", + "target": "sp|P67925|BLE_GEOSE" + }, + { + "sample": "Skin_28", + "orf": "C1-G33", + "contig": "Skin_28-C1", + "target": "sp|P05057|KANU_STAAU" + }, + { + "sample": "Skin_28", + "orf": "C1-G34", + "contig": "Skin_28-C1", + "target": "sp|P0C1L0|T431_STAA8" + }, + { + "sample": "Skin_28", + "orf": "C1-G38", + "contig": "Skin_28-C1", + "target": "sp|P96670|YDEM_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C1-G39", + "contig": "Skin_28-C1", + "target": "sp|P07944|PBP_STAAU" + }, + { + "sample": "Skin_28", + "orf": "C1-G40", + "contig": "Skin_28-C1", + "target": "sp|P0A0B2|MECR_STAEP" + }, + { + "sample": "Skin_28", + "orf": "C1-G41", + "contig": "Skin_28-C1", + "target": "sp|P68263|MECI_STAEP" + }, + { + "sample": "Skin_28", + "orf": "C1-G42", + "contig": "Skin_28-C1", + "target": "sp|P27159|XYLR_STAXY" + }, + { + "sample": "Skin_28", + "orf": "C2-G2", + "contig": "Skin_28-C2", + "target": "sp|Q6GKU3|DPO3B_STAAR" + }, + { + "sample": "Skin_28", + "orf": "C2-G5", + "contig": "Skin_28-C2", + "target": "sp|Q5HK03|GYRB_STAEQ" + }, + { + "sample": "Skin_28", + "orf": "C2-G6", + "contig": "Skin_28-C2", + "target": "sp|P05653|GYRA_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C2-G7", + "contig": "Skin_28-C2", + "target": "sp|Q2G2P8|NNRD_STAA8" + }, + { + "sample": "Skin_28", + "orf": "C2-G8", + "contig": "Skin_28-C2", + "target": "sp|A5INP9|HUTH_STAA9" + }, + { + "sample": "Skin_28", + "orf": "C2-G9", + "contig": "Skin_28-C2", + "target": "sp|A8YYT2|SYS_STAAT" + }, + { + "sample": "Skin_28", + "orf": "C2-G14", + "contig": "Skin_28-C2", + "target": "sp|P37484|GDPP_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C2-G16", + "contig": "Skin_28-C2", + "target": "sp|Q9X4C9|DNAB_GEOSE" + }, + { + "sample": "Skin_28", + "orf": "C2-G17", + "contig": "Skin_28-C2", + "target": "sp|Q5HJX8|PURA_STAAC" + }, + { + "sample": "Skin_28", + "orf": "C2-G19", + "contig": "Skin_28-C2", + "target": "sp|A6QD58|WALK_STAAE" + }, + { + "sample": "Skin_28", + "orf": "C2-G22", + "contig": "Skin_28-C2", + "target": "sp|A0A4Y1WBN6|YYCJ_BACAN" + }, + { + "sample": "Skin_28", + "orf": "C2-G26", + "contig": "Skin_28-C2", + "target": "sp|Q6GD38|DUS_STAAS" + }, + { + "sample": "Skin_28", + "orf": "C2-G31", + "contig": "Skin_28-C2", + "target": "sp|Q9Y6N5|SQOR_HUMAN" + }, + { + "sample": "Skin_28", + "orf": "C2-G32", + "contig": "Skin_28-C2", + "target": "sp|Q8NYV4|DUS_STAAW" + }, + { + "sample": "Skin_28", + "orf": "C2-G36", + "contig": "Skin_28-C2", + "target": "sp|P96726|YWQN_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C3-G1", + "contig": "Skin_28-C3", + "target": "sp|P39062|ACSA_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C3-G2", + "contig": "Skin_28-C3", + "target": "sp|Q65G33|ACUA_BACLD" + }, + { + "sample": "Skin_28", + "orf": "C3-G3", + "contig": "Skin_28-C3", + "target": "sp|Q5HF39|ACUC_STAAC" + }, + { + "sample": "Skin_28", + "orf": "C3-G4", + "contig": "Skin_28-C3", + "target": "sp|Q8NW33|CCPA_STAAW" + }, + { + "sample": "Skin_28", + "orf": "C3-G5", + "contig": "Skin_28-C3", + "target": "sp|P39912|AROG_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C3-G7", + "contig": "Skin_28-C3", + "target": "sp|Q7A0M4|Y1682_STAAW" + }, + { + "sample": "Skin_28", + "orf": "C3-G8", + "contig": "Skin_28-C3", + "target": "sp|P65476|MURC_STAAW" + }, + { + "sample": "Skin_28", + "orf": "C3-G10", + "contig": "Skin_28-C3", + "target": "sp|O34943|YTPR_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C3-G11", + "contig": "Skin_28-C3", + "target": "sp|Q7A528|Y1564_STAAN" + }, + { + "sample": "Skin_28", + "orf": "C3-G12", + "contig": "Skin_28-C3", + "target": "sp|O34357|YTPP_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C3-G13", + "contig": "Skin_28-C3", + "target": "sp|O34924|YTOP_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C3-G15", + "contig": "Skin_28-C3", + "target": "sp|O34760|YTNP_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C3-G16", + "contig": "Skin_28-C3", + "target": "sp|Q2FFZ0|TRMB_STAA3" + }, + { + "sample": "Skin_28", + "orf": "C3-G17", + "contig": "Skin_28-C3", + "target": "sp|C0SPC1|CCRZ_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C3-G18", + "contig": "Skin_28-C3", + "target": "sp|Q5HF24|DAAA_STAAC" + }, + { + "sample": "Skin_28", + "orf": "C3-G19", + "contig": "Skin_28-C3", + "target": "sp|Q7A522|PEPVL_STAAN" + }, + { + "sample": "Skin_28", + "orf": "C3-G21", + "contig": "Skin_28-C3", + "target": "sp|O32068|YTZG_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C3-G23", + "contig": "Skin_28-C3", + "target": "sp|Q795R8|YTFP_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C3-G26", + "contig": "Skin_28-C3", + "target": "sp|A6U2M4|SYL_STAA2" + }, + { + "sample": "Skin_28", + "orf": "C3-G27", + "contig": "Skin_28-C3", + "target": "sp|O34546|YTTB_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C3-G28", + "contig": "Skin_28-C3", + "target": "sp|O35008|YTQA_BACSU" + }, + { + "sample": "Skin_28", + "orf": "C3-G29", + "contig": "Skin_28-C3", + "target": "sp|Q2FXG9|MNMM_STAA8" + }, + { + "sample": "Skin_28", + "orf": "C3-G30", + "contig": "Skin_28-C3", + "target": "sp|Q7A514|ROT_STAAN" + }, + { + "sample": "Skin_28", + "orf": "C3-G35", + "contig": "Skin_28-C3", + "target": "sp|A8Z4J3|RISB_STAAT" + }, + { + "sample": "Skin_28", + "orf": "C3-G36", + "contig": "Skin_28-C3", + "target": "sp|Q6G8G1|RIBBA_STAAS" + }, + { + "sample": "Skin_28", + "orf": "C3-G37", + "contig": "Skin_28-C3", + "target": "sp|P50854|RISA_ACTPL" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G1", + "contig": "Urogenital_22-C1", + "target": "sp|Q04CX5|DNAA_LACDB" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G2", + "contig": "Urogenital_22-C1", + "target": "sp|O06672|DPO3B_STRPN" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G4", + "contig": "Urogenital_22-C1", + "target": "sp|Q1GC40|RECF_LACDA" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G5", + "contig": "Urogenital_22-C1", + "target": "sp|Q839Z1|GYRB_ENTFA" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G6", + "contig": "Urogenital_22-C1", + "target": "sp|O50628|GYRA_HALH5" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G7", + "contig": "Urogenital_22-C1", + "target": "sp|A8YW47|RS6_LACH4" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G8", + "contig": "Urogenital_22-C1", + "target": "sp|Q890K1|SSB_LACPL" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G9", + "contig": "Urogenital_22-C1", + "target": "sp|Q04CW7|RS18_LACDB" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G10", + "contig": "Urogenital_22-C1", + "target": "sp|A0A0H3GCG4|PDEA_LISM4" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G11", + "contig": "Urogenital_22-C1", + "target": "sp|Q1GC33|RL9_LACDA" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G12", + "contig": "Urogenital_22-C1", + "target": "sp|Q9X4C9|DNAB_GEOSE" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G21", + "contig": "Urogenital_22-C1", + "target": "sp|P77307|FETB_ECOLI" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G32", + "contig": "Urogenital_22-C1", + "target": "sp|Q830C9|Y2859_ENTFA" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G38", + "contig": "Urogenital_22-C1", + "target": "sp|Q59935|MANA_STRMU" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G39", + "contig": "Urogenital_22-C1", + "target": "sp|P82371|SCRK_LACLC" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G46", + "contig": "Urogenital_22-C1", + "target": "sp|F9UMX3|GLPF4_LACPL" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G48", + "contig": "Urogenital_22-C1", + "target": "sp|P10249|SUCP_STRMU" + }, + { + "sample": "Urogenital_22", + "orf": "C1-G49", + "contig": "Urogenital_22-C1", + "target": "sp|Q59495|SUCP_LEUME" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G1", + "contig": "Urogenital_22-C2", + "target": "sp|Q9L7L7|DNAA_MYCPA" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G2", + "contig": "Urogenital_22-C2", + "target": "sp|Q9L7L6|DPO3B_MYCPA" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G3", + "contig": "Urogenital_22-C2", + "target": "sp|Q9L7L5|RECF_MYCPA" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G4", + "contig": "Urogenital_22-C2", + "target": "sp|Q9L7L4|Y004_MYCPA" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G5", + "contig": "Urogenital_22-C2", + "target": "sp|Q9L7L3|GYRB_MYCPA" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G6", + "contig": "Urogenital_22-C2", + "target": "sp|P9WG46|GYRA_MYCTO" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G7", + "contig": "Urogenital_22-C2", + "target": "sp|P9WMA7|Y007_MYCTU" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G16", + "contig": "Urogenital_22-C2", + "target": "sp|P9WPL1|CP144_MYCTU" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G17", + "contig": "Urogenital_22-C2", + "target": "sp|P9WJF2|CWSA_MYCTO" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G18", + "contig": "Urogenital_22-C2", + "target": "sp|P19834|YI11_STRCL" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G19", + "contig": "Urogenital_22-C2", + "target": "sp|P65763|PPIA_MYCBO" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G20", + "contig": "Urogenital_22-C2", + "target": "sp|P9WMA2|Y010_MYCTO" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G21", + "contig": "Urogenital_22-C2", + "target": "sp|A0Q8S8|CRGA_MYCA1" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G23", + "contig": "Urogenital_22-C2", + "target": "sp|P9WN34|TRPG_MYCTO" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G24", + "contig": "Urogenital_22-C2", + "target": "sp|P54744|PKNB_MYCLE" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G25", + "contig": "Urogenital_22-C2", + "target": "sp|A5TY85|PKNA_MYCTA" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G26", + "contig": "Urogenital_22-C2", + "target": "sp|P9WKD0|PBPA_MYCTO" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G27", + "contig": "Urogenital_22-C2", + "target": "sp|P9WN99|RODA_MYCTU" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G28", + "contig": "Urogenital_22-C2", + "target": "sp|P9WHW4|PSTP_MYCTO" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G29", + "contig": "Urogenital_22-C2", + "target": "sp|P9WJB4|FHAB_MYCTO" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G30", + "contig": "Urogenital_22-C2", + "target": "sp|P71590|FHAA_MYCTU" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G32", + "contig": "Urogenital_22-C2", + "target": "sp|P63453|MBTL_MYCBO" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G33", + "contig": "Urogenital_22-C2", + "target": "sp|P0A4X9|MBTM_MYCBO" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G41", + "contig": "Urogenital_22-C2", + "target": "sp|P0DKR7|WHB5B_MYCTO" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G42", + "contig": "Urogenital_22-C2", + "target": "sp|P67705|Y023_MYCBO" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G44", + "contig": "Urogenital_22-C2", + "target": "sp|P9WMA1|Y025_MYCTU" + }, + { + "sample": "Urogenital_22", + "orf": "C2-G45", + "contig": "Urogenital_22-C2", + "target": "sp|P9WMB1|Y0026_MYCTU" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G1", + "contig": "Urogenital_22-C3", + "target": "sp|Q03UE4|DNAA_LEVBA" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G2", + "contig": "Urogenital_22-C3", + "target": "sp|P05649|DPO3B_BACSU" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G3", + "contig": "Urogenital_22-C3", + "target": "sp|P05650|RLBA_BACSU" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G4", + "contig": "Urogenital_22-C3", + "target": "sp|Q1WVP2|RECF_LIGS1" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G5", + "contig": "Urogenital_22-C3", + "target": "sp|Q839Z1|GYRB_ENTFA" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G6", + "contig": "Urogenital_22-C3", + "target": "sp|P0DG03|GYRA_STRPQ" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G7", + "contig": "Urogenital_22-C3", + "target": "sp|Q03UD8|RS6_LEVBA" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G8", + "contig": "Urogenital_22-C3", + "target": "sp|Q890K1|SSB_LACPL" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G9", + "contig": "Urogenital_22-C3", + "target": "sp|Q1WVN7|RS18_LIGS1" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G10", + "contig": "Urogenital_22-C3", + "target": "sp|O34948|YKWC_BACSU" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G14", + "contig": "Urogenital_22-C3", + "target": "sp|A0A086F3E3|TM175_CHRP1" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G15", + "contig": "Urogenital_22-C3", + "target": "sp|Q1WVR1|MSCL_LIGS1" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G18", + "contig": "Urogenital_22-C3", + "target": "sp|Q1WVQ8|MURE_LIGS1" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G19", + "contig": "Urogenital_22-C3", + "target": "sp|H8L902|ASL_ENTFU" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G20", + "contig": "Urogenital_22-C3", + "target": "sp|H8L901|RACD_ENTFU" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G21", + "contig": "Urogenital_22-C3", + "target": "sp|E3GC98|MDTG_ENTLS" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G23", + "contig": "Urogenital_22-C3", + "target": "sp|Q9ZF37|YSUP_LACHE" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G28", + "contig": "Urogenital_22-C3", + "target": "sp|P56067|CYSM_HELPY" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G29", + "contig": "Urogenital_22-C3", + "target": "sp|P0C2T9|METC_LACLC" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G31", + "contig": "Urogenital_22-C3", + "target": "sp|Q1WVT3|MSRB_LIGS1" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G39", + "contig": "Urogenital_22-C3", + "target": "sp|Q5XD45|Y533_STRP6" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G40", + "contig": "Urogenital_22-C3", + "target": "sp|P37478|WALR_BACSU" + }, + { + "sample": "Urogenital_22", + "orf": "C3-G46", + "contig": "Urogenital_22-C3", + "target": "sp|A0A0H2ZQT7|WALJ_STRP2" + } + ], + "fixture_sha256": "65352f9ae580c02293093f2e59057707ae1d1a1458c3a8c48bf5b9de5ac38d3d", + "source_release": "UniProtKB/Swiss-Prot Release 2026_01 of 28-Jan-2026" +} diff --git a/metapathways/report_assets/EDA_portal.html b/metapathways/report_assets/EDA_portal.html new file mode 100644 index 0000000..1d2ec24 --- /dev/null +++ b/metapathways/report_assets/EDA_portal.html @@ -0,0 +1,16 @@ + + +MetaPathways · EDA portal +

METAPATHWAYS / RESULTS

Explore and export

Follow the links between genes, annotations, MAGs and pathways. Build a subset for your next analysis.

+
+

+
+
+
Filter by linked results

Exact identifiers. These filters keep one row per record in the selected table. EC, reaction and database must occur on the same annotation. A pathway and entity must occur on the same pathway association. An entity alone selects its explicit MAG input genes.

+
Choose export columns
+

Filters combine with AND. Text “contains” is case-insensitive; identifiers use exact equality. Apply changes before exporting.

+
Connecting to the local database…
+
+ +

CSV export includes every matching row and your selected columns, not just this page. Missing values stay blank. Formula-like text is prefixed with an apostrophe for spreadsheet safety. Counts in different tables describe different units; do not sum duplicated annotation or pathway rows as gene counts.

+
diff --git a/metapathways/report_assets/portal.css b/metapathways/report_assets/portal.css new file mode 100644 index 0000000..62183aa --- /dev/null +++ b/metapathways/report_assets/portal.css @@ -0,0 +1 @@ +:root{font-family:system-ui,-apple-system,sans-serif;color:#20353d;background:#f4f7f8;font-size:15px}*{box-sizing:border-box}body{margin:0}header{padding:2.5rem 4vw;background:#123e48;color:white;display:flex;justify-content:space-between;gap:2rem;flex-wrap:wrap}h1{font-size:2.2rem;margin:.25rem 0}header p{max-width:760px}header nav{display:flex;gap:1.2rem;align-items:center;flex-wrap:wrap}header a{color:#d3f3f4}.eyebrow{letter-spacing:.15em;font-size:.75rem}main{padding:1.5rem 4vw}section{background:white;border:1px solid #dbe3e6;border-radius:8px;padding:1.4rem;margin-bottom:1.3rem}label{display:flex;flex-direction:column;gap:.4rem;font-weight:600}input,select,button{font:inherit;padding:.55rem .7rem;border:1px solid #a7b9bf;border-radius:4px;background:white;color:inherit}button{cursor:pointer}button:hover:not(:disabled){background:#e3f1f3}button:disabled{opacity:.5;cursor:default}button.primary{background:#086b78;color:white;border-color:#086b78}input[type=search]{min-width:300px}.toolbar{display:flex;gap:.7rem;align-items:end;flex-wrap:wrap;margin:.7rem 0}.toolbar strong{margin-right:auto;align-self:center}.filter{display:flex;gap:.6rem;align-items:center;margin:.7rem 0;flex-wrap:wrap}.related,#columns{display:flex;gap:.9rem;flex-wrap:wrap;padding:1rem 0}#columns label{flex-direction:row;align-items:center;font-weight:400}details{padding:.8rem 0;border-bottom:1px solid #dbe3e6}summary{cursor:pointer;font-weight:600}.hint,footer{font-size:.85rem;color:#526a74;line-height:1.6}.table-scroll{overflow:auto;max-height:65vh;border:1px solid #dbe3e6}table{width:100%;border-collapse:collapse;font-size:.85rem}td,th{border-bottom:1px solid #dbe3e6;padding:.65rem;text-align:left;vertical-align:top;min-width:100px;max-width:440px;overflow-wrap:anywhere}th{background:#ecf3f4;position:sticky;top:0;white-space:nowrap}th button{border:none;background:transparent;padding:0;font-weight:600}td.actions{min-width:170px}td.actions button,td.actions a{display:block;font-size:.78rem;padding:.25rem;margin:.15rem 0}a{color:#006a80}tr:nth-child(even){background:#f8fafb}.pagination{justify-content:center}#offline{border-left:5px solid #d89823}pre{white-space:pre-wrap}button:focus-visible,a:focus-visible,input:focus-visible,select:focus-visible{outline:3px solid #e4aa43;outline-offset:2px}.error{color:#a12928} [hidden]{display:none!important}@media(max-width:600px){header{padding:1.5rem}main{padding:1rem}input[type=search]{min-width:0}section{padding:.9rem}} diff --git a/metapathways/report_assets/portal.js b/metapathways/report_assets/portal.js new file mode 100644 index 0000000..09e16bd --- /dev/null +++ b/metapathways/report_assets/portal.js @@ -0,0 +1,112 @@ +'use strict'; +const $ = id => document.getElementById(id); +let metadata, active, offset = 0, total = 0, sort = '', descending = false, serial = 0; +const relatedKeys = ['pathway_id','entity_id','ec','reaction','reference_db']; +function element(tag, text) { const e=document.createElement(tag); if(text!==undefined)e.textContent=text; return e; } +function columns() { return metadata.views[$('table').value].columns.map(c=>c.name); } +function button(text, action) { const b=element('button',text); b.type='button'; b.onclick=action; return b; } +function api(path) { return new URL('../api/'+path, location.href).href; } +function queryURL(kind, spec) { return api(kind)+'?spec='+encodeURIComponent(JSON.stringify(spec)); } +function option(value, label=value) { const o=element('option',label); o.value=value; return o; } +function filterRow(column=columns()[0],op='contains',value='') { + const row=element('div'); row.className='filter'; + const c=element('select'); c.setAttribute('aria-label','Filter column'); columns().forEach(x=>c.append(option(x))); c.value=column; + const operator=element('select'); operator.setAttribute('aria-label','Comparison'); + Object.entries({contains:'contains',eq:'equals',ne:'does not equal',ge:'≥',le:'≤',gt:'>',lt:'<',missing:'is missing',present:'is present'}).forEach(([k,v])=>operator.append(option(k,v))); operator.value=op; + const input=element('input'); input.value=value; input.setAttribute('aria-label','Filter value'); + row.append(c,operator,input,button('Remove',()=>row.remove())); $('filters').append(row); +} +function configure(table, filters=[]) { + if(!metadata.views[table])table='samples'; + $('table').value=table; $('description').textContent=metadata.views[table].description; + $('filters').replaceChildren(); $('columns').replaceChildren(); $('search').value=''; + relatedKeys.forEach(k=>$('related_'+k).value=''); + $('relatedPanel').hidden=!(columns().includes('sample_id')&&columns().includes('orf_id')); + columns().forEach(c=>{const label=element('label');const box=element('input');box.type='checkbox';box.value=c;box.checked=true;label.append(box,document.createTextNode(c));$('columns').append(label);}); + filters.forEach(f=>filterRow(f.column,f.op,f.value)); sort='';descending=false;offset=0; +} +function specFromControls() { + const filters=Array.from($('filters').children).map(row=>({column:row.children[0].value,op:row.children[1].value,value:row.children[2].value})); + const related={}; if(!$('relatedPanel').hidden)relatedKeys.forEach(k=>{if($('related_'+k).value)related[k]=$('related_'+k).value;}); + const selected=Array.from($('columns').querySelectorAll('input:checked')).map(i=>i.value); + if(!selected.length)throw Error('Select at least one column.'); + return {table:$('table').value,search:$('search').value,filters,related,columns:selected,sort,descending,limit:100}; +} +async function apply(reset=true) { + const id=++serial; + $('status').className='';$('status').textContent='Searching…';$('export').disabled=true;$('saveQuery').disabled=true;$('previous').disabled=true;$('next').disabled=true; + try { + const spec=reset?specFromControls():{...active}; if(reset)offset=0; + const response=await fetch(queryURL('query',{...spec,columns:[],offset})); + const data=await response.json(); if(!response.ok)throw Error(data.error||'Query failed'); if(id!==serial)return; + active=spec; total=data.total; + render(data.rows,spec.columns); + $('status').textContent=total.toLocaleString()+' matching rows'; + $('page').textContent=total?`${offset+1}–${Math.min(offset+data.rows.length,total)} of ${total.toLocaleString()}`:'No matches'; + $('previous').disabled=offset===0;$('next').disabled=offset+100>=total; + $('export').disabled=false;$('saveQuery').disabled=false; + history.replaceState(null,'','#spec='+encodeURIComponent(JSON.stringify(spec))); + } catch(error) { if(id!==serial)return;$('status').className='error';$('status').textContent=error.message;$('previous').disabled=true;$('next').disabled=true; } +} +function drill(table, fields) { + const valid=metadata.views[table].columns.map(c=>c.name); + const filters=Object.entries(fields).filter(([k,v])=>valid.includes(k)&&v!==null&&v!==undefined).map(([column,value])=>({column,op:'eq',value})); + configure(table,filters);apply(); +} +function actions(row) { + const cell=element('td'); cell.className='actions'; + const scope={sample_id:row.sample_id}; + if(row.orf_id){ + cell.append(button('ORF annotations',()=>drill('annotation_explorer',{...scope,orf_id:row.orf_id})),button('Pathway links',()=>drill('pathway_gene_explorer',{...scope,orf_id:row.orf_id})),button('ORF abundance',()=>drill('abundance_explorer',{...scope,feature_type:'orf',feature_id:row.orf_id}))); + } + if(row.pathway_id)cell.append(button('Pathway genes',()=>drill('pathway_gene_explorer',{...scope,entity_id:row.entity_id,pathway_id:row.pathway_id}))); + if(row.contig_id)cell.append(button('Contig ORFs',()=>drill('orf_explorer',{...scope,contig_id:row.contig_id}))); + if(row.entity_id)cell.append(button('All mapped MAG ORFs',()=>drill('mag_orf_explorer',{...scope,entity_id:row.entity_id})),button('Entity pathways',()=>drill('pathway_explorer',{...scope,entity_id:row.entity_id})),button('MAG input genes',()=>drill('mag_gene_explorer',{...scope,entity_id:row.entity_id}))); + if(row.representative_orf_id)cell.append(button('Group members',()=>drill('orf_groups',{...scope,representative_orf_id:row.representative_orf_id}))); + if(row.annotation_id!==undefined)cell.append(button('EC / reaction terms',()=>drill('annotation_terms',{annotation_id:row.annotation_id}))); + if(row.source_id!==undefined)cell.append(button('Source record',()=>drill('sources',{source_id:row.source_id}))); + if(row.sample_id&&!row.orf_id&&!row.contig_id&&!row.pathway_id&&!row.entity_id)cell.append(button('Sample ORFs',()=>drill('orf_explorer',scope))); + if(row.path||row.source){ + const path=row.path||row.source; + if(typeof path==='string'&&!path.startsWith('/')&&!path.split('/').includes('..')) { + const a=element('a','Open source file');a.href='../'+path.split('/').map(encodeURIComponent).join('/');a.target='_blank';a.rel='noopener';cell.append(a); + } + } + return cell; +} +function render(rows,selected) { + const head=$('results').querySelector('thead'),body=$('results').querySelector('tbody'); head.replaceChildren();body.replaceChildren(); + const hr=element('tr');hr.append(element('th','Related results')); + selected.forEach(c=>{const th=element('th');th.append(button(c+(sort===c?(descending?' ↓':' ↑'):''),()=>{descending=sort===c?!descending:false;sort=c;apply();}));hr.append(th);});head.append(hr); + rows.forEach(row=>{const tr=element('tr');tr.append(actions(row));selected.forEach(c=>tr.append(element('td',row[c]===null?'':String(row[c]))));body.append(tr);}); +} +function saveJSON() { + const blob=new Blob([JSON.stringify({schema_version:metadata.schema_version,report_generated_utc:metadata.generated_utc,query:active},null,2)],{type:'application/json'}); + const link=element('a');link.href=URL.createObjectURL(blob);link.download='metapathways-query.json';link.click();setTimeout(()=>URL.revokeObjectURL(link.href),1000); +} +function fromHash() { + const params=new URLSearchParams(location.hash.slice(1)); + try { + const spec=params.has('spec')?JSON.parse(params.get('spec')):null; + configure(spec?.table||params.get('table')||'samples',spec?.filters||[]); + if(spec){$('search').value=spec.search||'';relatedKeys.forEach(k=>$('related_'+k).value=spec.related?.[k]||'');sort=spec.sort||'';descending=!!spec.descending; if(spec.columns)$('columns').querySelectorAll('input').forEach(i=>i.checked=spec.columns.includes(i.value));} + apply(); + } catch(error){$('status').textContent='Cannot read this saved query: '+error.message;} +} +async function init() { + if(location.protocol==='file:'){$('offline').hidden=false;$('status').textContent='Start the local portal to search these results.';document.querySelector('.controls').hidden=true;return;} + try { + const response=await fetch(api('meta'));if(!response.ok)throw Error('Open the URL printed by metapathways report --serve.');metadata=await response.json(); + Object.entries(metadata.views).forEach(([key,value])=>$('table').append(option(key,`${value.label} (${value.rows.toLocaleString()})`))); + const run=metadata.run_details||{}; + const details=[run.mp_version?'MP '+run.mp_version:'Report built with MP '+metadata.report_mp_version, + metadata.sample_paths.length+' samples',run.command,run.executor,run.status].filter(Boolean); + $('runDetails').textContent=details.join(' · '); + $('generated').textContent='Report updated '+new Date(metadata.generated_utc).toLocaleString(); + $('table').onchange=()=>{configure($('table').value);apply();};$('addFilter').onclick=()=>filterRow();$('apply').onclick=()=>apply();$('reset').onclick=()=>{configure($('table').value);apply();}; + $('search').onkeydown=e=>{if(e.key==='Enter')apply();};$('previous').onclick=()=>{offset=Math.max(0,offset-100);apply(false);};$('next').onclick=()=>{offset+=100;apply(false);}; + $('export').onclick=()=>{const a=element('a');a.href=queryURL('export',active);a.download='metapathways-subset.csv';a.click();};$('saveQuery').onclick=saveJSON; + window.addEventListener('hashchange',fromHash);fromHash(); + }catch(error){$('status').className='error';$('status').textContent=error.message;} +} +init(); diff --git a/metapathways/report_server.py b/metapathways/report_server.py new file mode 100644 index 0000000..cbee801 --- /dev/null +++ b/metapathways/report_server.py @@ -0,0 +1,258 @@ +"""Loopback-only, read-only SQL queries and streamed CSV exports for MP reports.""" +import argparse +from contextlib import closing +import csv +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +import io +import json +from pathlib import Path +import secrets +import sqlite3 +import threading +import time +from urllib.parse import parse_qs, unquote, urlsplit, quote +import webbrowser + +from metapathways.reporting import VIEWS, build_report + + +def identifier(value): + return '"' + value.replace('"', '""') + '"' + + +def connect(database): + db = sqlite3.connect(database.as_uri() + '?mode=ro', uri=True) + db.row_factory = sqlite3.Row + db.execute('PRAGMA query_only=ON') + deadline = time.monotonic() + 120 + db.set_progress_handler(lambda: int(time.monotonic() > deadline), 10000) + return db + + +def query(db, spec, export=False): + table = spec.get('table', 'samples') + if table not in VIEWS: + raise ValueError('Unknown table') + details = list(db.execute(f'PRAGMA table_info({identifier(table)})')) + columns = [r[1] for r in details] + numeric = {r[1] for r in details if r[2] in ('REAL','INTEGER')} | {'linked_orf_count'} + selected = spec.get('columns') or columns + if not isinstance(selected, list) or len(selected) > len(columns) or any(c not in columns for c in selected): + raise ValueError('Unknown selected column') + where, values = [], [] + search = spec.get('search', '') + if not isinstance(search, str) or len(search) > 1000: + raise ValueError('Search is too long') + if search: + where.append('(' + ' OR '.join(f'instr(lower(CAST(t.{identifier(c)} AS TEXT)),lower(?))>0' for c in columns) + ')') + values.extend([search] * len(columns)) + filters = spec.get('filters', []) + if not isinstance(filters, list) or len(filters) > 30: + raise ValueError('At most 30 filters are supported') + for f in filters: + column, op, value = f.get('column'), f.get('op'), f.get('value', '') + if column not in columns: + raise ValueError('Unknown filter column') + expr = 't.' + identifier(column) + if op == 'missing': + where.append(f'({expr} IS NULL OR {expr}=\'\')'); continue + if op == 'present': + where.append(f'({expr} IS NOT NULL AND {expr}!=\'\')'); continue + if not isinstance(value, (str, int, float)) or len(str(value)) > 1000: + raise ValueError('Invalid filter value') + if op == 'contains': + where.append(f'instr(lower(CAST({expr} AS TEXT)),lower(?))>0') + elif op in ('eq','ne'): + where.append(f'{expr} {"=" if op == "eq" else "!="} ?') + elif op in ('lt','le','gt','ge'): + if column not in numeric: + raise ValueError('Numeric comparisons require a numeric column') + try: + value = float(value) + except (ValueError, TypeError): + raise ValueError('Numeric comparison requires a number') + where.append(f'CAST({expr} AS REAL) ' + {'lt':'<','le':'<=','gt':'>','ge':'>='}[op] + ' ?') + else: + raise ValueError('Unknown filter operator') + values.append(value) + # Related filters use EXISTS: keep one ORF/annotation row rather than multiply it. + related = spec.get('related') or {} + if not isinstance(related, dict) or set(related) - {'pathway_id','entity_id','ec','reaction','reference_db'}: + raise ValueError('Unknown related filter') + if related: + if not {'sample_id','orf_id'} <= set(columns): + raise ValueError('Related filters require an ORF-based table') + if any(not isinstance(v, str) or len(v) > 1000 for v in related.values()): + raise ValueError('Invalid related filter') + if related.get('pathway_id'): + expression = 'EXISTS(SELECT 1 FROM pathway_orfs p WHERE p.sample_id=t.sample_id AND p.orf_id=t.orf_id AND p.pathway_id=?' + values.append(related['pathway_id']) + if related.get('entity_id'): + expression += ' AND p.entity_id=?'; values.append(related['entity_id']) + where.append(expression + ')') + elif related.get('entity_id'): + where.append('EXISTS(SELECT 1 FROM entity_orfs e WHERE e.sample_id=t.sample_id AND e.orf_id=t.orf_id AND e.entity_id=?)') + values.append(related['entity_id']) + if any(related.get(k) for k in ('ec','reaction','reference_db')): + expression = 'EXISTS(SELECT 1 FROM annotations a WHERE a.sample_id=t.sample_id AND a.orf_id=t.orf_id' + if related.get('reference_db'): + expression += ' AND a.reference_db=?'; values.append(related['reference_db']) + for kind in ('ec','reaction'): + if related.get(kind): + expression += ' AND EXISTS(SELECT 1 FROM annotation_terms x WHERE x.annotation_id=a.annotation_id AND x.term_type=? AND x.term=?)' + values.extend(['EC' if kind == 'ec' else 'reaction', related[kind]]) + where.append(expression + ')') + base = ' FROM ' + identifier(table) + ' t' + (' WHERE ' + ' AND '.join(where) if where else '') + sort = spec.get('sort') or columns[0] + if sort not in columns: + raise ValueError('Unknown sort column') + order = ' ORDER BY t.' + identifier(sort) + (' DESC' if spec.get('descending') else ' ASC') + for column in ('sample_id','orf_id','entity_id','pathway_id','annotation_id','source_id'): + if column in columns and column != sort: + order += ',t.' + identifier(column) + sql = 'SELECT ' + ','.join('t.'+identifier(c) for c in selected) + base + order + if export: + return selected, db.execute(sql, values) + limit, offset = int(spec.get('limit',100)), int(spec.get('offset',0)) + if not 1 <= limit <= 500 or offset < 0: + raise ValueError('Invalid page limits') + total = db.execute('SELECT COUNT(*)' + base, values).fetchone()[0] + result = db.execute(sql + ' LIMIT ? OFFSET ?', values+[limit,offset]) + return dict(columns=selected, rows=[dict(r) for r in result], total=total, offset=offset, limit=limit) + + +def csv_value(value): + # Keep numeric negatives numeric, and prevent text from becoming spreadsheet formulas. + if isinstance(value, str) and value.lstrip().startswith(('=', '+', '-', '@')): + return "'" + value + return value + + +class ReportServer(ThreadingHTTPServer): + daemon_threads = True + def __init__(self, root, port=0): + self.root = Path(root).resolve() + self.token = secrets.token_urlsafe(24) + self.slots = threading.BoundedSemaphore(4) + super().__init__(('127.0.0.1', port), Handler) + + @property + def url(self): + return f'http://127.0.0.1:{self.server_port}/{self.token}/reports/EDA_portal.html' + + +class Handler(BaseHTTPRequestHandler): + def log_message(self, fmt, *args): + # Queries can include scientific identifiers; keep the terminal concise. + if len(args) > 1 and str(args[1]).startswith(('4','5')): + print('Portal request failed:', args[1], flush=True) + + def send(self, value, kind='application/json', status=200): + raw = json.dumps(value).encode() if kind == 'application/json' else value + self.send_response(status) + self.send_header('Content-Type', kind) + self.send_header('Content-Length', str(len(raw))) + self.send_header('Cache-Control','no-store') + self.send_header('X-Content-Type-Options','nosniff') + self.end_headers(); self.wfile.write(raw) + + def do_GET(self): + try: + self.handle_get() + except (BrokenPipeError, ConnectionResetError): + pass + except (ValueError, KeyError, TypeError, sqlite3.Error, OSError) as exc: + if getattr(self, '_response_started', False): + self.close_connection = True + else: + self.send(dict(error=str(exc)), status=400) + + def send_response(self, code, message=None): + self._response_started = True + super().send_response(code, message) + + def handle_get(self): + expected = f'127.0.0.1:{self.server.server_port}' + if self.headers.get('Host') != expected or self.headers.get('Origin') not in (None, f'http://{expected}'): + self.send(dict(error='Local same-origin access only'), status=403); return + parsed = urlsplit(self.path) + prefix = '/' + self.server.token + '/' + if not parsed.path.startswith(prefix): + self.send(dict(error='Not found'), status=404); return + path = unquote(parsed.path[len(prefix):]) + root = self.server.root + if path == 'api/meta': + self.send(json.loads((root/'reports/schema.json').read_text())); return + if path in ('api/query','api/export'): + raw = parse_qs(parsed.query).get('spec',['{}'])[0] + if len(raw) > 20000: + raise ValueError('Query is too large') + spec = json.loads(raw) + if not isinstance(spec, dict): + raise ValueError('Query must be an object') + if not self.server.slots.acquire(blocking=False): + self.send(dict(error='Four queries are already active; try again shortly'), status=429); return + try: + with closing(connect(root/'reports/results.sqlite')) as db: + if path == 'api/query': + self.send(query(db, spec)); return + columns, cursor = query(db, spec, export=True) + self.send_response(200) + self.send_header('Content-Type','text/csv; charset=utf-8') + self.send_header('Content-Disposition','attachment; filename="metapathways-subset.csv"') + self.send_header('Cache-Control','no-store') + self.end_headers() + buffer=io.StringIO(); writer=csv.writer(buffer) + writer.writerow(columns) + for row in cursor: + writer.writerow([csv_value(v) for v in row]) + if buffer.tell() > 65536: + self.wfile.write(buffer.getvalue().encode()); buffer.seek(0); buffer.truncate() + self.wfile.write(buffer.getvalue().encode()) + finally: + self.server.slots.release() + return + target=(root/path).resolve() + try: + relative=target.relative_to(root) + except ValueError: + self.send(dict(error='Not found'), status=404); return + if any(p.startswith('.') for p in relative.parts) or not target.is_file(): + self.send(dict(error='Not found'), status=404); return + if relative.parts[0] != 'reports': + with closing(connect(root/'reports/results.sqlite')) as db: + if not db.execute('SELECT 1 FROM files WHERE path=?',(relative.as_posix(),)).fetchone(): + self.send(dict(error='Not inventoried'), status=404); return + import mimetypes + self.send_response(200) + self.send_header('Content-Type',mimetypes.guess_type(target.name)[0] or 'application/octet-stream') + self.send_header('Content-Length',str(target.stat().st_size)) + self.send_header('X-Content-Type-Options','nosniff') + self.end_headers() + with target.open('rb') as stream: + for block in iter(lambda: stream.read(65536), b''): + self.wfile.write(block) + + +def main(argv=None): + parser=argparse.ArgumentParser(description='Build a navigable report from existing MP outputs; no analyses are rerun.') + parser.add_argument('-o','--output_dir',required=True,help='sample output or parent containing sample outputs') + parser.add_argument('--serve',action='store_true',help='open the local searchable portal and serve until Ctrl-C') + parser.add_argument('--no-rebuild',action='store_true',help='use an existing report snapshot') + parser.add_argument('--port',type=int,default=0,help='local port [automatically selected]') + parser.add_argument('--no-browser',action='store_true',help='print the local URL without opening a browser') + args=parser.parse_args(argv) + root=Path(args.output_dir).expanduser().resolve() + if not args.no_rebuild: + build_report(root) + if not (root/'reports/results.sqlite').is_file(): + parser.error('No report database exists; omit --no-rebuild') + if args.serve: + with ReportServer(root,args.port) as server: + print(f'EDA portal: {server.url}\nLocal access only. Ctrl-C stops the portal.',flush=True) + if not args.no_browser: + webbrowser.open(server.url) + try: + server.serve_forever() + except KeyboardInterrupt: + print('Portal stopped.',flush=True) diff --git a/metapathways/reporting.py b/metapathways/reporting.py new file mode 100644 index 0000000..cd45b50 --- /dev/null +++ b/metapathways/reporting.py @@ -0,0 +1,508 @@ +"""Read existing MP results into a sample-scoped, queryable report database.""" +import csv +from contextlib import closing +import fcntl +import hashlib +import html +import json +import os +from pathlib import Path +import shutil +import sqlite3 +import tempfile +from datetime import datetime, timezone +from urllib.parse import quote + +SCHEMA_VERSION = 3 +SCHEMA = ''' +PRAGMA foreign_keys=ON; +CREATE TABLE samples(sample_id TEXT PRIMARY KEY, output_path TEXT NOT NULL); +CREATE TABLE sources(source_id INTEGER PRIMARY KEY, path TEXT UNIQUE, bytes INTEGER, sha256 TEXT, role TEXT); +CREATE TABLE contigs(sample_id TEXT, contig_id TEXT, original_id TEXT, length INTEGER, + PRIMARY KEY(sample_id,contig_id), FOREIGN KEY(sample_id) REFERENCES samples); +CREATE TABLE orfs(sample_id TEXT, orf_id TEXT, contig_id TEXT, length INTEGER, start INTEGER, end INTEGER, + strand TEXT, target TEXT, product TEXT, taxonomy TEXT, reference_db TEXT, lca_taxonomy TEXT, annotation_present INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY(sample_id,orf_id), FOREIGN KEY(sample_id,contig_id) REFERENCES contigs); +CREATE TABLE annotations(annotation_id INTEGER PRIMARY KEY, sample_id TEXT, orf_id TEXT, + reference_db TEXT, target TEXT, product TEXT, score REAL, ec TEXT, reaction TEXT, source_id INTEGER, + FOREIGN KEY(sample_id,orf_id) REFERENCES orfs, FOREIGN KEY(source_id) REFERENCES sources); +CREATE TABLE annotation_taxonomy(sample_id TEXT, orf_id TEXT, reference_db TEXT, target TEXT, + taxid TEXT, taxonomy TEXT, lca_taxonomy TEXT, source_id INTEGER, + PRIMARY KEY(sample_id,orf_id,reference_db,target), FOREIGN KEY(source_id) REFERENCES sources); +CREATE TABLE annotation_terms(annotation_id INTEGER, term_type TEXT, term TEXT, + PRIMARY KEY(annotation_id,term_type,term), FOREIGN KEY(annotation_id) REFERENCES annotations); +CREATE TABLE entities(sample_id TEXT, entity_id TEXT, entity_type TEXT, pathway_status TEXT, last_task_status TEXT, + PRIMARY KEY(sample_id,entity_id), FOREIGN KEY(sample_id) REFERENCES samples); +CREATE TABLE contig_mags(sample_id TEXT, contig_id TEXT, entity_id TEXT, original_mag_id TEXT, source_id INTEGER, + PRIMARY KEY(sample_id,contig_id,entity_id), FOREIGN KEY(sample_id,contig_id) REFERENCES contigs, + FOREIGN KEY(sample_id,entity_id) REFERENCES entities, FOREIGN KEY(source_id) REFERENCES sources); +CREATE INDEX original_contigs ON contigs(sample_id,original_id); +CREATE TABLE entity_orfs(sample_id TEXT, entity_id TEXT, orf_id TEXT, + PRIMARY KEY(sample_id,entity_id,orf_id), FOREIGN KEY(sample_id,entity_id) REFERENCES entities, + FOREIGN KEY(sample_id,orf_id) REFERENCES orfs); +CREATE TABLE orf_groups(sample_id TEXT, representative_orf_id TEXT, member_orf_id TEXT, + PRIMARY KEY(sample_id,representative_orf_id,member_orf_id), + FOREIGN KEY(sample_id,representative_orf_id) REFERENCES orfs, + FOREIGN KEY(sample_id,member_orf_id) REFERENCES orfs); +CREATE TABLE pathways(sample_id TEXT, entity_id TEXT, pathway_id TEXT, name TEXT, score REAL, + reactions INTEGER, covered_reactions INTEGER, reported_orf_count INTEGER, source_id INTEGER, + PRIMARY KEY(sample_id,entity_id,pathway_id), FOREIGN KEY(sample_id,entity_id) REFERENCES entities, + FOREIGN KEY(source_id) REFERENCES sources); +CREATE TABLE pathway_orfs(sample_id TEXT, entity_id TEXT, pathway_id TEXT, orf_id TEXT, + PRIMARY KEY(sample_id,entity_id,pathway_id,orf_id), + FOREIGN KEY(sample_id,entity_id,pathway_id) REFERENCES pathways, + FOREIGN KEY(sample_id,orf_id) REFERENCES orfs); +CREATE TABLE abundance(sample_id TEXT, feature_type TEXT, feature_id TEXT, measurement TEXT, value REAL, + source_id INTEGER, PRIMARY KEY(sample_id,feature_type,feature_id,measurement), + FOREIGN KEY(sample_id) REFERENCES samples, FOREIGN KEY(source_id) REFERENCES sources); +CREATE TABLE abundance_explorer(sample_id TEXT, feature_type TEXT, feature_id TEXT, + orf_id TEXT, contig_id TEXT, length_bp REAL, count REAL, mean_coverage REAL, + coverage_variance REAL, trimmed_mean_coverage REAL, rpkm REAL, tpm REAL, source_id INTEGER, + PRIMARY KEY(sample_id,feature_type,feature_id), + FOREIGN KEY(sample_id) REFERENCES samples, FOREIGN KEY(source_id) REFERENCES sources); +CREATE TABLE issues(issue_id INTEGER PRIMARY KEY, sample_id TEXT, source TEXT, message TEXT); +CREATE TABLE files(path TEXT PRIMARY KEY, bytes INTEGER, modified_utc TEXT); +CREATE TABLE execution(run_id TEXT, command TEXT, task TEXT, label TEXT, status TEXT, + elapsed_seconds REAL, error TEXT, source TEXT); +CREATE INDEX annotations_orf ON annotations(sample_id,orf_id); +CREATE INDEX annotations_reference ON annotations(reference_db); +CREATE INDEX terms_value ON annotation_terms(term_type,term); +CREATE INDEX entity_orfs_orf ON entity_orfs(sample_id,orf_id); +CREATE INDEX pathway_orfs_orf ON pathway_orfs(sample_id,orf_id); +CREATE INDEX orfs_contig ON orfs(sample_id,contig_id); +CREATE VIEW orf_explorer AS + SELECT o.*, c.original_id AS original_contig_id, c.length AS contig_length + FROM orfs o LEFT JOIN contigs c USING(sample_id,contig_id); +CREATE VIEW annotation_explorer AS + SELECT a.*, o.contig_id, t.taxid, + COALESCE(t.taxonomy, 'Not computed') AS taxonomy, + COALESCE(t.lca_taxonomy, 'Not computed') AS lca_taxonomy, + t.source_id AS taxonomy_source_id, c.original_id AS original_contig_id + FROM annotations a JOIN orfs o USING(sample_id,orf_id) + LEFT JOIN annotation_taxonomy t ON t.sample_id=a.sample_id AND t.orf_id=a.orf_id + AND t.reference_db=a.reference_db AND t.target=a.target + LEFT JOIN contigs c ON c.sample_id=o.sample_id AND c.contig_id=o.contig_id; +CREATE VIEW pathway_explorer AS + SELECT p.*, e.entity_type, + (SELECT COUNT(*) FROM pathway_orfs g WHERE g.sample_id=p.sample_id AND g.entity_id=p.entity_id + AND g.pathway_id=p.pathway_id) AS linked_orf_count + FROM pathways p JOIN entities e USING(sample_id,entity_id); +CREATE VIEW pathway_gene_explorer AS + SELECT g.*, p.name AS pathway_name, p.score AS pathway_score, e.entity_type, + o.contig_id, o.product, o.taxonomy, o.annotation_present + FROM pathway_orfs g JOIN pathways p USING(sample_id,entity_id,pathway_id) + JOIN orfs o USING(sample_id,orf_id) JOIN entities e USING(sample_id,entity_id); +CREATE VIEW mag_orf_explorer AS + SELECT m.sample_id,m.entity_id,m.contig_id,o.orf_id,o.product,o.taxonomy,o.annotation_present,m.source_id + FROM contig_mags m JOIN orfs o USING(sample_id,contig_id); +CREATE VIEW mag_gene_explorer AS + SELECT m.*, e.entity_type, o.contig_id, o.product, o.taxonomy, o.annotation_present + FROM entity_orfs m JOIN entities e USING(sample_id,entity_id) JOIN orfs o USING(sample_id,orf_id); +''' +VIEWS = { + 'samples': ('Samples', 'One row per sample output directory.'), + 'contigs': ('Contigs', 'One row per sample and contig; original identifiers come from the mapping file.'), + 'orf_explorer': ('ORFs and taxonomy', 'One row per sample and ORF; the primary annotation and reported taxonomy are preserved.'), + 'annotation_explorer': ('Functional annotations', 'One row per reference annotation; taxonomy belongs to this database and target. LCA uses only hits from this database.'), + 'annotation_terms': ('EC and reaction terms', 'One row per annotation and EC/reaction term; join by annotation_id.'), + 'entities': ('Communities and MAGs', 'Pathway output availability and recorded task status; unavailable is not biological absence.'), + 'contig_mags': ('Contig-to-MAG membership', 'Explicit original contig-to-MAG assignments, joined through the contig identifier map.'), + 'mag_orf_explorer': ('MAG ORFs', 'All reported ORFs on explicitly mapped MAG contigs; requires the preserved contig-to-MAG map.'), + 'mag_gene_explorer': ('MAG input genes', 'Only genes explicitly listed in each MAG Pathway Tools input; not complete MAG membership.'), + 'orf_groups': ('Collapsed ORF groups', 'Explicit representative/member mappings from ptools/orf_map.txt; no membership is inferred.'), + 'pathway_explorer': ('Pathways', 'One row per sample, community/MAG and pathway. Reported scores are not recalculated.'), + 'pathway_gene_explorer': ('Pathway genes', 'One row per explicit pathway/ORF association, without multiplying by reference hits.'), + 'abundance_explorer': ('Read abundance', 'One row per sample and feature. Count and normalization follow the source tool; blank measurements are unavailable, not zero.'), + 'abundance': ('Read abundance: raw measurements', 'Original measurement names and values, one row per measurement; use Read abundance for the wide table.'), + 'execution': ('Execution history', 'Every retained invocation; repeated or reused tasks are not independent biological results.'), + 'issues': ('Import notes', 'Missing files, unmatched identifiers and other limitations detected while reading outputs.'), + 'sources': ('Indexed sources', 'Relative paths and SHA-256 checksums of the files used to build the database.'), + 'files': ('All output files', 'Output inventory, excluding report products and temporary workflow state.'), +} + + +def number(value, integer=False): + if value in (None, '', 'nan', 'NA', 'None'): + return None + result = float(value) + if not __import__('math').isfinite(result): + return None + return int(result) if integer else result + + +def rows(path): + """TSV header normalization; preserve blanks and literal identifier strings.""" + with path.open(newline='', encoding='utf-8-sig') as stream: + reader = csv.DictReader(stream, delimiter='\t') + if reader.fieldnames: + reader.fieldnames = [h.lstrip('# ').strip() for h in reader.fieldnames] + for line, row in enumerate(reader, 2): + if None in row: + raise ValueError(f'{path}:{line}: more values than header columns') + yield {k: v if v is not None else '' for k, v in row.items()} + + +def sample_dirs(root): + def is_sample(p): + return any((p / name).is_dir() for name in ('results/annotation_table', 'preprocessed', 'ptools')) + if is_sample(root): + return [root] + return sorted(p for p in root.iterdir() if p.is_dir() and not p.name.startswith('.') and is_sample(p)) + + +class Importer: + def __init__(self, db, root): + self.db, self.root = db, root + self.source_ids = {} + self.source_stats = {} + self.pgdb_status = {} + # Ignore incomplete or stale pathway tables from a failed/skipped retry. + for summary in sorted(root.glob('logs/analysis_wf/*/summary.json'), key=lambda p: p.stat().st_mtime_ns): + for task in json.loads(summary.read_text()).get('tasks', []): + if task.get('sample') and task.get('entity'): + self.pgdb_status[(task['sample'], task['entity'])] = task.get('status') + + def issue(self, sample, source, message): + self.db.execute('INSERT INTO issues(sample_id,source,message) VALUES(?,?,?)', (sample, str(source), message)) + + def source(self, path, role): + # Reporting must never publish arbitrary files outside the chosen output tree. + path.resolve().relative_to(self.root) + relative = path.relative_to(self.root).as_posix() + if relative in self.source_ids: + return self.source_ids[relative] + before = path.stat() + self.source_stats[path] = (before.st_size, before.st_mtime_ns) + digest = hashlib.sha256() + with path.open('rb') as stream: + for block in iter(lambda: stream.read(1024 * 1024), b''): + digest.update(block) + cur = self.db.execute('INSERT OR IGNORE INTO sources(path,bytes,sha256,role) VALUES(?,?,?,?)', + (relative, path.stat().st_size, digest.hexdigest(), role)) + identifier = self.db.execute('SELECT source_id FROM sources WHERE path=?', (relative,)).fetchone()[0] + self.source_ids[relative] = identifier + return identifier + + def one(self, sample, folder, pattern, role, required=False): + matches = sorted(folder.glob(pattern)) + if len(matches) > 1: + raise ValueError(f'Ambiguous {role} in {folder}: {[p.name for p in matches]}') + if not matches: + if required: + self.issue(sample, folder.relative_to(self.root), f'Missing {role}; related fields may be unavailable.') + return None + self.source(matches[0], role) + return matches[0] + + def orf(self, sample, identifier): + if not identifier: + raise ValueError(f'Empty ORF identifier in sample {sample}') + self.db.execute('INSERT OR IGNORE INTO orfs(sample_id,orf_id) VALUES(?,?)', (sample, identifier)) + + def sample(self, directory): + s = directory.name if directory == self.root else directory.relative_to(self.root).as_posix() + self.db.execute('INSERT INTO samples VALUES(?,?)', (s, directory.relative_to(self.root).as_posix())) + self.db.execute('INSERT INTO entities(sample_id,entity_id,entity_type,pathway_status) VALUES(?,?,?,?)', (s, 'community', 'community', 'unavailable')) + mapping = self.one(s, directory/'preprocessed', '*.mapping.txt', 'contig identifier mapping', True) + if mapping: + with mapping.open(newline='') as f: + for r in csv.reader(f, delimiter='\t'): + if r: + if len(r) != 3: + raise ValueError(f'Expected three contig mapping columns: {mapping}') + self.db.execute('INSERT INTO contigs VALUES(?,?,?,?)', (s, r[0], r[1], number(r[2], True))) + annotation_dir = directory/'results/annotation_table' + primary = self.one(s, annotation_dir, '*.functional_and_taxonomic_table.txt', 'primary ORF annotations', True) + if primary: + for r in rows(primary): + contig = r['Contig_Name'] + self.db.execute('INSERT OR IGNORE INTO contigs(sample_id,contig_id,length) VALUES(?,?,?)', + (s, contig, number(r['Contig_length'], True))) + self.db.execute('INSERT INTO orfs VALUES(?,?,?,?,?,?,?,?,?,?,?,?,1)', + (s, r['ORF_ID'], contig, number(r['ORF_length'], True), number(r['start'], True), + number(r['end'], True), r['strand'], r['target'], r['product'], r['taxonomy'], r.get('reference_db'), r.get('lca_taxonomy'))) + taxonomy = self.one(s, annotation_dir, '*.annotation_taxonomy.tsv', 'per-reference protein taxonomy') + if taxonomy: + source_id = self.source(taxonomy, 'per-reference protein taxonomy') + for r in rows(taxonomy): + self.db.execute('INSERT INTO annotation_taxonomy VALUES(?,?,?,?,?,?,?,?)', + (s, r['orf_id'], r['reference_db'], r['target'], r['taxid'], + r['taxonomy'], r['lca_taxonomy'], source_id)) + # EC_RXN_map is the richer form of .1.txt; do not ingest both and duplicate hits. + hits = self.one(s, annotation_dir, '*.EC_RXN_map.tsv', 'reference annotation and EC/reaction mapping') + if hits is None: + hits = self.one(s, annotation_dir, '*.1.txt', 'reference annotations', True) + if hits: + source_id = self.source(hits, 'reference annotations') + last_orf = None + for r in rows(hits): + # MP's compact .1.txt leaves repeated ORF cells blank. + last_orf = r['orf_id'] or last_orf + self.orf(s, last_orf) + cur = self.db.execute('INSERT INTO annotations(sample_id,orf_id,reference_db,target,product,score,ec,reaction,source_id) VALUES(?,?,?,?,?,?,?,?,?)', + (s, last_orf, r['ref dbname'], r['target'], r['product'], number(r['value']), r.get('EC',''), r.get('RXN',''), source_id)) + for kind, field in [('EC','EC'), ('reaction','RXN')]: + for term in r.get(field, '').split('|'): + if term and term != 'NONE': + self.db.execute('INSERT OR IGNORE INTO annotation_terms VALUES(?,?,?)', (cur.lastrowid, kind, term)) + groups = directory/'ptools/orf_map.txt' + if groups.is_file(): + self.source(groups, 'collapsed ORF groups') + with groups.open(newline='') as stream: + for group in csv.reader(stream, delimiter='\t'): + group = [x for x in group if x] + if not group: + continue + for identifier in group: + self.orf(s, identifier) + self.db.execute('INSERT OR IGNORE INTO orf_groups VALUES(?,?,?)', (s, group[0], identifier)) + for mag in sorted((directory/'magsplitter/results').glob('*')): + if not mag.is_dir(): + continue + kind = 'unbinned' if 'non_binned' in mag.name else 'MAG' + self.db.execute('INSERT OR IGNORE INTO entities(sample_id,entity_id,entity_type,pathway_status) VALUES(?,?,?,?)', (s, mag.name, kind, 'unavailable')) + pf = mag/'0.pf' + if pf.is_file(): + self.source(pf, 'MAG Pathway Tools input genes') + with pf.open() as stream: + for line in stream: + if line.startswith('ID\t'): + identifier = line.rstrip('\r\n').split('\t', 1)[1] + self.orf(s, identifier) + self.db.execute('INSERT OR IGNORE INTO entity_orfs VALUES(?,?,?)', (s, mag.name, identifier)) + mag_map = directory/'magsplitter/contig_to_mag.tsv' + if mag_map.is_file(): + source_id = self.source(mag_map, 'original contig-to-MAG membership') + unmatched = 0 + with mag_map.open(newline='') as stream: + for row in csv.reader(stream, delimiter='\t'): + if not row: + continue + if len(row) != 2 or not all(row): + raise ValueError(f'Expected original-contig and MAG columns without a header: {mag_map}') + original, original_mag_id = row + # MAGSplitter names output directories by replacing periods. + mag_id = original_mag_id.replace('.', '_') + self.db.execute('INSERT OR IGNORE INTO entities(sample_id,entity_id,entity_type,pathway_status) VALUES(?,?,?,?)', (s, mag_id, 'MAG', 'unavailable')) + contigs = self.db.execute('SELECT contig_id FROM contigs WHERE sample_id=? AND original_id=?', (s, original)).fetchall() + if not contigs: + unmatched += 1 + for (contig,) in contigs: + self.db.execute('INSERT OR IGNORE INTO contig_mags VALUES(?,?,?,?,?)', (s, contig, mag_id, original_mag_id, source_id)) + if unmatched: + self.issue(s, mag_map.relative_to(self.root), f'{unmatched} input contig assignments do not match retained contigs (for example after QC).') + elif (directory/'magsplitter').is_dir(): + self.issue(s, 'magsplitter/contig_to_mag.tsv', 'Full contig-to-MAG membership is unavailable; MAG input genes alone are incomplete.') + pgdb = directory/'results/pgdb' + for path in sorted(pgdb.glob('community/*_pwy.tsv')) + sorted(pgdb.glob('MAGs/*/*_pwy.tsv')): + entity = 'community' if path.parent.name == 'community' and path.parent.parent == pgdb else path.parent.name + self.db.execute('INSERT OR IGNORE INTO entities(sample_id,entity_id,entity_type,pathway_status) VALUES(?,?,?,?)', (s, entity, 'community' if entity == 'community' else 'MAG', 'unavailable')) + self.db.execute('UPDATE entities SET pathway_status=? WHERE sample_id=? AND entity_id=?', ('available', s, entity)) + status = self.pgdb_status.get((s, entity)) + if status and status not in ('SUCCESS', 'ALREADY_COMPUTED'): + self.db.execute('UPDATE entities SET pathway_status=? WHERE sample_id=? AND entity_id=?', ('unavailable', s, entity)) + self.issue(s, path.relative_to(self.root), f'Pathway table excluded because the latest PGDB task is {status}.') + continue + source_id = self.source(path, 'pathway inference') + for r in rows(path): + self.db.execute('INSERT INTO pathways VALUES(?,?,?,?,?,?,?,?,?)', + (s, entity, r['PWY_NAME'], r['PWY_COMMON_NAME'], number(r['PWY_SCORE']), + number(r['NUM_REACTIONS'], True), number(r['NUM_COVERED_REACTIONS'], True), number(r['ORF_COUNT'], True), source_id)) + identifiers = {x.strip() for x in r['ORFS'].split(',') if x.strip() and x.strip() != 'nan'} + for identifier in sorted(identifiers): + self.orf(s, identifier) + self.db.execute('INSERT INTO pathway_orfs VALUES(?,?,?,?)', (s, entity, r['PWY_NAME'], identifier)) + if number(r['ORF_COUNT'], True) != len(identifiers): + self.issue(s, path.relative_to(self.root), f"{entity}/{r['PWY_NAME']}: reported ORF count differs from unique listed ORFs.") + known_rna_features = set() + for path in sorted((directory/'results/rpkm').glob('*.contig_counts.tsv')) + sorted((directory/'results/rpkm').glob('*.orf_counts.tsv')): + feature = 'contig' if path.name.endswith('.contig_counts.tsv') else 'orf' + source_id = self.source(path, 'read abundance') + for r in rows(path): + identifier = r['Contig'] if feature == 'contig' else r['Gene_ID'] + if feature == 'orf': + if r.get('feature') in ('tRNA', 'rRNA'): + known_rna_features.add(identifier) + self.orf(s, identifier) + if r.get('seqname'): + self.db.execute('INSERT OR IGNORE INTO contigs(sample_id,contig_id) VALUES(?,?)', (s, r['seqname'])) + self.db.execute('UPDATE orfs SET contig_id=COALESCE(contig_id,?) WHERE sample_id=? AND orf_id=?', (r['seqname'],s,identifier)) + else: + self.db.execute('INSERT OR IGNORE INTO contigs(sample_id,contig_id) VALUES(?,?)', (s,identifier)) + fields = {k:v for k,v in r.items() if k != 'Contig'} if feature == 'contig' else {k:r[k] for k in ('Length','Count','RPKM','TPM') if k in r} + for key, value in fields.items(): + self.db.execute('INSERT INTO abundance VALUES(?,?,?,?,?,?)', (s, feature, identifier, key, number(value), source_id)) + cleaned = {} + names = {'Length':'length_bp', 'Read Count':'count', 'Count':'count', + 'Mean':'mean_coverage', 'Variance':'coverage_variance', + 'Trimmed Mean':'trimmed_mean_coverage', 'RPKM':'rpkm', 'TPM':'tpm'} + for key, value in fields.items(): + # Only strip recognized terminal metric names, never guess from a path. + metric = next((name for name in sorted(names, key=len, reverse=True) + if key == name or key.endswith(' ' + name)), None) + if metric is None: + continue # Original fields remain in the raw measurements table. + column = names[metric] + if column in cleaned: + raise ValueError(f'Multiple abundance columns map to {column} in {path}; cannot combine read sets.') + cleaned[column] = number(value) + orf_id = identifier if feature == 'orf' else None + contig_id = identifier if feature == 'contig' else r.get('seqname') + if feature == 'orf' and not contig_id: + row = self.db.execute('SELECT contig_id FROM orfs WHERE sample_id=? AND orf_id=?', (s, identifier)).fetchone() + contig_id = row[0] if row else None + self.db.execute('INSERT INTO abundance_explorer VALUES(?,?,?,?,?,?,?,?,?,?,?,?,?)', + (s, feature, identifier, orf_id, contig_id, + *(cleaned.get(k) for k in ('length_bp','count','mean_coverage','coverage_variance','trimmed_mean_coverage','rpkm','tpm')), + source_id)) + missing = sum(identifier not in known_rna_features for (identifier,) in + self.db.execute('SELECT orf_id FROM orfs WHERE sample_id=? AND annotation_present=0', (s,))) + if missing: + self.issue(s, directory.relative_to(self.root), f'{missing} referenced ORF identifiers lack primary annotations; placeholders preserve these relationships.') + self.db.commit() + + def inventory(self): + for directory, dirs, files in os.walk(self.root, followlinks=False): + dirs[:] = sorted(d for d in dirs if not d.startswith('.') and d not in ('reports','work','conda-cache') and not (Path(directory)/d).is_symlink()) + for name in sorted(files): + path = Path(directory)/name + if not path.is_file() or path.is_symlink() or name.startswith('.'): + continue + stat = path.stat() + self.db.execute('INSERT INTO files VALUES(?,?,?)', + (path.relative_to(self.root).as_posix(), stat.st_size, datetime.fromtimestamp(stat.st_mtime, timezone.utc).isoformat())) + # Summaries record optional failures even when Nextflow completed normally. + summaries = list(self.root.glob('logs/*/*/summary.json')) + list(self.root.glob('*/logs/*/*/summary.json')) + for path in sorted(summaries, key=lambda p: p.stat().st_mtime_ns): + source = path.relative_to(self.root).as_posix() + data = json.loads(path.read_text()) + for task in data.get('tasks', []): + sample_root = path.parent.parent.parent.parent + sample_id = sample_root.name if sample_root == self.root else sample_root.relative_to(self.root).as_posix() + if path.parent.parent.name == 'ptools' or task.get('entity'): + self.db.execute('UPDATE entities SET last_task_status=? WHERE sample_id=? AND entity_id=?', + (task.get('status'), task.get('sample', sample_id), task.get('entity', task.get('label')))) + self.db.execute('INSERT INTO execution VALUES(?,?,?,?,?,?,?,?)', + (path.parent.name, path.parent.parent.name, task.get('task',''), task.get('label',''), + task.get('status',''), task.get('elapsed_seconds'), task.get('error',''), source)) + self.db.commit() + + +def run_details(root): + """Describe the latest workflow summary without mistaking report version for run version.""" + from itertools import islice + import re + candidates = list(root.glob('logs/*/*/summary.json')) + list(root.glob('*/logs/*/*/summary.json')) + if not candidates: + return {} + path = max(candidates, key=lambda p: p.stat().st_mtime_ns) + summary = json.loads(path.read_text()) + version = summary.get('mp_version') + if not version: + # Older summaries did not record a version. Match the specific run ID + # in CLI log preambles, rather than borrowing a newer installation version. + log_root = path.parent.parent.parent + for log in sorted((log_root/'cli').glob('*.log'), reverse=True): + with log.open(errors='replace') as stream: + preamble = ''.join(islice(stream, 80)) + if path.parent.name in preamble: + match = re.search(r'RUNNING MetaPathways: v([^\s]+)', preamble) + if match: + version = match.group(1) + break + return dict(mp_version=version, command=path.parent.parent.name, + executor=summary.get('resources', {}).get('executor'), status=summary.get('status')) + + +def metadata(db): + result = {} + for name, (label, description) in VIEWS.items(): + result[name] = dict(label=label, description=description, + columns=[dict(name=r[1], type=r[2]) for r in db.execute(f'PRAGMA table_info("{name}")')], + rows=db.execute(f'SELECT COUNT(*) FROM "{name}"').fetchone()[0]) + tables = {} + for (name,) in db.execute("SELECT name FROM sqlite_master WHERE type='table' ORDER BY name"): + tables[name] = dict(columns=[dict(name=r[1], type=r[2], primary_key_order=r[5]) for r in db.execute(f'PRAGMA table_info("{name}")')], + foreign_keys=[dict(id=r[0], sequence=r[1], table=r[2], column=r[3], references=r[4]) for r in db.execute(f'PRAGMA foreign_key_list("{name}")')]) + return dict(schema_version=SCHEMA_VERSION, generated_utc=datetime.now(timezone.utc).isoformat(), views=result, tables=tables) + + +def atomic_text(path, value): + temporary = path.with_name('.'+path.name+'.tmp') + temporary.write_text(value, encoding='utf-8') + temporary.replace(path) + + +def write_report(db, root, reports, info): + esc = html.escape + cards = ''.join(f'
  • {esc(view["label"])}: {view["rows"]:,} rows
  • ' for name, view in info['views'].items()) + samples = ''.join(f'{esc(s)}{esc(p)}' for s,p in db.execute('SELECT * FROM samples ORDER BY sample_id')) + notes = ''.join(f'
  • {esc(str(s))}: {esc(m)}
  • ' for s,m in db.execute('SELECT sample_id,message FROM issues LIMIT 100')) + # Link files rather than relying on web-server directory listings. + links = ''.join(f'
  • {esc(p)}
  • ' for (p,) in db.execute("SELECT path FROM files WHERE path LIKE '%/trace.tsv' OR path LIKE '%/report.html' OR path LIKE '%/timeline.html' OR path LIKE '%/summary.json' ORDER BY path")) + document = f'''MetaPathways run report + +

    MetaPathways run report

    Generated {esc(info['generated_utc'])}. An inventory of existing outputs; it does not rerun or interpret analyses.

    +

    Open EDA portal · Relational results database · Schema and table definitions · All output files

    +

    For searching and CSV export: metapathways report -o OUTPUT --serve --no-rebuild. The HTML report and file links also work without the server.

    +

    Samples

    {samples}
    SampleOutput directory
    +

    Available results

      {cards}

    Counts describe table rows, not necessarily distinct genes or pathways. Sample and entity identifiers scope every biological relationship.

    +

    Availability and import notes

      {notes or '
    • No import notes.
    • '}

    Up to 100 notes shown; the portal includes all notes. Missing pathways are not evidence of biological absence. Expected MAG failures remain in execution history.

    +

    Nextflow run details

      {links or '
    • No Nextflow records found in these outputs.
    • '}
    ''' + atomic_text(reports/'MP_run_report.html', document) + with (reports/'output_inventory.tsv').open('w', newline='') as f: + writer=csv.writer(f, delimiter='\t'); writer.writerow(['path','bytes','modified_utc']) + writer.writerows(db.execute('SELECT * FROM files ORDER BY path')) + + +def build_report(output): + root = Path(output).expanduser().resolve() + if not root.is_dir(): + raise ValueError(f'Output directory does not exist: {root}') + samples = sample_dirs(root) + if not samples: + raise ValueError(f'No sample outputs found directly in {root} or its immediate child directories') + reports = root/'reports' + reports.mkdir(exist_ok=True) + with (reports/'.build.lock').open('w') as lock: + try: + fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError: + raise RuntimeError(f'Another report is being built in {reports}') + with tempfile.TemporaryDirectory(prefix='.build-', dir=reports) as temporary: + database = Path(temporary)/'results.sqlite' + with closing(sqlite3.connect(database)) as db: + # Build only in a disposable staging file; publish after validation. + db.execute('PRAGMA journal_mode=MEMORY') + db.execute('PRAGMA synchronous=OFF') + db.execute('PRAGMA foreign_keys=ON') + db.executescript('BEGIN;\n' + SCHEMA) + db.execute('PRAGMA cache_size=-32768') + importer = Importer(db, root) + for directory in samples: + print(f'Indexing report tables: {directory.name}', flush=True) + importer.sample(directory) + importer.inventory() + for source, expected in importer.source_stats.items(): + current = source.stat() + if (current.st_size, current.st_mtime_ns) != expected: + raise ValueError(f"Source changed during report generation: {source}; rebuild when outputs are stable") + if db.execute('PRAGMA foreign_key_check').fetchone(): + raise ValueError('Report database contains invalid relationships') + info = metadata(db) + info['output_root'] = str(root) + info['sample_paths'] = [str(p) for p in samples] + from metapathways._version import __version__ + info['report_mp_version'] = __version__ + info['run_details'] = run_details(root) + db.execute('ANALYZE'); db.commit() + write_report(db, root, reports, info) + with database.open('rb') as stream: + os.fsync(stream.fileno()) + database.replace(reports/'results.sqlite') + atomic_text(reports/'schema.json', json.dumps(info, indent=2)+'\n') + assets = Path(__file__).parent/'report_assets' + for name in ('EDA_portal.html','portal.js','portal.css'): + shutil.copyfile(assets/name, reports/name) + print(f'Report: {reports / "MP_run_report.html"}', flush=True) + return reports diff --git a/metapathways/resources/ptools_reaction_blacklist.json b/metapathways/resources/ptools_reaction_blacklist.json new file mode 100644 index 0000000..c844c02 --- /dev/null +++ b/metapathways/resources/ptools_reaction_blacklist.json @@ -0,0 +1,12 @@ +{ + "TRANS-RXN8J2-121": { + "reason": "Pathway Tools 29.5 imports protein substrates before name matching, then dereferences a missing raw-gene record for BSU21480-MONOMER.", + "evidence": "Reproduced with Skin_13 C137456-G1 alone; removing the explicit METACYC line permits a complete build, with the same reaction recovered by name matching.", + "image_sha256": "be2ab905fa6b6d0eeba6b056d9682565001ac9a416c462eb1a8da900e46cdb8b" + }, + "RXN8J2-204": { + "reason": "Pathway Tools 29.5 fails during enzyme-name matching after explicit reaction import.", + "evidence": "Two isolated MPDB reaction screens failed with a NIL-reference error; no-reaction control succeeded.", + "image_sha256": "be2ab905fa6b6d0eeba6b056d9682565001ac9a416c462eb1a8da900e46cdb8b" + } +} diff --git a/metapathways/sysutil.py b/metapathways/sysutil.py index 30ea244..66ddc71 100644 --- a/metapathways/sysutil.py +++ b/metapathways/sysutil.py @@ -70,8 +70,15 @@ def pathDelim(): def getstatusoutput(cmd): """Return (status, output) of executing cmd in a shell.""" - pipe = os.popen(cmd + " 2>&1", "r") - text = pipe.read() + pipe = os.popen("{ " + cmd + "\n} 2>&1", "r") + if os.environ.get('METAPATHWAYS_STREAM_TOOLS') == '1': + chunks = [] + for line in pipe: + print(line, end='', flush=True) + chunks.append(line) + text = ''.join(chunks) + else: + text = pipe.read() sts = pipe.close() if sts is None: sts = 0 diff --git a/metapathways/test_data.py b/metapathways/test_data.py new file mode 100644 index 0000000..c3b1589 --- /dev/null +++ b/metapathways/test_data.py @@ -0,0 +1,54 @@ +"""Copy the bundled test inputs and reference seeds into a writable workspace.""" +import argparse +from pathlib import Path +import shutil + +SEEDS = ('functional/swissprot_test', 'taxonomic/SILVA_SSU_test', + 'taxonomic/SILVA_LSU_test', 'test_reference.json') + + +def prepare(output, fixtures=None): + fixtures = Path(fixtures) if fixtures else Path(__file__).parent / 'regtests' + root = Path(output).expanduser().resolve() + inputs = fixtures / 'cami_test' + files = [(p, root / 'cami-test' / p.relative_to(inputs)) + for p in sorted(inputs.rglob('*')) if p.is_file()] + if not files or not (inputs / 'all.tsv').is_file(): + raise ValueError('The installed test bundle is incomplete; reinstall MetaPathways.') + files.extend((fixtures / 'test_db' / name, root / 'MPDB' / name) for name in SEEDS) + # Preflight the complete copy before writing anything. Never follow a link into + # an installation or replace a user's modified inputs/reference files. + for source, target in files: + if not source.is_file(): + raise ValueError(f'Missing bundled test file: {source}') + for p in (target, *target.parents): + if p == root: + break + if p.is_symlink(): + raise ValueError(f'Test destination contains a symbolic link: {p}; choose a new directory.') + if any(p.exists() and not p.is_dir() for p in target.parents): + raise ValueError(f'Test destination parent is not a directory: {target.parent}') + if target.exists() and (not target.is_file() or target.read_bytes() != source.read_bytes()): + raise ValueError(f'Test file differs: {target}; choose a new directory.') + for source, target in files: + target.parent.mkdir(parents=True, exist_ok=True) + if not target.exists(): + # Exclusive creation also protects against a concurrent preparation. + with source.open('rb') as src, target.open('xb') as dst: + shutil.copyfileobj(src, dst) + return root + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('-o', '--output_dir', required=True, + help='workspace for cami-test/ inputs and MPDB/ reference seeds') + args = parser.parse_args(argv) + root = prepare(args.output_dir) + print(f'Test workspace: {root}') + print('Inputs: cami-test/all.tsv (three samples), single.tsv, pair.tsv') + print('Next: change to this workspace and run metapathways build_db --test -d MPDB') + + +if __name__ == '__main__': + main() diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..cb60049 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,3 @@ +[build-system] +requires = ["setuptools>=77", "wheel"] +build-backend = "setuptools.build_meta" diff --git a/requirements-workflow.txt b/requirements-workflow.txt new file mode 100644 index 0000000..2ea876c --- /dev/null +++ b/requirements-workflow.txt @@ -0,0 +1,4 @@ +# Workflow dependencies installed automatically by the source package. +# Conda builds bundle these same immutable sources into the release artifact. +magsplitter @ https://github.com/hallamlab/MAGSplitter/archive/a83bcd0c6479d3fda15025262b981e348c082bc9.tar.gz#sha256=4a7a6a6936e4edc6d4a6b27de1acff29f32156ac181901c83babc1bbd9da8ee6 +camelot-frs @ https://bitbucket.org/tomeraltman/camelot-frs/get/30a774c7fe7bb8a88b5cd6cadfac1c6d258efc96.tar.gz#sha256=eba9efcc91b897dd3bc70ccc7bcb7d26494e8d1b023743498dff0dc00133600d diff --git a/scripts/check_docs.py b/scripts/check_docs.py new file mode 100644 index 0000000..157a1b7 --- /dev/null +++ b/scripts/check_docs.py @@ -0,0 +1,26 @@ +#!/usr/bin/env python3 +"""Check repository documentation links and the generated CLI reference.""" +from pathlib import Path +import re +import subprocess +import sys +from urllib.parse import unquote, urlsplit + +ROOT=Path(__file__).resolve().parents[1] +files=[ROOT/'README.md', *sorted((ROOT/'docs').glob('*.md')), ROOT/'docker/README.quay.md'] +errors=[] +for source in files: + content=re.sub(r'```.*?```','',source.read_text(),flags=re.S) + for link in re.findall(r'\]\(([^\s)]+)\)',content): + url=urlsplit(link) + if 'github.com/hallamlab/MetaPathways/blob/HEAD/' in link: + errors.append(f'{source.relative_to(ROOT)}: default-branch link can mismatch this docs version: {link}') + if url.scheme or url.netloc or not url.path: + continue + target=(source.parent/unquote(url.path)).resolve() + if not target.exists(): + errors.append(f'{source.relative_to(ROOT)}: missing link {link}') +if errors: + raise SystemExit('\n'.join(errors)) +subprocess.run([sys.executable,str(ROOT/'scripts/generate_cli_docs.py'),'--check'],check=True) +print(f'Local links checked in {len(files)} Markdown documents.') diff --git a/scripts/check_docs_html.py b/scripts/check_docs_html.py new file mode 100644 index 0000000..57296c3 --- /dev/null +++ b/scripts/check_docs_html.py @@ -0,0 +1,62 @@ +#!/usr/bin/env python3 +"""Validate local links and fragments in the built documentation.""" +import argparse +import re +from html.parser import HTMLParser +from pathlib import Path +from urllib.parse import unquote, urlsplit + + +class Page(HTMLParser): + def __init__(self, text): + super().__init__() + self.ids = set() + self.links = [] + self.feed(text) + + def handle_starttag(self, tag, attrs): + attrs = dict(attrs) + if 'id' in attrs: + self.ids.add(attrs['id']) + if tag == 'a' and 'href' in attrs: + self.links.append(attrs['href']) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('directory', type=Path) + parser.add_argument('--readme', type=Path, help='also check hosted documentation links in the README') + args = parser.parse_args() + root = args.directory.resolve() + pages = {p.resolve(): Page(p.read_text()) for p in root.rglob('*.html')} + if not pages: + raise SystemExit(f'No HTML pages found in {root}') + errors = [] + for source, page in pages.items(): + for link in page.links: + url = urlsplit(link) + if url.scheme or url.netloc: + continue + target = (source.parent / unquote(url.path)).resolve() if url.path else source + if target.is_dir(): + target /= 'index.html' + if not target.exists(): + errors.append(f'{source.name}: missing {link}') + elif url.fragment and target in pages and unquote(url.fragment) not in pages[target].ids: + errors.append(f'{source.name}: missing anchor {link}') + if args.readme: + for link in re.findall(r'\]\((https://metapathways\.readthedocs\.io/[^\s)]*)\)', args.readme.read_text()): + url = urlsplit(link) + relative = url.path.removeprefix('/en/latest/').lstrip('/') or 'index.html' + target = root / relative + if target not in pages: + errors.append(f'README: missing documentation page {link}') + elif url.fragment and unquote(url.fragment) not in pages[target].ids: + errors.append(f'README: missing documentation anchor {link}') + if errors: + raise SystemExit('\n'.join(sorted(set(errors)))) + print(f'Local links and anchors checked in {len(pages)} HTML pages.') + + +if __name__ == '__main__': + main() diff --git a/scripts/check_installed_assets.py b/scripts/check_installed_assets.py new file mode 100644 index 0000000..ab14d3d --- /dev/null +++ b/scripts/check_installed_assets.py @@ -0,0 +1,22 @@ +#!/usr/bin/env python3 +"""Check the installed MP payload outside the source checkout (no biology run).""" +import hashlib +import json +from pathlib import Path +import metapathways +from metapathways.analysis_workflow import read_manifest, validate + +root = Path(metapathways.__file__).resolve().parent +for name in ('test_data.py', 'nextflow.py', 'nf_worker.py', 'nf_databases.py', 'analysis_workflow.py', + 'reporting.py', 'report_server.py', 'protein_taxonomy.py', 'pt_container.py', 'pt_sequences.py', + 'bin/fastal', 'bin/fastdb', 'bin/metacount'): + path = root / name + assert path.is_file() and path.stat().st_size, f'Missing package asset: {path}' +assets = root / 'report_assets' +assert list(assets.glob('*.html')) and list(assets.glob('*.js')), assets +fixture = root / 'regtests/cami_test' +for path, digest in json.loads((fixture / 'provenance.json').read_text())['sha256'].items(): + assert hashlib.sha256((fixture / path).read_bytes()).hexdigest() == digest, path +for label, count in [('single', 1), ('pair', 2), ('all', 3)]: + assert len(validate(read_manifest(fixture / f'{label}.tsv'))) == count, label +print(f'Installed workflow modules, binaries, report assets and CAMI fixtures verified: {root}') diff --git a/scripts/check_release_controls.py b/scripts/check_release_controls.py new file mode 100644 index 0000000..d05f4da --- /dev/null +++ b/scripts/check_release_controls.py @@ -0,0 +1,30 @@ +#!/usr/bin/env python3 +"""Reject accidental publication before allocating build runners.""" +import os +from release import tag_version + + +def validate(event, repository, ref_type, ref_name, release_tag='', **selections): + if not all(isinstance(value, bool) for value in selections.values()): + raise ValueError('Publication selections must be booleans') + if not any(selections.values()): + return + if event != 'workflow_dispatch': + raise ValueError('Publication requires a manual workflow dispatch') + if repository.lower() != 'hallamlab/metapathways': + raise ValueError('Publication is restricted to hallamlab/MetaPathways') + tag = release_tag or (ref_name if ref_type == 'tag' else '') + tag_version(tag) # Build/artifact verification also checks its version and commit. + + +if __name__ == '__main__': + selections = {} + for name in ('PUBLISH_ANACONDA', 'PUBLISH_QUAY', 'PUBLISH_GITHUB'): + value = os.environ.get(name, 'false').lower() + if value not in ('true', 'false'): + raise SystemExit(f'{name} must be true or false') + selections[name] = value == 'true' + validate(os.environ.get('GITHUB_EVENT_NAME', ''), os.environ.get('GITHUB_REPOSITORY', ''), + os.environ.get('GITHUB_REF_TYPE', ''), os.environ.get('GITHUB_REF_NAME', ''), + os.environ.get('RELEASE_TAG_INPUT', ''), **selections) + print('Publication selections validated; unselected destinations will not publish.') diff --git a/scripts/check_workflow_style.py b/scripts/check_workflow_style.py new file mode 100644 index 0000000..0673f1b --- /dev/null +++ b/scripts/check_workflow_style.py @@ -0,0 +1,66 @@ +#!/usr/bin/env python3 +"""Check the shared MP nodal workflow contract without third-party dependencies.""" +from pathlib import Path +import re +import sys +import xml.etree.ElementTree as ET + +ROOT = Path(__file__).resolve().parents[1] +NS = '{http://www.w3.org/2000/svg}' + +def check(path): + root = ET.parse(path).getroot() + source = path.read_text() + errors = [] + if 'Times New Roman' not in source or 'serif' not in source: + errors.append('use the shared Times serif typography') + if any(font in source for font in ('Arial', 'Helvetica', 'sans-serif')): + errors.append('workflow labels must retain serif typography') + if not any(n.get('fill', '').upper() == '#CCCCCC' for n in root.iter(NS+'rect')): + errors.append('keep the restrained gray header / legend') + circles = list(root.iter(NS + 'circle')) + for color, role in [('#DAE8FC', 'input'), ('#D5E8D4', 'output')]: + count = sum(n.get('fill', '').upper() == color for n in circles) + if count < 2: + errors.append(f'keep circular {role} nodes beyond the legend') + squares = [] + for n in root.iter(NS + 'rect'): + try: + w, h = float(n.get('width', 0)), float(n.get('height', 0)) + if 18 <= w <= 40 and abs(w-h) < 0.1 and float(n.get('rx', 0)) == 0: + squares.append(n) + except ValueError: + pass + if len(squares) < 3: + errors.append('keep the numbered square module spine') + numbered = [n for n in root.iter() if n.tag in (NS+'text', NS+'tspan') + and re.fullmatch(r'\d+', (n.text or '').strip())] + if len(numbered) < 2: + errors.append('keep module numbers') + diamonds = [n for n in root.iter(NS+'path') + if len(re.findall('[Ll]', n.get('d', ''))) >= 3 + and n.get('fill', '').upper() == '#F5F5F5'] + if len(diamonds) < 3: + errors.append('keep diamond compute nodes beyond the legend') + if not list(root.iter(NS+'marker')): + errors.append('keep directional connectors') + return errors + +def main(): + paths = [ROOT / 'docs/assets' / name for name in + ('workflow.svg', 'workflow-main.svg', 'workflow-brief.svg', 'workflow-detailed.svg')] + paths = [p for p in paths if p.exists()] + if not paths: + print('No primary workflow SVGs in this checkout; apply docs/WORKFLOW_STYLE.md to new figures.') + return 0 + failed = False + for path in paths: + errors = check(path) + failed |= bool(errors) + print(('FAIL' if errors else 'PASS') + ' ' + str(path.relative_to(ROOT))) + for error in errors: + print(' ' + error) + return int(failed) + +if __name__ == '__main__': + sys.exit(main()) diff --git a/scripts/containers.py b/scripts/containers.py index 9b124df..83da7c0 100644 --- a/scripts/containers.py +++ b/scripts/containers.py @@ -64,7 +64,7 @@ def build(args): integration.chmod(0o777) release.run("docker", "run", "--rm", "--volume", f"{integration}:/work", image, "bash", "-c", "umask 0000; trap 'find /work -mindepth 1 -exec chmod a+rwX {} +' EXIT; " - "metapathways build_db --test && metapathways run --test", + "metapathways build_db --test --memory '2 GB' --max_memory '4 GB' --max_cpus 2 && metapathways run --test --memory '2 GB' --max_memory '4 GB' --max_cpus 2", log=output / "docker-integration.log") receipt = release.validate_run(integration) for name in ["metapathways_steps_log.txt", "errors_warnings_log.txt"]: @@ -132,7 +132,7 @@ def push(args): tags = list(dict.fromkeys([tag.removeprefix("v"), tag, value, f"v{value}"])) tags += [] if "rc" in value else ["latest"] for tag in tags: - target = f"{IMAGE}:{tag}" + target = f"{os.environ.get('QUAY_REPOSITORY', IMAGE)}:{tag}" release.run("docker", "tag", image, target) release.run("docker", "push", target) @@ -150,7 +150,7 @@ def attach(args): bundle = Path(temp) / f"metapathways-{receipt['version']}-container-validation.zip" with release.zipfile.ZipFile(bundle, "w", release.zipfile.ZIP_DEFLATED) as archive: for path in sorted(output.iterdir()): - if path.suffix != ".sif": + if path.suffix != ".sif" and path.name != "docker-image.tar.gz": info = release.zipfile.ZipInfo(path.name, (1980, 1, 1, 0, 0, 0)) archive.writestr(info, path.read_bytes()) for path in [output / receipt["sif"], output / "container-SHA256SUMS", bundle]: diff --git a/scripts/generate_cli_docs.py b/scripts/generate_cli_docs.py new file mode 100644 index 0000000..1969c38 --- /dev/null +++ b/scripts/generate_cli_docs.py @@ -0,0 +1,34 @@ +#!/usr/bin/env python3 +"""Generate the GitHub CLI reference from the installed source parser help.""" +import argparse +import os +from pathlib import Path +import subprocess +import sys + +ROOT=Path(__file__).resolve().parents[1] +COMMANDS=('prepare_test','run','analysis_wf','build_db','mag_split','build_pt','screen_pt','ptools','report') + + +def render(): + env=dict(os.environ, PYTHONPATH=str(ROOT), COLUMNS='100') + text='# Complete CLI reference\n\n[User guide](index.md) · [Workflow guide](workflow.md)\n\n' + text+='Generated by `python scripts/generate_cli_docs.py`. Run `metapathways COMMAND --help` for the help of your installed revision.\n\n' + for command in COMMANDS: + code='import sys; from metapathways.pipeline import main; sys.argv=["metapathways"]+sys.argv[1:]; main()' + result=subprocess.run([sys.executable,'-c',code,command,'--help'],cwd=ROOT,env=env,check=True,text=True,capture_output=True) + text+=f'## {command}\n\n```text\n{result.stdout.rstrip()}\n```\n\n' + return '\n'.join(line.rstrip() for line in text.splitlines()).rstrip() + '\n' + + +def main(): + p=argparse.ArgumentParser();p.add_argument('--check',action='store_true');args=p.parse_args() + target=ROOT/'docs/cli-reference.md';expected=render() + if args.check: + if not target.is_file() or target.read_text()!=expected: + raise SystemExit('CLI reference is stale; run python scripts/generate_cli_docs.py') + print('CLI reference matches current parser help.') + else: + target.write_text(expected) + +if __name__=='__main__':main() diff --git a/scripts/prepare_cami_test.py b/scripts/prepare_cami_test.py new file mode 100644 index 0000000..68717c9 --- /dev/null +++ b/scripts/prepare_cami_test.py @@ -0,0 +1,159 @@ +#!/usr/bin/env python3 +"""Prepare tiny real CAMI inputs; requires minimap2, not MetaPathways or Pathway Tools. + +Select three abundant genomes per sample, retain up to 50 kb of one contig each, +then retain intact original read pairs with a primary MAPQ >=20 alignment to +those regions from the first 250,000 pairs. This is deliberately coverage-biased +interface test data, never an abundance or accuracy benchmark. +""" +import argparse +import csv +import gzip +import hashlib +import json +import re +import subprocess +import tempfile +from pathlib import Path + +SAMPLES = ('Urogenital_22', 'Gastrointestinal_5', 'Skin_28') +FIELDS = ('sample_id', 'assembly', 'read_layout', 'reads_1', 'reads_2', 'mag_map') + + +def fasta(path): + with open(path) as handle: + name, seq = None, [] + for line in handle: + if line.startswith('>'): + if name is not None: + yield name, ''.join(seq) + name, seq = line[1:].split()[0], [] + else: + seq.append(line.strip()) + if name is not None: + yield name, ''.join(seq) + + +def pair_records(handle): + while True: + a = [handle.readline() for _ in range(4)] + if not a[0]: + return + b = [handle.readline() for _ in range(4)] + for record in (a, b): + if (not record[0].startswith('@') or not record[2].startswith('+') + or len(record[1].strip()) != len(record[3].strip()) or not record[3]): + raise ValueError('Invalid/truncated FASTQ') + if not a[0].split()[0].endswith('/1') or a[0].split()[0][:-2] != b[0].split()[0][:-2] or not b[0].split()[0].endswith('/2'): + raise ValueError('Source FASTQ is not adjacent /1, /2 pairs') + yield a, b + + +def write_gzip(path, text): + # No timestamp or source filename: reproducible compressed payload. + with open(path, 'wb') as raw: + with gzip.GzipFile(fileobj=raw, mode='wb', filename='', mtime=0) as handle: + handle.write(text.encode()) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--source-manifest', required=True) + parser.add_argument('--output', required=True) + parser.add_argument('--minimap2', default='minimap2') + args = parser.parse_args() + out = Path(args.output) + if out.exists(): + parser.error('Output already exists; choose a new directory') + out.mkdir(parents=True) + for part in ('assemblies', 'reads', 'mag_maps'): + (out / 'inputs' / part).mkdir(parents=True) + with open(args.source_manifest) as handle: + sources = {r['sample_id']: r for r in csv.DictReader(handle, delimiter='\t')} + rows, stats = [], [] + with tempfile.TemporaryDirectory(prefix='mp-cami-test-') as tmp: + tmp = Path(tmp) + for sample in SAMPLES: + source = sources[sample] + if source['read_layout'] != 'interleaved' or source.get('reads_2'): + raise ValueError(f'{sample}: expected original interleaved CAMI reads') + assembly = Path(source['assembly']) + mapping = assembly.parent / 'gsa_mapping.tsv' + with mapping.open() as handle: + candidates = [r for r in csv.DictReader(handle, delimiter='\t') + if int(r['end_position']) - int(r['start_position']) + 1 >= 20000] + candidates.sort(key=lambda r: (-int(r['number_reads']) / (int(r['end_position']) - int(r['start_position']) + 1), r['#anonymous_contig_id'])) + chosen, genomes = {}, set() + for r in candidates: + if r['genome_id'] not in genomes: + chosen[r['#anonymous_contig_id']] = r + genomes.add(r['genome_id']) + if len(chosen) == 3: + break + if len(chosen) != 3: + raise ValueError(f'{sample}: fewer than three suitable genomes') + sequences = {} + for name, seq in fasta(assembly): + if name in chosen: + sequences[name] = seq[:50000] + if len(sequences) == len(chosen): + break + if sequences.keys() != chosen.keys(): + raise ValueError(f'{sample}: missing selected contigs') + fa = ''.join(f'>{name}\n{sequences[name]}\n' for name in sorted(sequences)) + (tmp / 'assembly.fasta').write_text(fa) + with gzip.open(source['reads_1'], 'rt') as inp, (tmp / 'r1.fq').open('w') as r1, (tmp / 'r2.fq').open('w') as r2: + scanned = 0 + for a, b in pair_records(inp): + r1.writelines(a) + r2.writelines(b) + scanned += 1 + if scanned == 250000: + break + cmd = [args.minimap2, '-ax', 'sr', '-t', '2', str(tmp / 'assembly.fasta'), str(tmp / 'r1.fq'), str(tmp / 'r2.fq')] + with (tmp / 'align.sam').open('w') as sam, (out / f'{sample}.selection.log').open('w') as log: + subprocess.run(cmd, stdout=sam, stderr=log, check=True) + hits, per_contig = {}, {name: set() for name in sequences} + with (tmp / 'align.sam').open() as sam: + for line in sam: + if line.startswith('@'): + continue + f = line.split('\t') + flag = int(f[1]) + if flag & (4 | 256 | 2048) or int(f[4]) < 20: + continue + name = re.sub(r'/[12]$', '', f[0]) + hits.setdefault(name, set()).add(f[2]) + r1_out, r2_out, kept = [], [], 0 + with (tmp / 'r1.fq').open() as r1, (tmp / 'r2.fq').open() as r2: + for _ in range(scanned): + a, b = [r1.readline() for _ in range(4)], [r2.readline() for _ in range(4)] + name = a[0].split()[0][1:-2] + if name in hits: + r1_out.extend(a) + r2_out.extend(b) + kept += 1 + for contig in hits[name]: + per_contig[contig].add(name) + if kept == 3000: + break + if any(len(v) < 10 for v in per_contig.values()): + raise ValueError(f'{sample}: insufficient matched pairs: {[(k, len(v)) for k, v in per_contig.items()]}') + write_gzip(out / 'inputs' / 'assemblies' / f'{sample}.fasta.gz', fa) + write_gzip(out / 'inputs' / 'reads' / f'{sample}_R1.fastq.gz', ''.join(r1_out)) + write_gzip(out / 'inputs' / 'reads' / f'{sample}_R2.fastq.gz', ''.join(r2_out)) + (out / 'inputs' / 'mag_maps' / f'{sample}.tsv').write_text(''.join(f'{name}\tCAMI_{re.sub("[^A-Za-z0-9_]", "_", chosen[name]["genome_id"])}\n' for name in sorted(sequences))) + rows.append(dict(zip(FIELDS, (sample, f'inputs/assemblies/{sample}.fasta.gz', 'paired', f'inputs/reads/{sample}_R1.fastq.gz', f'inputs/reads/{sample}_R2.fastq.gz', f'inputs/mag_maps/{sample}.tsv')))) + stats.append({'sample_id': sample, 'source_assembly': str(assembly), 'source_reads': source['reads_1'], 'source_mapping': str(mapping), 'scanned_pairs': scanned, 'retained_pairs': kept, 'assembly_bases': sum(map(len, sequences.values())), 'regions': [{**chosen[name], 'retained_contig_start_1based': 1, 'retained_contig_end_1based': len(sequences[name]), 'aligned_pairs': len(per_contig[name])} for name in sorted(sequences)]}) + print(f'{sample}: {stats[-1]["assembly_bases"]} bases, {kept} pairs, 3 genome bins', flush=True) + for label, subset in [('single', rows[:1]), ('pair', rows[1:]), ('all', rows)]: + with (out / f'{label}.tsv').open('w') as handle: + writer = csv.DictWriter(handle, fieldnames=FIELDS, delimiter='\t', lineterminator='\n') + writer.writeheader() + writer.writerows(subset) + provenance = {'source_dataset': 'CAMI II human-associated short-read gold-standard assemblies and simulated reads', 'source_doi': '10.4126/FRL01-006425518', 'selection': __doc__, 'minimap2_version': subprocess.check_output([args.minimap2, '--version'], text=True).strip(), 'samples': stats, 'sha256': {str(p.relative_to(out)): hashlib.sha256(p.read_bytes()).hexdigest() for p in sorted(out.rglob('*')) if p.is_file()}} + (out / 'provenance.json').write_text(json.dumps(provenance, indent=2) + '\n') + + +if __name__ == '__main__': + main() diff --git a/scripts/release.py b/scripts/release.py index 0e2bffa..e1c1010 100644 --- a/scripts/release.py +++ b/scripts/release.py @@ -125,6 +125,10 @@ def prepare(args): readme.write_text(re.sub(r"https://img.shields.io/badge/Version-[^)]*", f"https://img.shields.io/badge/Version-{value}-blue.svg", readme.read_text())) + citation = ROOT / "CITATION.cff" + if citation.exists(): + text = re.sub(r"^version:.*\n?", "", citation.read_text(), flags=re.M) + citation.write_text(text.rstrip() + f'\nversion: "{value}"\n') print(f"Prepared {value}, Conda build {build_number}. Review and commit before publishing.") @@ -147,9 +151,33 @@ def validate_run(directory): return {"successful_stages": sorted(STAGES), "nonempty_outputs": OUTPUTS} +def validate_citation(data, value): + if not isinstance(data, dict) or data.get('version') != value: + raise ValueError('CITATION.cff must match the release version.') + authors = data.get('authors') + if not isinstance(authors, list) or not authors: + raise ValueError('CITATION.cff must contain the reviewed manuscript authors.') + names = set() + for author in authors: + if not isinstance(author, dict) or not all( + isinstance(author.get(key), str) and author[key].strip() + for key in ('family-names', 'given-names', 'affiliation') + ): + raise ValueError('Each citation author needs a name and reviewed affiliation.') + name = (author['family-names'].strip().casefold(), author['given-names'].strip().casefold()) + if name in names: + raise ValueError('Duplicate citation author; review CITATION.cff.') + names.add(name) + + def build(args): clean() value = version() + import yaml + citation = ROOT / 'CITATION.cff' + if not citation.is_file() or (ROOT / '.zenodo.json').exists(): + raise ValueError('Release requires CITATION.cff without an overriding .zenodo.json.') + validate_citation(yaml.safe_load(citation.read_text()), value) number = recipe_build((ROOT / "conda_recipe/meta_template.yaml").read_text()) tag = release_tag(value, number) if args.tag and args.tag != tag: @@ -220,11 +248,16 @@ def build(args): "assert Version(version('setuptools')) >= Version('83.0.0'); " "print({n: version(n) for n in ['urllib3', 'setuptools']})", cwd=testdir, log=output / "security-dependencies.log") + run(*runner, "magsplitter", "--help", cwd=testdir, log=output / "magsplitter.log") + run(*runner, "python", "-c", "import camelot_frs", cwd=testdir, log=output / "camelot.log") + run(*runner, "metapathways", "prepare_test", "-o", "test", cwd=testdir, log=output / "test-inputs.log") run(*runner, "metapathways", "version", cwd=testdir, log=output / "cli-version.log") - run(*runner, "metapathways", "build_db", "--test", cwd=testdir, + run(*runner, "python", snapshot / "scripts/check_installed_assets.py", + cwd=testdir, log=output / "installed-assets.log") + run(*runner, "metapathways", "build_db", "--test", "--memory", "2 GB", "--max_memory", "4 GB", "--max_cpus", "2", cwd=testdir, log=output / "build-db.log") - run(*runner, "metapathways", "run", "--test", cwd=testdir, + run(*runner, "metapathways", "run", "--test", "--memory", "2 GB", "--max_memory", "4 GB", "--max_cpus", "2", cwd=testdir, log=output / "pipeline.log") validation.update(validate_run(testdir), scope="core-integration") for name in ["metapathways_steps_log.txt", "errors_warnings_log.txt"]: @@ -375,8 +408,9 @@ def publish(args): clean() value = version() branch = run("git", "branch", "--show-current", capture=True) - if branch != "dev": - raise ValueError("Publish from the dev branch.") + target_branch = getattr(args, "branch", "main") + if target_branch not in ("main", "dev") or branch != target_branch: + raise ValueError(f"Publish from the {target_branch} branch after PR review and testing.") url = run("git", "remote", "get-url", "--push", args.remote, capture=True) if url.removesuffix(".git").rstrip("/") not in ( f"git@github.com:{REPOSITORY}", f"https://github.com/{REPOSITORY}", @@ -394,8 +428,8 @@ def publish(args): raise ValueError(f"Local {tag} must be an annotated tag.") else: run("git", "tag", "-a", tag, "-m", f"MetaPathways {value}") - run("git", "push", "--atomic", args.remote, "HEAD:refs/heads/dev", f"refs/tags/{tag}") - print(f"CI will build, test, and publish {tag}: https://github.com/{REPOSITORY}/actions") + run("git", "push", "--atomic", args.remote, f"HEAD:refs/heads/{target_branch}", f"refs/tags/{tag}") + print(f"CI will build and test {tag}; publication requires manual selection: https://github.com/{REPOSITORY}/actions") def upload_conda(args): @@ -405,7 +439,7 @@ def upload_conda(args): raise ValueError("Expected exactly one validated Conda package.") label = "rc" if "rc" in manifest["version"] else "main" # anaconda-client reads BINSTAR_API_TOKEN; never put credentials on the command line. - run("anaconda", "upload", "--user", "hallamlab", "--label", label, packages[0]) + run("anaconda", "upload", "--user", os.environ.get("ANACONDA_OWNER", "hallamlab"), "--label", label, packages[0]) def container_command(args): @@ -435,11 +469,13 @@ def main(): p.add_argument("tag") p.add_argument("--output", default=str(ROOT / "dist/release")) p.set_defaults(func=github_release) - p = commands.add_parser("publish", help="Push dev and its release tag using existing Git credentials.") + p = commands.add_parser("publish", help="Push the reviewed main/dev branch and its release tag using existing Git credentials.") p.add_argument("--remote", default="origin") + p.add_argument("--branch", choices=("main", "dev"), default="main", + help="Reviewed release branch [main]; never a feature branch.") p.set_defaults(func=publish) for command, action in [("container-build", "build"), ("container-push", "push"), - ("container-attach", "attach"), ("quay-description", "description")]: + ("container-verify", "verify"), ("container-attach", "attach"), ("quay-description", "description")]: p = commands.add_parser(command) p.add_argument("--output", default=str(ROOT / "dist/release")) p.add_argument("--container-output", default=str(ROOT / "dist/containers")) diff --git a/scripts/render_detailed_workflow.py b/scripts/render_detailed_workflow.py new file mode 100644 index 0000000..3d24114 --- /dev/null +++ b/scripts/render_detailed_workflow.py @@ -0,0 +1,96 @@ +"""Render a code-based software workflow in the shared Hallam documentation style. + +The JSON describes public software behavior, not manuscript results. Module rows +are conceptual groups; branching lanes represent evidence that converges. +""" +import html +import json + + +def render(root, stem="detailed-nodal-workflow", output="workflow-detailed"): + data = json.loads((root / f'docs/diagrams/{stem}.json').read_text()) + width = 1720 + heights = [290 if 'lanes' in row else 190 for row in data['rows']] + height = 310 + sum(heights) + parts = [f'', + f'{data["title"]} complete software workflow', + 'Numbered conceptual modules with inputs, computational steps, data products and optional branches. See the workflow guide for source-code references and execution order.', + '', + '', + ''] + + def text(x, y, label, size=22, anchor='middle'): + for i, line in enumerate(label.split('\n')): + parts.append(f'{html.escape(line)}') + + def rect(x,y,w,h,fill,stroke,rx=0): + parts.append(f'') + + def wire(points, arrow=True): + path='M'+' L'.join(f'{x} {y}' for x,y in points) + marker=' marker-end="url(#arrow)"' if arrow else '' + parts.append(f'') + + def symbol(x,y,kind): + if kind=='compute': + parts.append(f'') + else: + fill,stroke={'input':('#DAE8FC','#6C8EBF'),'output':('#D5E8D4','#82B366'),'data':('#F5F5F5','#666666')}[kind] + parts.append(f'') + + def node(x,y,n): + symbol(x,y,n['kind']) + text(x,y-25-(len(n['label'].split('\n'))-1)*25,n['label']) + if n['tool']:text(x,y+37,n['tool'],19) + + # Title first, then inputs and the symbol key; module geometry stays fixed. + rect(20,20,1680,78,'#CCCCCC','#666666',16) + text(860,53,data['controller'],28) + text(860,82,data['execution'],20) + rect(20,120,975,130,'#DAE8FC','#6C8EBF',16) + text(45,151,'Inputs',28,'start') + for i,line in enumerate(data['inputs']):text(45,187+i*29,line,21,'start') + rect(1030,120,670,110,'#CCCCCC','#666666',16) + for x,label,kind in [(1090,'Module',None),(1220,'Compute','compute'),(1350,'Data','data'),(1480,'Input','input'),(1610,'Output','output')]: + text(x,152,label,21) + if kind:symbol(x,189,kind) + else:rect(x-12,177,24,24,'#F5F5F5','#111111') + top=290 + previous_y = None + for index,(row,span) in enumerate(zip(data['rows'],heights),1): + y=top+span/2-15 + if previous_y is not None: + wire([(326, previous_y + 14), (326, y - 14)]) + previous_y = y + # Module names are explicitly wrapped to fit the same label column. + import textwrap + title='\n'.join(textwrap.wrap(row['title'],22)) + text(160,y-(len(title.split('\n'))-1)*15,title,27) + rect(313,y-13,26,26,'#F5F5F5','#111111') + text(326,y+7,str(index),20) + if 'lanes' in row: + ys=[y-64,y+64] + wire([(339,y),(350,y)],False) + for lane,ly in zip(row['lanes'],ys): + wire([(350,y),(350,ly),(448,ly)]) + for x,n in zip([460,820,1160],lane):node(x,ly,n) + wire([(473,ly),(807,ly)]) + wire([(833,ly),(1147,ly)]) + wire([(1173,ly),(1370,ly),(1370,y)],False) + wire([(1370,y),(1548,y)]) + node(1560,y,dict(label=row['result'],tool='',kind='output')) + else: + nodes=row['nodes'];xs=[460+i*1100/(len(nodes)-1) for i in range(len(nodes))] + wire([(339,y),(448,y)]) + for x,n in zip(xs,nodes):node(x,y,n) + for a,b in zip(xs,xs[1:]):wire([(a+13,y),(b-13,y)]) + top+=span + parts.extend(['','']) + source=root/f'docs/assets/{output}.svg' + source.write_text('\n'.join(parts)+'\n') + return source,width,height + + +if __name__ == "__main__": + from pathlib import Path + render(Path(__file__).resolve().parents[1]) diff --git a/scripts/render_workflow_diagrams.py b/scripts/render_workflow_diagrams.py new file mode 100644 index 0000000..7140632 --- /dev/null +++ b/scripts/render_workflow_diagrams.py @@ -0,0 +1,85 @@ +#!/usr/bin/env python3 +"""Rebuild shared-scale SVG previews from Mermaid and publication SVG sources. + +Run from the repository root after installing docs/diagram-requirements.txt +and running `python -m playwright install chromium`. Rendering downloads the +pinned Mermaid bundle; documentation builds use the committed SVGs offline. +""" +from pathlib import Path +import json +import argparse +import xml.etree.ElementTree as ET +from playwright.sync_api import sync_playwright + +ROOT = Path(__file__).resolve().parents[1] +NS = '{http://www.w3.org/2000/svg}' +ET.register_namespace('', NS[1:-1]) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--mermaid-js", type=Path, help="Local Mermaid 11.12.1 bundle") + args = parser.parse_args() + config = json.loads((ROOT / 'docs/diagrams/figures.json').read_text()) + from render_detailed_workflow import render + detailed_svg, detailed_width, detailed_height = render(ROOT) + rendered = [] + with sync_playwright() as p: + browser = p.chromium.launch(headless=True) + page = browser.new_page() + page.goto('about:blank') + if args.mermaid_js: + page.add_script_tag(path=str(args.mermaid_js.resolve())) + else: + page.add_script_tag(url='https://cdn.jsdelivr.net/npm/mermaid@11.12.1/dist/mermaid.min.js') + page.evaluate('mermaid.initialize({startOnLoad:false,securityLevel:"strict"})') + for item in config['figures']: + code = (ROOT / 'docs/diagrams' / (item['name'] + '.mmd')).read_text() + svg = page.evaluate('''async ({code,id}) => { + const result = await mermaid.render(id, code); + const div = document.createElement('div'); + div.innerHTML = result.svg; + return new XMLSerializer().serializeToString(div.firstElementChild); + }''', {'code': code, 'id': ROOT.name.lower() + '-' + item['name']}) + rendered.append((item['name'], ET.fromstring(svg), config['mermaid_scale'])) + page.set_content('' + detailed_svg.read_text() + '') + page.pdf(path=str(detailed_svg.with_suffix('.pdf')), width=f'{detailed_width}px', height=f'{detailed_height}px', print_background=True, margin=dict(top='0', right='0', bottom='0', left='0')) + browser.close() + for name in ['workflow', 'workflow-brief', 'workflow-detailed']: + source_name = 'workflow-main' if name == 'workflow' else name + source = ROOT / 'docs/assets' / (source_name + '.svg') + if source.exists(): + rendered.append((name, ET.parse(source).getroot(), 1)) + width = config['canvas_width'] + destination = ROOT / 'docs/assets/diagrams' + destination.mkdir(exist_ok=True) + # Check every diagram before replacing any previews. Keep the common width + # synchronized between ASPIRE and MetaPathways if a future figure needs more. + for name, element, scale in rendered: + natural_width = float(element.get('viewBox').split()[2]) * scale + if natural_width > width - 40: + raise SystemExit(f'{name} needs {natural_width + 40:g} canvas units; update the shared canvas width.') + for name, element, scale in rendered: + _, _, w, h = map(float, element.get('viewBox').split()) + w *= scale + h *= scale + canvas = ET.Element(NS + 'svg', { + 'viewBox': f'0 0 {width} {h + 40:g}', 'width': str(width), + 'height': f'{h + 40:g}', 'role': 'img', + }) + ET.SubElement(canvas, NS + 'title').text = name.replace('-', ' ').title() + ET.SubElement(canvas, NS + 'rect', { + 'width': str(width), 'height': f'{h + 40:g}', 'fill': '#ffffff', + }) + element.set('x', f'{(width - w) / 2:g}') + element.set('y', '20') + element.set('width', f'{w:g}') + element.set('height', f'{h:g}') + element.attrib.pop('style', None) + canvas.append(element) + (destination / (name + '.svg')).write_text(ET.tostring(canvas, encoding='unicode') + '\n') + print(f'Rendered {name} on {width}-unit canvas') + + +if __name__ == '__main__': + main() diff --git a/setup.py b/setup.py index ef163f8..67a5890 100755 --- a/setup.py +++ b/setup.py @@ -1,6 +1,14 @@ import os from pathlib import Path from setuptools import setup, find_packages +from setuptools.dist import Distribution + + +class PlatformDistribution(Distribution): + """The bundled FAST/metacount executables are native Linux binaries.""" + + def has_ext_modules(self): + return True PACKAGE_ROOT = Path(os.path.realpath(__file__)).parent NAME = "metapathways".lower() @@ -15,7 +23,7 @@ } VERSION = ver_dict['version'].strip('"') CLASSIFIERS = [ - "Development Status :: 2 - Pre-Alpha", + "Development Status :: 5 - Production/Stable", "Environment :: Console", "Intended Audience :: Science/Research", "Natural Language :: English", @@ -30,6 +38,7 @@ def read(fname): if __name__ == "__main__": setup( + distclass=PlatformDistribution, name=NAME, version=VERSION, author=ver_dict['author'], @@ -41,8 +50,8 @@ def read(fname): license=ver_dict['license'], keywords="metagenomics pipeline", url="https://github.com/hallamlab/MetaPathways/", - packages=find_packages(), - package_data={'': ['metapathways/regtests/**/*'],}, + packages=find_packages(include=["metapathways", "metapathways.*"]), + package_data={'': ['metapathways/regtests/**/*'], 'metapathways': ['report_assets/*', 'resources/*.json']}, scripts=["bin/metapathways-install-deps.sh", "bin/metapathways-data-install.sh", "dev/metacount", @@ -65,11 +74,11 @@ def read(fname): entry_points={"console_scripts": ENTRY_POINTS}, long_description=read("README.md"), include_package_data=True, - data_files=[('data_file_test', ['README.md', 'Makefile'])], classifiers=CLASSIFIERS, extras_require={ "test": ["pytest", "pytest-cov", "tox"], }, python_requires=">=3.10", - install_requires=[], + install_requires=[line.strip() for line in read("requirements-workflow.txt").splitlines() + if line.strip() and not line.startswith("#")], ) diff --git a/tests/release/test_controls.py b/tests/release/test_controls.py new file mode 100644 index 0000000..103d0e0 --- /dev/null +++ b/tests/release/test_controls.py @@ -0,0 +1,31 @@ +import itertools +from pathlib import Path +import sys +import unittest + +sys.path.insert(0, str(Path(__file__).resolve().parents[2] / 'scripts')) +from check_release_controls import validate + + +class ControlTests(unittest.TestCase): + def test_every_destination_combination_requires_manual_tagged_publication(self): + for flags in itertools.product((False, True), repeat=3): + selections = dict(zip(('anaconda', 'quay', 'github'), flags)) + with self.subTest(flags=flags): + validate('workflow_dispatch', 'hallamlab/MetaPathways', 'tag', 'v3.5.2', **selections) + validate('workflow_dispatch', 'hallamlab/MetaPathways', 'branch', 'dev', 'v3.5.2', **selections) + if any(flags): + for event, repo, kind, ref in ( + ('push', 'hallamlab/MetaPathways', 'tag', 'v3.5.2'), + ('workflow_dispatch', 'hallamlab/MetaPathways', 'branch', 'dev'), + ('workflow_dispatch', 'fork/MetaPathways', 'tag', 'v3.5.2'), + ('workflow_dispatch', 'hallamlab/MetaPathways', 'tag', 'invalid'), + ): + with self.assertRaises(ValueError): + validate(event, repo, kind, ref, **selections) + else: + validate('push', 'fork/MetaPathways', 'branch', 'feature', **selections) + + def test_strings_cannot_enable_publication(self): + with self.assertRaises(ValueError): + validate('workflow_dispatch', 'hallamlab/MetaPathways', 'tag', 'v3.5.2', anaconda='false') diff --git a/tests/release/test_release.py b/tests/release/test_release.py index 51030f2..e8add80 100644 --- a/tests/release/test_release.py +++ b/tests/release/test_release.py @@ -15,6 +15,16 @@ class ReleaseTests(unittest.TestCase): + def test_citation_requires_reviewed_unambiguous_authors(self): + author = {'given-names': 'Example', 'family-names': 'Author', 'affiliation': 'Reviewed institution'} + release.validate_citation({'version': '3.5.2', 'authors': [author]}, '3.5.2') + for data in ({}, {'version': '3.5.1', 'authors': [author]}, + {'version': '3.5.2', 'authors': []}, + {'version': '3.5.2', 'authors': [author, author]}, + {'version': '3.5.2', 'authors': [{'name': 'GitHub display name'}]}): + with self.subTest(data=data), self.assertRaises(ValueError): + release.validate_citation(data, '3.5.2') + def test_explicit_versions_only(self): for value in ["3.5.0", "v3.5.1", "3.6.0rc1"]: self.assertEqual(release.valid_version(value), value.removeprefix("v")) @@ -40,9 +50,11 @@ def test_preparation_preserves_unrelated_changes(self): (root / "conda_recipe/meta_template.yaml").write_text("build:\n number: 0\n") (root / "README.md").write_text( "User edit\nhttps://img.shields.io/badge/Version-3.5-blue.svg)\n") + (root / "CITATION.cff").write_text('title: MetaPathways\nversion: "3.5.0"\n') with patch.object(release, "ROOT", root): release.prepare(SimpleNamespace(version="3.5.1rc1", build_number=2)) self.assertEqual(release.version(root), "3.5.1rc1") + self.assertIn('version: "3.5.1rc1"', (root / "CITATION.cff").read_text()) self.assertIn("User edit", (root / "README.md").read_text()) self.assertIn("number: 2", (root / "conda_recipe/meta_template.yaml").read_text()) with patch.object(release, "ROOT", root): @@ -51,6 +63,7 @@ def test_preparation_preserves_unrelated_changes(self): with patch.object(release, "ROOT", root): release.prepare(SimpleNamespace(version="3.5.1", build_number=None)) self.assertIn("number: 0", (root / "conda_recipe/meta_template.yaml").read_text()) + self.assertEqual((root / "CITATION.cff").read_text(), 'title: MetaPathways\nversion: "3.5.1"\n') def fixture(self, root): sample = root / "test/k12_test" @@ -94,7 +107,7 @@ def test_dirty_checkout_stops_build_before_artifacts(self): release.build(SimpleNamespace()) def test_existing_remote_tag_never_overwritten(self): - responses = ["", "dev", "git@github.com:hallamlab/MetaPathways.git", + responses = ["", "main", "git@github.com:hallamlab/MetaPathways.git", "abc refs/tags/v3.5.0"] with patch.object(release, "recipe_build", return_value=0), patch.object(release, "version", return_value="3.5.0"), patch.object( release, "run", side_effect=responses @@ -103,13 +116,21 @@ def test_existing_remote_tag_never_overwritten(self): release.publish(SimpleNamespace(remote="origin")) self.assertFalse(any("push" in c.args or "tag" in c.args for c in run.call_args_list)) + def test_feature_branch_cannot_publish(self): + with patch.object(release, "version", return_value="3.5.2"), patch.object( + release, "run", side_effect=["", "feat/example"] + ) as run: + with self.assertRaisesRegex(ValueError, "main branch"): + release.publish(SimpleNamespace(remote="origin")) + self.assertFalse(any("push" in c.args or "tag" in c.args for c in run.call_args_list)) + def test_publish_pushes_annotated_tag_and_branch_atomically(self): with tempfile.TemporaryDirectory() as temp: root = Path(temp) remote, checkout = root / "remote.git", root / "checkout" subprocess.run(["git", "init", "--bare", str(remote)], check=True, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) - subprocess.run(["git", "init", "-b", "dev", str(checkout)], check=True, + subprocess.run(["git", "init", "-b", "main", str(checkout)], check=True, stdout=subprocess.DEVNULL) (checkout / "README.md").write_text("Release fixture\n") (checkout / "conda_recipe").mkdir() @@ -140,7 +161,7 @@ def local_run(*args, **kwargs): ["git", "cat-file", "-t", "v3.5.0"], cwd=remote, text=True).strip() self.assertEqual(kind, "tag") branch = subprocess.check_output( - ["git", "rev-parse", "dev"], cwd=remote, text=True).strip() + ["git", "rev-parse", "main"], cwd=remote, text=True).strip() tagged = subprocess.check_output( ["git", "rev-parse", "v3.5.0^{commit}"], cwd=remote, text=True).strip() self.assertEqual(branch, tagged) diff --git a/tests/test_analysis_workflow.py b/tests/test_analysis_workflow.py new file mode 100644 index 0000000..52a8833 --- /dev/null +++ b/tests/test_analysis_workflow.py @@ -0,0 +1,226 @@ +"""Multi-sample discovery and DAG tests; no biological tools are executed.""" +import csv +import json +import os +from pathlib import Path +import tempfile +import unittest +from unittest.mock import patch + +from metapathways import analysis_workflow as wf, nextflow as nf, pipeline + + +class AnalysisTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + self.root = Path(self.tmp.name) + self.inputs = self.root / 'inputs' + for folder in ('assemblies', 'reads', 'mag_maps'): + (self.inputs / folder).mkdir(parents=True) + self.write('assemblies/Alpha.fa', '>ctg1\nACGT\n') + self.write('assemblies/Beta.fna', '>ctg2\nACGT\n') + self.write('reads/Alpha_R1.fq', '@r/1\nAC\n+\nII\n') + self.write('reads/Alpha_R2.fq', '@r/2\nGT\n+\nII\n') + self.write('reads/Beta_interleaved.fastq', '@b/1\nAC\n+\nII\n') + self.write('mag_maps/Alpha.tsv', 'ctg1\tmb2.1\n') + self.write('mag_maps/Beta.tsv', 'ctg2\tmb2.1\n') + + def write(self, name, text): + (self.inputs / name).write_text(text) + + def args(self, *extra): + return wf.parser().parse_args(['analysis_wf', '-i', str(self.inputs), + '-o', str(self.root / 'out'), '-d', '/fake/db', *extra]) + + def rows(self): + return wf.validate(wf.discover(self.args())) + + def test_discovery_pairs_reads_and_normalizes_mag_ids(self): + rows = self.rows() + self.assertEqual([r['sample_id'] for r in rows], ['Alpha', 'Beta']) + self.assertEqual([r['read_layout'] for r in rows], ['paired', 'interleaved']) + self.assertEqual(rows[0]['entities'], ['mb2_1']) + self.assertNotEqual(rows[0]['reads_1'], rows[1]['reads_1']) + + def test_missing_mate_points_to_documentation(self): + (self.inputs / 'reads/Alpha_R2.fq').unlink() + with self.assertRaisesRegex(ValueError, 'analysis.html#custom-analysis-manifest'): + self.rows() + + def test_orphan_and_ambiguous_files_fail(self): + for name, text in [('reads/unknown_single.fq', 'x'), + ('reads/Alpha_single.fq', 'x'), + ('assemblies/Alpha.fasta', '>ctg1\nAC\n'), + ('mag_maps/unknown.tsv', 'c\tm\n')]: + with self.subTest(name=name): + self.write(name, text) + with self.assertRaises(ValueError): + self.rows() + (self.inputs / name).unlink() + + def test_hardlinked_mates_fail(self): + mate = self.inputs / 'reads/Alpha_R2.fq' + mate.unlink() + os.link(self.inputs / 'reads/Alpha_R1.fq', mate) + with self.assertRaisesRegex(ValueError, 'same file'): + self.rows() + + def test_bad_maps_fail_before_jobs(self): + for text in ('ctg_missing\tMAG1\n', 'ctg1\tcommunity\n', + 'ctg1\tmb2.1\nctg2\tmb2_1\n', 'ctg1\tMAG1\nctg1\tMAG2\n'): + with self.subTest(text=text): + self.write('mag_maps/Alpha.tsv', text) + with self.assertRaises(ValueError): + self.rows() + + def test_manifest_relative_paths_and_mixed_optional_branches(self): + manifest = self.inputs / 'custom.tsv' + with manifest.open('w', newline='') as handle: + writer = csv.writer(handle, delimiter='\t') + writer.writerow(wf.FIELDS) + writer.writerow(['DifferentID', 'assemblies/Alpha.fa', 'none', '', '', '']) + writer.writerow(['Beta', 'assemblies/Beta.fna', 'single', 'reads/Beta_interleaved.fastq', '', 'mag_maps/Beta.tsv']) + rows = wf.validate(wf.read_manifest(manifest)) + self.assertEqual(rows[1]['read_layout'], 'none') + self.assertEqual(rows[1]['entities'], []) + self.assertTrue(Path(rows[0]['assembly']).is_absolute()) + + def test_resolved_manifest_guards_resume_and_aliases(self): + output = self.root / 'out' + output.mkdir() + rows = self.rows() + staged = wf.save_inputs(rows, output) + alias = Path(staged[0]['assembly']) + self.assertTrue(alias.is_symlink()) + stamp = alias.lstat().st_mtime_ns + wf.save_inputs(rows, output) + self.assertEqual(alias.lstat().st_mtime_ns, stamp) + with self.assertRaisesRegex(ValueError, 'different inputs'): + wf.save_inputs(rows[:1], output) + + def test_dependencies_are_per_sample_and_pgdb_failures_optional(self): + tasks = [] + for row in self.rows(): + annotation = [nf.task(row['sample_id'] + ':path', 'path', [], + context={'name': 'PATHOLOGIC_INPUT'})] + tasks += annotation + wf.downstream(row, self.root / 'out', annotation, self.args(), '/fake/image.sif') + nf.ordered(tasks) + indexed = {t['id']: t for t in tasks} + self.assertEqual(indexed['Alpha:pgdb:community']['dependencies'], ['Alpha:path']) + self.assertEqual(indexed['Alpha:pgdb:mb2_1']['dependencies'], ['Alpha:mag_split']) + self.assertEqual(indexed['Beta:mag_split']['dependencies'], ['Beta:path']) + for sample in ('Alpha', 'Beta'): + split = indexed[f'{sample}:mag_split'] + self.assertEqual(split['cache_version'], 'authoritative-mag-coordinates-v1') + self.assertFalse(split['adopt_existing']) + self.assertIn(str(self.root / f'out/{sample}/results/annotation_table/{sample}.ptinput.tsv'), split['inputs']) + self.assertTrue(indexed['Alpha:pgdb:mb2_1']['allow_failure']) + self.assertFalse(indexed['Alpha:pgdb:community']['allow_failure']) + self.assertEqual(indexed['Alpha:pgdb:mb2_1']['cpus'], 1) + self.assertEqual(indexed['Alpha:pgdb:community']['memory'], '16 GB') + for task in tasks: + self.assertTrue(all(d.split(':')[0] == task['id'].split(':')[0] for d in task['dependencies'])) + + def test_one_memory_request_applies_to_all_workflow_jobs_unless_overridden(self): + row = self.rows()[0] + annotation = [nf.task('Alpha:path', 'path', [], context={'name': 'PATHOLOGIC_INPUT'})] + for executor in ('local', 'slurm'): + args = self.args('--memory', '64 GB', '--executor', executor) + tasks = wf.downstream(row, self.root / 'out', annotation, args, '/fake/image.sif') + self.assertTrue(all(t['memory'] == '64 GB' for t in tasks)) + self.assertTrue(all(t['cpus'] == 1 for t in tasks)) + args.ptools_memory = '4 GB' + tasks = wf.downstream(row, self.root / 'out', annotation, args, '/fake/image.sif') + for t in tasks: + self.assertEqual(t['memory'], '4 GB' if ':pgdb:' in t['id'] else '64 GB') + + def test_standalone_mag_split_invalidates_legacy_cache_and_tracks_coordinates(self): + base = self.root / 'sample' + (base / 'results/annotation_table').mkdir(parents=True) + (base / 'preprocessed').mkdir() + (base / 'results/annotation_table/sample.ORF_annotation_table.txt').touch() + (base / 'preprocessed/sample.mapping.txt').touch() + with patch('sys.argv', ['metapathways', 'mag_split', '-o', str(base), + '-m', str(self.inputs / 'mag_maps/Alpha.tsv')]), \ + patch.object(nf, 'launch') as launch: + pipeline.mag_split() + task = launch.call_args.args[0][0] + self.assertEqual(task['cache_version'], 'authoritative-mag-coordinates-v1') + self.assertFalse(task['adopt_existing']) + self.assertIn(str(base / 'results/annotation_table/sample.ptinput.tsv'), task['inputs']) + + def test_one_launch_contains_all_samples_with_distinct_reads(self): + planned = [] + def prepare(args, parser): + planned.append(args) + name = Path(args.input_file).stem + return [nf.task(name + ':path', name, [], context={'name': 'PATHOLOGIC_INPUT'})], args.output_dir + with patch.object(pipeline, 'prepare_annotation', side_effect=prepare), \ + patch.object(wf.shutil, 'which', return_value='/mock/tool'), \ + patch.object(nf, 'launch') as launch: + wf.main(['-i', str(self.inputs), '-o', str(self.root / 'out'), '-d', '/fake/db', + '--skip_ptools', '--threads', '8', '--max_cpus', '32', '--dryrun']) + launch.assert_called_once() + self.assertEqual(len(launch.call_args.args[0]), 4) + self.assertNotEqual(planned[0].fwd_fastq, planned[1].fwd_fastq) + self.assertFalse(planned[0].interleaved) + self.assertTrue(planned[1].interleaved) + self.assertTrue(launch.call_args.kwargs['dryrun']) + + def test_invalid_input_never_launches(self): + (self.inputs / 'reads/Alpha_R2.fq').unlink() + with patch.object(nf, 'launch') as launch, self.assertRaises(ValueError): + wf.main(['-i', str(self.inputs), '-o', str(self.root / 'out'), '-d', '/fake/db', '--skip_ptools']) + launch.assert_not_called() + + def test_compact_cleanup_is_per_sample_terminal_and_resume_skips_complete(self): + def prepare(args, parser): + name = Path(args.input_file).stem + return [nf.task(name+':path', name, [], sample=name, context={'name':'PATHOLOGIC_INPUT'}), + nf.task(name+':tpm', name, [], sample=name)], args.output_dir + command = ['-i', str(self.inputs), '-o', str(self.root/'out'), '-d', '/fake/db', + '--skip_ptools', '--compact_results', '--dryrun'] + with patch.object(pipeline, 'prepare_annotation', side_effect=prepare), \ + patch.object(wf.shutil, 'which', return_value='/mock/tool'), \ + patch.object(nf, 'launch') as launch: + wf.main(command) + tasks = launch.call_args.args[0] + for sample in ('Alpha','Beta'): + cleanup = next(t for t in tasks if t['id'] == sample+':compact_results') + self.assertEqual(set(cleanup['dependencies']), {sample+':path', sample+':tpm', sample+':mag_split'}) + with patch('metapathways.compact_results.marker_state', side_effect=['complete', None]): + wf.main(command) + self.assertTrue(all(t['sample']=='Beta' for t in launch.call_args.args[0])) + + def test_compact_rejects_shared_cache_and_retained_work(self): + for flags in (['--keep_work'], ['--work_dir','/tmp/work'], ['--conda_cache','/tmp/cache']): + with self.assertRaisesRegex(ValueError, 'cannot be combined'): + wf.main(['--compact_results', *flags]) + + def test_planners_do_not_share_parameter_state(self): + from metapathways.jobscreator import ContextCreator + a = ContextCreator({'g': {'key': 'a'}}, {'NUM_CPUS': 8}) + b = ContextCreator({'g': {'key': 'b'}}, {'NUM_CPUS': 3}) + self.assertEqual(a.params.get('g', 'key'), 'a') + self.assertEqual(b.params.get('g', 'key'), 'b') + self.assertEqual((a.configs.NUM_CPUS, b.configs.NUM_CPUS), (8, 3)) + + def test_missing_mag_input_is_skipped_without_launching_tool(self): + from metapathways.nf_worker import execute + task = nf.task('mag', 'mag', ['must-not-run'], inputs=['absent'], + outputs=['absent-output'], skip_if_missing=str(self.root / 'absent/0.pf'), + receipt=str(self.root / 'cache.json')) + previous = Path.cwd() + try: + os.chdir(self.root) + with patch('metapathways.nf_worker.subprocess.run') as run: + self.assertEqual(execute(task), 0) + run.assert_not_called() + self.assertEqual(json.loads(Path('receipt.json').read_text())['status'], 'SKIPPED') + finally: + os.chdir(previous) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_annotation_taxonomy_status.py b/tests/test_annotation_taxonomy_status.py new file mode 100644 index 0000000..667d298 --- /dev/null +++ b/tests/test_annotation_taxonomy_status.py @@ -0,0 +1,147 @@ +import csv +import io +import shlex +import tempfile +import unittest +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import MagicMock, patch + +from metapathways.LCAComputation import LCAComputation +from metapathways.MetaPathways_create_reports_fast import create_annotation, print_orf_table +from metapathways.protein_taxonomy import hit_taxid, hit_taxonomy, raw_lca + + +def tree(): + lca = LCAComputation([]) + lca.id_to_name = {'1': 'root', '2': 'Bacteria', '3': 'Species A', '4': 'Species B', '5': 'Archaea'} + lca.name_to_id = {v: k for k, v in lca.id_to_name.items()} + lca.taxid_to_ptaxid = {i: [p, 0, 0] for i, p in [('1','1'), ('2','1'), ('3','2'), ('4','2'), ('5','1')]} + return lca + + +class TaxonomyStatusTests(unittest.TestCase): + def test_rna_only_and_empty_ptinput_keep_coordinate_schema(self): + from metapathways.MetaPathways_create_genbank_ptinput import ptinput_dataframe + rna = {'rna1': {'id': 'rna1', 'seqname': 'contig1', + 'start': 2, 'end': 9, 'strand': '+'}} + frame = ptinput_dataframe(rna, {'contig1': 'A' * 20}) + self.assertEqual(frame.loc['rna1', 'contig_length'], 20) + empty = ptinput_dataframe({}, {}) + self.assertTrue(empty.empty) + self.assertTrue({'id', 'seqname', 'start', 'end', 'strand', 'contig_length'} <= set(empty)) + + def test_orf_map_preserves_unannotated_cds_across_batches(self): + output = io.StringIO() + with tempfile.TemporaryDirectory() as directory: + print_orf_table({'swissprot_test': {}}, {'C1-G1': 'sample-C1'}, directory, output) + print_orf_table({'swissprot_test': {'C2-G1': [ + {'query': 'C2-G1', 'product': 'enzyme'}]}}, + {'C2-G1': 'sample-C2', 'C2-G2': 'sample-C2'}, directory, output) + rows = list(csv.reader(io.StringIO(output.getvalue()), delimiter='\t')) + self.assertEqual(rows, [['# ORF_ID', 'CONTIG_ID', 'swissprot_test'], + ['C1-G1', 'sample-C1', ''], ['C2-G1', 'sample-C2', 'enzyme'], + ['C2-G2', 'sample-C2', '']]) + + def test_empty_orf_map_has_header(self): + output = io.StringIO() + with tempfile.TemporaryDirectory() as directory: + print_orf_table({'swissprot_test': {}}, {}, directory, output) + self.assertEqual(output.getvalue(), '# ORF_ID\tCONTIG_ID\tswissprot_test\n') + + def test_report_task_lists_selected_inputs_and_checkpoints_taxonomy(self): + from metapathways.jobscreator import ContextCreator + creator = ContextCreator.__new__(ContextCreator) + creator.configs = SimpleNamespace(REFDBS='/db', CREATE_ANNOT_REPORTS='report-script') + creator.params = MagicMock() + creator.params.get.return_value = 'yes' + creator.get_dbs = lambda: ['swissprot', 'metacyc'] + sample = SimpleNamespace(genbank_dir='/out/genbank', sample_name='sample', + output_results_annotation_table_dir='/out/annotations', blast_results_dir='/out/blast', algorithm='FAST') + context = creator.create_report_files_cmd(sample)[0] + command = shlex.split(context.commands[0]) + self.assertNotIn('-D', command) + self.assertEqual([command[i+1] for i,x in enumerate(command) if x == '-d'], ['swissprot','metacyc']) + self.assertEqual(context.outputs['annotation_taxonomy'], '/out/annotations/sample.annotation_taxonomy.tsv') + + def test_taxid_header_forms(self): + for text in ('Protein OS=Name OX=3 GN=g', 'Protein OS Name OX 3 GN g'): + self.assertEqual(hit_taxid({'product': text}, 'swissprot'), '3') + self.assertEqual(hit_taxid({'product':'protein', 'comment':'OS=Name OX=3'}, 'swissprot_test'), '3') + self.assertEqual(hit_taxid({'product':'protein TaxID=4'}, 'uniref50'), '4') + self.assertEqual(hit_taxid({'target':'5.protein'}, 'eggnog'), '5') + self.assertIsNone(hit_taxid({'product':'OX=bad'}, 'swissprot')) + self.assertIsNone(hit_taxid({'product':'OX=3bad'}, 'swissprot')) + + def test_missing_unknown_and_disabled_are_distinct(self): + lca = tree() + self.assertEqual(hit_taxonomy({'product':'OX=999'}, 'swissprot', lca), ('999','Unclassified')) + self.assertEqual(hit_taxonomy({}, 'swissprot', lca), ('','Unclassified')) + self.assertEqual(hit_taxonomy({'product':'OX=3'}, 'metacyc', lca), ('','Not computed')) + self.assertEqual(hit_taxonomy({'product':'OX=1'}, 'swissprot', lca), ('1','root')) + + def test_lca_thresholds_and_real_root(self): + lca = tree() + hits = [{'product':'OX=3', 'bitscore':100}, {'product':'OX=4', 'bitscore':95}, + {'product':'OX=5', 'bitscore':20}, {'product':'OX=999', 'bitscore':100}] + self.assertEqual(raw_lca(hits, 'swissprot', lca), '2') + self.assertEqual(raw_lca([{'product':'OX=3','bitscore':100}, {'product':'OX=5','bitscore':100}], 'swissprot', lca), '1') + self.assertIsNone(raw_lca([{'product':'OX=3','bitscore':20}], 'swissprot', lca)) + self.assertIsNone(raw_lca([{'product':'OX=999','bitscore':100}], 'swissprot', lca)) + self.assertEqual(raw_lca(hits, 'swissprot', lca), '2') # LCA scratch counters were cleared. + + def test_report_generation_is_independent_of_database_order(self): + from metapathways.MetaPathways_create_reports_fast import main + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + ncbi = root/'tree.tsv' + ncbi.write_text('root\t1\t1\nBacteria\t2\t1\nSpecies A\t3\t2\nSpecies B\t4\t2\nArchaea\t5\t1\n') + gff = root/'annot.gff' + gff.write_text(''.join( + f'C1\tprodigal\tCDS\t{start}\t{start+299}\t.\t+\t0\tID=G{i};orf_length=300;contig_length=1000;sourcedb=swissprot;target=S{i};product=protein\n' + for i,start in [(1,1),(2,400)])) + header = '#query\ttarget\tq_length\tbitscore\tbsr\texpect\taln_length\tidentity\tec\tproduct\n' + swiss = root/'swiss.parsed.txt' + swiss.write_text(header+''.join(f'G{i}\tS{i}\t300\t100\t1\t0\t300\t100\t\tProtein OS=Name OX={i+2} GN=g\n' for i in (1,2))) + uniref = root/'uniref.parsed.txt' + uniref.write_text(header+''.join(f'G{i}\tU{i}\t300\t100\t1\t0\t300\t100\t\tProtein n=1 Tax=Archaea TaxID=5 RepID=foo\n' for i in (1,2))) + eggnog = root/'eggnog.parsed.txt' + eggnog.write_text(header+'G1\t3.protein\t300\t100\t1\t0\t300\t100\t\tprotein\n') + expected = None + for iteration, databases in enumerate(([('swissprot',swiss),('uniref50',uniref),('eggnog',eggnog)], [('eggnog',eggnog),('uniref50',uniref),('swissprot',swiss)])): + output = root/str(iteration) + args = ['--input-annotated-gff',str(gff),'--output-dir',str(output),'--ncbi-taxonomy-map',str(ncbi),'-s','sample'] + for db, filename in databases: + args += ['-d',db,'-b',str(filename)] + main(args) + with (output/'sample.annotation_taxonomy.tsv').open() as stream: + rows = sorted((r['orf_id'],r['reference_db'],r['taxonomy'],r['lca_taxonomy']) for r in csv.DictReader(stream, delimiter='\t')) + self.assertIn(('G1','swissprot','Species A','Bacteria'), rows) + self.assertIn(('G1','uniref50','Archaea','Archaea'), rows) + self.assertIn(('G1','eggnog','Species A','root'), rows) # Support must not leak from SwissProt. + if expected is not None: + self.assertEqual(rows, expected) + expected = rows + + def test_primary_annotation_uses_its_own_database_and_target(self): + lca = tree() + reader = MagicMock() + reader.__iter__.return_value = iter(['contig1']) + reader.orf_dictionary = {'contig1': [{ + 'id': 'orf1', 'seqname': 'contig1', 'feature': 'CDS', 'sourcedb':'swissprot', + 'product': 'protein', 'orf_length':90, 'start':1, 'end':90, + 'contig_length':100, 'strand':'+', 'target':'hit1', + }]} + hits = {'swissprot': {'orf1':[{'target':'hit1','product':'OX=3'}]}, + 'uniref50': {'orf1':[{'target':'hit1','product':'TaxID=5'}]}} + taxons = {'swissprot':{'orf1':'Bacteria'}, 'uniref50':{'orf1':'Archaea'}} + with tempfile.TemporaryDirectory() as directory: + with patch('metapathways.MetaPathways_create_reports_fast.mputils.GffFileParser', return_value=reader): + create_annotation(hits, list(hits), 'unused.gff', directory, taxons, + {'orf1': True}, {}, lca, sample_name='sample') + row = (Path(directory)/'sample.functional_and_taxonomic_table.txt').read_text().strip().split('\t') + self.assertEqual(row[-3:], ['Species A','swissprot','Bacteria']) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_atomic_json.py b/tests/test_atomic_json.py new file mode 100644 index 0000000..0922c0e --- /dev/null +++ b/tests/test_atomic_json.py @@ -0,0 +1,46 @@ +"""Exercise overlapping publishers without relying on shared filesystem locks.""" +from concurrent.futures import ThreadPoolExecutor +import json +from pathlib import Path +import tempfile +import threading +import unittest +from unittest.mock import patch + +from metapathways.nf_worker import atomic_json + + +class AtomicJsonTests(unittest.TestCase): + def test_overlapping_writers_publish_complete_independent_records(self): + with tempfile.TemporaryDirectory() as directory: + target = Path(directory) / 'cache.json' + barrier = threading.Barrier(8) + replace = Path.replace + staged = [] + + def simultaneous_replace(source, destination): + staged.append(source) + barrier.wait(timeout=10) + return replace(source, destination) + + records = [{'writer': i, 'payload': str(i) * 10000} for i in range(8)] + with patch.object(Path, 'replace', simultaneous_replace): + with ThreadPoolExecutor(max_workers=8) as pool: + list(pool.map(lambda record: atomic_json(target, record), records)) + self.assertEqual(len(set(staged)), 8) + self.assertIn(json.loads(target.read_text()), records) + self.assertEqual(list(Path(directory).iterdir()), [target]) + + def test_failed_publication_preserves_previous_record_and_cleans_temp(self): + with tempfile.TemporaryDirectory() as directory: + target = Path(directory) / 'cache.json' + atomic_json(target, {'old': True}) + with patch.object(Path, 'replace', side_effect=OSError('publication failed')): + with self.assertRaisesRegex(OSError, 'publication failed'): + atomic_json(target, {'new': True}) + self.assertEqual(json.loads(target.read_text()), {'old': True}) + self.assertEqual(list(Path(directory).iterdir()), [target]) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_blast_hit_counts.py b/tests/test_blast_hit_counts.py new file mode 100644 index 0000000..5706938 --- /dev/null +++ b/tests/test_blast_hit_counts.py @@ -0,0 +1,24 @@ +import tempfile +import unittest +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import MagicMock, patch +from metapathways import MetaPathways_parse_blast as module + +class BlastHitCountsTests(unittest.TestCase): + def test_unique_counts_preserve_current_legacy_and_custom_identifiers(self): + queries = ['C1-G1', 'C1-G1', 'C2-G1', 'sampleA_1_2', 'sampleB_1_2', 'custom-gene'] + fields = ['target', 'q_length', 'bitscore', 'bsr', 'expect', 'aln_length', 'identity', 'ec', 'product'] + records = [dict(query=q, **{f: 'value' for f in fields}) for q in queries] + parser = MagicMock() + parser.__iter__.return_value = iter(records) + with tempfile.TemporaryDirectory() as d: + output = Path(d) / 'parsed.tsv' + opts = SimpleNamespace(ec_maps=None, taxonomy=False, parsed_output=str(output)) + with patch.object(module, 'BlastOutputParser', return_value=parser): + hits, unique = module.process_blastoutput('test', 'unused', 'unused', 'unused', opts) + self.assertEqual((hits, unique), (6, 5)) + self.assertEqual([line.split('\t')[0] for line in output.read_text().splitlines()[1:]], queries) + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_compact_results.py b/tests/test_compact_results.py new file mode 100644 index 0000000..dace9ec --- /dev/null +++ b/tests/test_compact_results.py @@ -0,0 +1,89 @@ +"""Destructive cleanup is tested only on temporary fixtures.""" +import json +import sqlite3 +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch +from metapathways.compact_results import compact, marker_state, MARKER +from metapathways.reporting import build_report +from metapathways.analysis_workflow import compact_task, parser +import test_reporting + + +class CompactTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + self.root = Path(self.tmp.name) + fixture = test_reporting.ReportTests() + fixture.root = self.root + fixture.fixture('alpha') + self.base = self.root/'alpha' + for name in ('bwa/reads.sorted.bam', 'preprocessed/alpha.fasta', + 'orf_prediction/alpha.faa', 'blast_results/raw.FASTout', + 'results/pgdb/community/alphacyc.tar.bz2', 'ptools/0.fasta'): + path = self.base/name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text('large intermediate') + + def tables(self): + build_report(self.root) + with sqlite3.connect(self.root/'reports/results.sqlite') as db: + self.assertEqual(db.execute('pragma integrity_check').fetchone()[0], 'ok') + self.assertFalse(db.execute('pragma foreign_key_check').fetchall()) + return {t:db.execute(f'select * from {t} order by 1,2').fetchall() for t in ( + 'contigs','orfs','annotations','annotation_taxonomy','entities', + 'contig_mags','entity_orfs','orf_groups','pathways','pathway_orfs', + 'abundance','abundance_explorer')} + + def test_report_rebuild_equivalence_and_external_symlink_safety(self): + outside = self.root/'external'; outside.mkdir(); (outside/'important').write_text('keep') + (self.base/'external_link').symlink_to(outside, target_is_directory=True) + before = self.tables() + compact(self.base, 'key') + self.assertEqual(before, self.tables()) + self.assertEqual((outside/'important').read_text(), 'keep') + self.assertFalse((self.base/'external_link').exists()) + self.assertFalse((self.base/'bwa').exists()) + self.assertTrue((self.base/'results/pgdb/community/alphacyc.tar.bz2').exists()) + self.assertTrue((self.base/'logs/ptools/run1/summary.json').is_file()) + self.assertEqual(marker_state(self.base, 'key'), 'complete') + compact(self.base, 'key') + with self.assertRaisesRegex(ValueError, 'new output'): + marker_state(self.base, 'different') + with self.assertRaisesRegex(ValueError, 'new output'): + marker_state(self.base, 'key', force=True) + + def test_validation_failure_does_not_delete_data(self): + (self.base/'preprocessed/test.mapping.txt').unlink() + with self.assertRaisesRegex(ValueError, 'Missing required'): + compact(self.base, 'key') + self.assertTrue((self.base/'bwa/reads.sorted.bam').is_file()) + self.assertFalse((self.base/MARKER).exists()) + + def test_interrupted_deletion_can_resume(self): + original = Path.unlink + def interrupt(path, *a, **kw): + if path.name == 'reads.sorted.bam': + raise OSError('simulated interruption') + return original(path, *a, **kw) + with patch.object(Path, 'unlink', interrupt): + with self.assertRaisesRegex(OSError, 'simulated'): + compact(self.base, 'key') + self.assertEqual(marker_state(self.base, 'key'), 'compacting') + compact(self.base, 'key') + self.assertEqual(marker_state(self.base, 'key'), 'complete') + self.tables() + + def test_cleanup_is_opt_in_and_waits_for_all_dependencies(self): + args = parser().parse_args(['analysis_wf']) + self.assertFalse(args.compact_results) + t = compact_task('alpha', self.root, 'key', ['alpha:tpm','alpha:pgdb:community','alpha:pgdb:MAG1'], '4 GB') + self.assertEqual(t['dependencies'], ['alpha:tpm','alpha:pgdb:community','alpha:pgdb:MAG1']) + self.assertFalse(t.get('allow_failure', False)) + self.assertEqual(t['cpus'], 1) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_compact_storage.py b/tests/test_compact_storage.py new file mode 100644 index 0000000..2c425bf --- /dev/null +++ b/tests/test_compact_storage.py @@ -0,0 +1,108 @@ +import json +import os +from pathlib import Path +import subprocess +import sys +import tarfile +import tempfile +from types import SimpleNamespace +import unittest +from unittest.mock import patch + +from metapathways.compact_storage import compact_pgdb, publish_file, scratch_root +from metapathways.compact_results import resume_key +from metapathways import nextflow as nf + + +class StorageTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + self.root = Path(self.tmp.name) + self.scratch = self.root/'scratch' + self.scratch.mkdir() + self.output = self.root/'shared' + self.env = patch.dict(os.environ, METAPATHWAYS_COMPACT_SCRATCH=str(self.scratch)) + self.env.start() + self.addCleanup(self.env.stop) + + def build(self, directory): + directory = Path(directory) + self.assertTrue(directory.is_relative_to(self.scratch)) + for suffix in ('cyc.tar.bz2', '_pwy.tsv', '_pwy2orf.tsv'): + (directory/('test'+suffix)).write_text('result') + for folder in ('diagnostics', '1.0/data'): + (directory/folder).mkdir(parents=True) + for i in range(100): + (directory/folder/f'{i}.txt').write_text('evidence') + + def test_success_publishes_tables_archive_diagnostics_without_extracted_tree(self): + compact_pgdb(self.output, 'test', self.build) + self.assertEqual(len(list(self.output.iterdir())), 4) + self.assertFalse(list(self.scratch.iterdir())) + with tarfile.open(self.output/'diagnostics.tar.gz') as archive: + self.assertEqual(archive.extractfile('diagnostics/0.txt').read(), b'evidence') + + def test_failed_build_archives_recovery_and_does_not_publish_partial_tables(self): + def fail(directory): + self.build(directory) + raise RuntimeError('inference failed') + with self.assertRaisesRegex(RuntimeError, 'inference failed'): + compact_pgdb(self.output, 'test', fail) + files = list(self.output.iterdir()) + self.assertEqual(len(files), 1) + with tarfile.open(files[0]) as archive: + self.assertTrue(any(n.endswith('/1.0/data/0.txt') for n in archive.getnames())) + self.assertFalse(list(self.scratch.iterdir())) + + def test_failed_copy_never_overwrites_valid_destination(self): + source = self.root/'source'; source.write_text('new') + dest = self.root/'dest'; dest.write_text('old') + def fail(src, target): + Path(target).write_text('partial') + raise OSError('quota') + with patch('metapathways.compact_storage.shutil.copyfile', side_effect=fail): + with self.assertRaises(OSError): + publish_file(source, dest) + self.assertEqual(dest.read_text(), 'old') + self.assertFalse(list(self.root.glob('.*.tmp'))) + + def test_slurm_requires_known_scratch_and_accepts_override(self): + with patch.dict(os.environ, {'SLURM_JOB_ID':'1'}, clear=True): + with self.assertRaisesRegex(RuntimeError, '--scratch_dir'): + scratch_root() + self.assertEqual(scratch_root(str(self.scratch)), self.scratch) + with patch.dict(os.environ, {'SLURM_JOB_ID':'1', 'SLURM_TMPDIR':str(self.scratch)}, clear=True): + self.assertEqual(scratch_root(), self.scratch) + + def test_resource_changes_preserve_compact_sample_key(self): + assembly = self.root/'in.fasta'; assembly.write_text('>a\nACGT\n') + row = dict(assembly=str(assembly), reads_1='', reads_2='', mag_map='') + args = SimpleNamespace(skip_ptools=True, image=None, refdb_dir=str(self.root/'db'), + threads=8, max_tasks=100, max_cpus=None, max_memory=None, + partition='a', scratch_dir=None) + key = resume_key(args, row) + args.max_tasks = 1; args.max_cpus = 16; args.max_memory = '64 GB' + args.partition = 'b'; args.scratch_dir = '/local' + self.assertEqual(key, resume_key(args, row)) + args.threads = 4 + self.assertNotEqual(key, resume_key(args, row)) + + def test_worker_scratch_is_cleaned_and_checkpoint_survives(self): + product = self.root/'result.txt' + task = nf.task('sample:stage', 'stage', [f'printf result > {product}'], outputs=[str(product)]) + task.update(receipt=str(self.root/'checkpoint.json'), compact_results=True, + scratch_dir=str(self.scratch), invocation_receipt=str(self.root/'invocation.json')) + manifest = self.root/'tasks.json'; manifest.write_text(json.dumps([task])) + env = dict(os.environ, PYTHONPATH=str(Path(nf.__file__).resolve().parent.parent)) + for expected in ('SUCCESS', 'ALREADY_COMPUTED'): + result = subprocess.run([sys.executable, '-m', 'metapathways.nf_worker', '--execute', + str(manifest), task['id']], cwd=self.root, env=env, + capture_output=True, text=True) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + self.assertEqual(json.loads((self.root/'invocation.json').read_text())['status'], expected) + self.assertFalse(list(self.scratch.iterdir())) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_fast_isolation.py b/tests/test_fast_isolation.py new file mode 100644 index 0000000..b3fc358 --- /dev/null +++ b/tests/test_fast_isolation.py @@ -0,0 +1,81 @@ +import concurrent.futures +from pathlib import Path +import shlex +import subprocess +import tempfile +import threading +from types import SimpleNamespace +import unittest +from unittest.mock import patch + +from metapathways import MetaPathways_func_search as search + + +class FastIsolationTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix='mp fast ') + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + + def options(self, name): + db = self.root/name + Path(str(db)+'.prj').write_text('numofsequences=1\n') + return SimpleNamespace(last_db=str(db), last_o=str(self.root/(name+'.out')), + last_query=str(self.root/'query.faa'), last_executable='fastal', last_f='2', + num_threads='2', num_hits='10', run_mode='pervol') + + def test_simultaneous_searches_cannot_share_sort_files(self): + barrier = threading.Barrier(2) + roots = [] + def fake(command): + args = shlex.split(command) + root = Path(args[args.index('-X')+1]) + roots.append(root) + # Model FAST processes selecting exactly the same internal filename. + collision = root/'same-second-LASTtemp0' + collision.write_text(args[-2]) + barrier.wait(timeout=5) + Path(args[args.index('-o')+1]).write_text(collision.read_text()) + return 0, '' + options = [self.options('metacyc'), self.options('swissprot')] + with patch.object(search.sysutils, 'getstatusoutput', side_effect=fake), concurrent.futures.ThreadPoolExecutor(2) as pool: + results = list(pool.map(search._execute_FAST, options)) + self.assertTrue(all(r[0] == 0 for r in results)) + self.assertEqual(len(set(roots)), 2) + self.assertTrue(all(not p.exists() for p in roots)) + for opt in options: + self.assertEqual(Path(opt.last_o).read_text(), opt.last_db) + + def test_failed_first_volume_stops_without_publishing_partial_results(self): + opt = self.options('multi') + Path(opt.last_db+'.prj').write_text('volumes=2\n') + Path(opt.last_o).write_text('previous output') + with patch.object(search.sysutils, 'getstatusoutput', return_value=(1, 'failure')) as run: + self.assertEqual(search._execute_FAST(opt), (1, 'failure')) + self.assertEqual(run.call_count, 1) + self.assertEqual(Path(opt.last_o).read_text(), 'previous output') + self.assertFalse(list(self.root.glob('.fast-*'))) + + def test_bundled_fast_parallel_searches_return_only_their_own_targets(self): + binary = Path(search.__file__).parent/'bin' + sequence = 'MKWVTFISLLFLFSSAYSRGVFRRDTHKSEIAHRFKDLGE' + query = self.root/'query.faa' + query.write_text('>query\n'+sequence+'\n') + options = [] + for name in ('meta_fixture', 'swiss_fixture'): + opt = self.options(name) + fasta = self.root/(name+'.faa') + fasta.write_text('>'+name+'\n'+sequence+'\n') + subprocess.run([str(binary/'fastdb'), '-p', opt.last_db, str(fasta)], check=True, capture_output=True) + opt.last_executable = str(binary/'fastal') + options.append(opt) + with concurrent.futures.ThreadPoolExecutor(2) as pool: + results = list(pool.map(search._execute_FAST, options)) + self.assertTrue(all(r[0] == 0 for r in results), results) + for opt in options: + targets = {line.split('\t')[1] for line in Path(opt.last_o).read_text().splitlines()} + self.assertEqual(targets, {Path(opt.last_db).name}) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_metacyc_db.py b/tests/test_metacyc_db.py new file mode 100644 index 0000000..830f41f --- /dev/null +++ b/tests/test_metacyc_db.py @@ -0,0 +1,140 @@ +"""Licensed-reference preparation using synthetic MetaCyc records.""" +import csv +import json +from pathlib import Path +import subprocess +import tempfile +import unittest +from unittest.mock import patch + +from metapathways import metacyc_db as mc +from metapathways.nf_databases import plan + + +class MetaCycTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.source = self.root/'29.5/data' + self.source.mkdir(parents=True) + (self.source/'protseq.fsa').write_text('>gnl|META|M1 protein\nMKWVTFISLLFLFSSAYSRG\n') + self.dat('proteins', [('M1', [('COMPONENT-OF', 'C1'), ('COMPONENT-OF', 'C2')]), + ('C1', [('CATALYZES', 'E1')]), ('C2', [('CATALYZES', 'E2')])]) + self.dat('enzrxns', [('E1', [('REACTION', 'R1')]), ('E2', [('REACTION', 'R2')])]) + self.dat('reactions', [('R1', []), ('R2', [])]) + self.dat('compounds', [('A', []), ('B', [])]) + self.dat('classes', [('Pathways', [('TYPES', 'Generalized-Reactions')]), + ('Generalized-Reactions', [('TYPES', 'FRAMES')]), + ('Class1', [('TYPES', 'Pathways')])]) + self.dat('pathways', [(p, [('TYPES', 'Class1'), ('REACTION-LAYOUT', + '(R1 (:LEFT-PRIMARIES A) (:DIRECTION :L2R) (:RIGHT-PRIMARIES B))')]) for p in ['P1', 'P2']]) + + def dat(self, name, records): + (self.source/(name+'.dat')).write_text(''.join( + 'UNIQUE-ID - '+identifier+'\nCOMMON-NAME - '+identifier+' name\n' + + ''.join(k+' - '+v+'\n' for k, v in values) + '//\n' for identifier, values in records)) + + def test_existing_scripts_keep_reactions_from_all_complexes(self): + out = self.root/'tables' + mc.make_tables(self.source, out) + with (out/mc.TABLES[0]).open() as stream: + pairs = {(r['MC'], r['RXN']) for r in csv.DictReader(stream, delimiter='\t')} + self.assertEqual(pairs, {('M1', 'R1'), ('M1', 'R2')}) + with (out/mc.TABLES[2]).open() as stream: + rows = list(csv.DictReader(stream, delimiter='\t')) + self.assertEqual(len(rows), 2) + self.assertEqual(rows[0]['MetaCyc_Ontology_IDs'], 'Class1|P1') + self.assertEqual(rows[1]['MetaCyc_Ontology_IDs'], 'Class1|P2') + + def test_incomplete_fasta_only_source_is_rejected(self): + (self.source/'proteins.dat').unlink() + with self.assertRaisesRegex(ValueError, 'protseq.fsa alone is insufficient'): + mc.source_path(self.source/'protseq.fsa') + + def test_screen_concurrency_and_aggregate_reservations(self): + import argparse + from metapathways.nf_databases import screen_task + args = argparse.Namespace(executor='local', max_tasks=32, max_cpus=None, max_memory=None) + with patch('metapathways.nextflow.local_capacity', return_value=(32, '256 GB')): + task = screen_task(self.root/'db', self.root/'pt.sif', '8 GB', resources=args) + self.assertIn('--max_tasks 32', task['commands'][0]) + self.assertEqual(task['cpus'], 32) + from metapathways.nextflow import memory_bytes + self.assertEqual(memory_bytes(task['memory']), memory_bytes('256 GB')) + args.max_cpus, args.max_memory = 12, '32 GB' + with patch('metapathways.nextflow.local_capacity', return_value=(32, '256 GB')): + task = screen_task(self.root/'db', self.root/'pt.sif', '8 GB', resources=args) + self.assertEqual(task['cpus'], 4) + self.assertIn('--max_tasks 4', task['commands'][0]) + self.assertEqual(memory_bytes(task['memory']), memory_bytes('32 GB')) + + def test_slurm_screen_reserves_selected_concurrency(self): + import argparse + from metapathways.nf_databases import screen_task + args = argparse.Namespace(executor='slurm', max_tasks=8, max_cpus=None, max_memory=None) + task = screen_task(self.root/'db', self.root/'pt.sif', '4 GB', resources=args) + self.assertEqual(task['cpus'], 8) + from metapathways.nextflow import memory_bytes + self.assertEqual(memory_bytes(task['memory']), memory_bytes('32 GB')) + + @patch('metapathways.nextflow.local_capacity', return_value=(4, '32 GB')) + def test_metacyc_screen_default_and_opt_out(self, _): + image = self.root/'pt.sif' + image.write_bytes(b'licensed-image-fixture') + tasks = plan(self.root/'db', ['metacyc'], 'fast', metacyc_source=self.source, screen_image=image) + self.assertEqual([t['id'] for t in tasks], ['directories', 'prepare_metacyc', 'screen_metacyc']) + self.assertIn('--publish', tasks[-1]['commands'][0]) + tasks = plan(self.root/'db', ['metacyc'], 'fast', metacyc_source=self.source, skip_pt_screen=True) + self.assertEqual(len(tasks), 2) + + def test_screen_rejects_insufficient_memory(self): + from metapathways.nf_databases import screen_resources + with patch('metapathways.nextflow.local_capacity', return_value=(2, '7 GB')): + with self.assertRaisesRegex(ValueError, 'available screening budget'): + screen_resources(None, '16 GB') + + def test_metacyc_only_plan_does_not_download_unrelated_references(self): + tasks = plan(self.root/'db', ['metacyc'], 'fast', metacyc_source=self.source, skip_pt_screen=True) + self.assertEqual([t['id'] for t in tasks], ['directories', 'prepare_metacyc']) + self.assertFalse(tasks[-1]['adopt_existing']) + self.assertTrue(all(any(n in o for o in tasks[-1]['outputs']) for n in mc.TABLES)) + with patch('metapathways.pt_container.registered_image', return_value=None): + with self.assertRaisesRegex(ValueError, 'licensed source'): + plan(self.root, ['metacyc'], 'fast') + + def test_failed_index_does_not_replace_existing_reference(self): + db = self.root/'db' + (db/'functional').mkdir(parents=True) + old = db/'functional/metacyc' + old.write_text('previous reference') + real_run = subprocess.run + def run(command, **kwargs): + if command[0] == 'fastdb': + raise subprocess.CalledProcessError(1, command) + return real_run(command, **kwargs) + with patch.object(mc.subprocess, 'run', side_effect=run): + with self.assertRaises(subprocess.CalledProcessError): + mc.prepare(self.source, db, 'fast') + self.assertEqual(old.read_text(), 'previous reference') + self.assertFalse(list(db.glob('.metacyc-build-*'))) + self.assertFalse((self.source/mc.TABLES[0]).exists()) + + def test_success_publishes_matching_tables_and_provenance(self): + db = self.root/'db' + real_run = subprocess.run + def run(command, **kwargs): + if command[0] == 'fastdb': + Path(command[2]+'.prj').write_text('fixture index') + return subprocess.CompletedProcess(command, 0) + return real_run(command, **kwargs) + with patch.object(mc.subprocess, 'run', side_effect=run): + mc.prepare(self.source, db, 'fast') + record = json.loads((db/'functional_categories/MetaCyc_provenance.json').read_text()) + self.assertEqual(record['release'], '29.5') + self.assertEqual(record['source_sha256']['protseq.fsa'], mc.digest(db/'functional/metacyc')) + self.assertFalse((self.source/mc.TABLES[0]).exists()) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_nextflow.py b/tests/test_nextflow.py new file mode 100644 index 0000000..10e97b7 --- /dev/null +++ b/tests/test_nextflow.py @@ -0,0 +1,355 @@ +import argparse +import contextlib +import io +import json +import os +from pathlib import Path +import subprocess +import tempfile +import tarfile +from types import SimpleNamespace +import unittest +from unittest.mock import patch + +from metapathways import nextflow as nf + + +class SchedulingTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.task = nf.task('test', 'test', ['echo hello'], outputs=[str(self.root / 'result')], cpus=8) + + def args(self, **kw): + p = argparse.ArgumentParser() + nf.add_resources(p) + a = p.parse_args([]) + for k, v in kw.items(): + setattr(a, k, v) + return a + + def test_local_uses_available_cpus(self): + with patch.object(nf, 'local_capacity', return_value=(48, '128 GB')): + text, limits = nf.configuration([self.task], self.args(), self.root) + self.assertEqual(limits['max_cpus'], 48) + self.assertEqual(limits['max_tasks'], 48) + self.assertIn('executor.cpus = 48', text) + + def test_slurm_bounds_submissions_by_cpu_and_memory(self): + args = self.args(executor='slurm', account='lab', partition='compute') + text, limits = nf.configuration([self.task], args, self.root) + self.assertEqual(limits['max_tasks'], 4) + self.assertIn('--account=lab', text) + self.assertIn('--nodes=1', text) + self.assertIn("executor.submitRateLimit = '6/1min'", text) + args.max_cpus = 16 + self.assertEqual(nf.configuration([self.task], args, self.root)[1]['max_tasks'], 2) + args.max_memory = '16 GB' + self.assertEqual(nf.configuration([self.task], args, self.root)[1]['max_tasks'], 1) + + def test_local_job_ceiling_uses_detected_capacity_without_user_totals(self): + args = self.args(max_tasks=100) + with patch.object(nf, 'local_capacity', return_value=(32, '128 GB')): + config, limits = nf.configuration([self.task], args, self.root) + self.assertEqual(limits['max_tasks'], 100) + self.assertEqual(limits['max_cpus'], 32) + self.assertEqual(limits['max_memory'], '128 GB') + self.assertIn('executor.queueSize = 100', config) + self.assertIn('executor.cpus = 32', config) + self.assertIn("executor.memory = '128 GB'", config) + + def test_slurm_job_limit_needs_no_implicit_aggregate_budgets(self): + tasks = [nf.task('threaded', 'threaded', [], cpus=8, memory='64 GB'), + nf.task('serial', 'serial', [], cpus=1, memory='64 GB')] + args = self.args(executor='slurm', account='lab', max_tasks=100) + with patch.object(nf, 'local_capacity', side_effect=AssertionError('Slurm must not use headnode capacity')): + config, limits = nf.configuration(tasks, args, self.root) + self.assertEqual(limits['max_tasks'], 100) + self.assertIsNone(limits['max_cpus']) + self.assertIsNone(limits['max_memory']) + self.assertIn('executor.queueSize = 100', config) + self.assertNotIn('process.queue =', config) + self.assertIn('--account=lab', config) + self.assertIn('--nodes=1', config) + args.max_memory = '128 GB' + self.assertEqual(nf.configuration(tasks, args, self.root)[1]['max_tasks'], 2) + args.max_memory, args.max_cpus = None, 24 + self.assertEqual(nf.configuration(tasks, args, self.root)[1]['max_tasks'], 3) + args.max_cpus, args.max_tasks = None, None + self.assertEqual(nf.configuration(tasks, args, self.root)[1]['max_tasks'], 4) + + def test_invalid_slurm_options_and_oversized_tasks_rejected(self): + with self.assertRaisesRegex(ValueError, 'requires --account'): + nf.configuration([self.task], self.args(executor='slurm'), self.root) + with self.assertRaises(argparse.ArgumentTypeError): + nf.configuration([self.task], self.args(executor='slurm', account='lab; echo unsafe', partition='compute'), self.root) + with self.assertRaisesRegex(ValueError, 'more memory'): + nf.configuration([self.task], self.args(max_memory='1 GB', max_cpus=8), self.root) + with self.assertRaisesRegex(ValueError, 'threads'): + nf.configuration([self.task], self.args(max_cpus=1), self.root) + + def test_dependency_order_and_cycle_check(self): + a = nf.task('a','a',[]) + b = nf.task('b','b',[],dependencies=['a']) + self.assertEqual([t['id'] for t in nf.ordered([b,a])], ['a','b']) + a['dependencies'] = ['b'] + with self.assertRaisesRegex(ValueError, 'Cyclic'): + nf.ordered([a,b]) + + def test_large_workflow_is_split_into_bounded_modules(self): + tasks = [nf.task(str(i), f'task {i}', ['true']) for i in range(6470)] + files = nf.render_modules(tasks, self.root / 'tasks.json') + modules = [text for name, text in files.items() if name != 'main.nf'] + self.assertEqual(len(modules), 102) + self.assertEqual(sum(text.count('process TASK_') for text in modules), 6470) + self.assertTrue(all(text.count('process TASK_') <= 64 for text in modules)) + self.assertNotIn('process TASK_', files['main.nf']) + + def test_module_boundaries_preserve_individual_dependencies(self): + tasks = [nf.task('a', 'a', []), nf.task('b', 'b', []), + nf.task('c', 'c', [], dependencies=['a']), + nf.task('d', 'd', [], dependencies=['c', 'b'])] + files = nf.render_modules(tasks, self.root / 'tasks.json', batch_size=2) + self.assertIn('BATCH_0001(BATCH_0000.out.done_TASK_0000, BATCH_0000.out.done_TASK_0001)', files['main.nf']) + module = files['modules/BATCH_0001.nf'] + self.assertIn('TASK_0002(upstream_0)', module) + self.assertIn('TASK_0003(TASK_0002.out, upstream_1)', module) + self.assertNotIn('collect', module) + tasks[2]['dependencies'] = ['missing'] + with self.assertRaisesRegex(ValueError, 'Unknown dependency'): + nf.render_modules(tasks, self.root / 'tasks.json') + + def test_compact_sample_fan_in_is_wired_inside_sample_modules(self): + tasks = [] + for sample in range(49): + ids = [f'{sample}:pgdb:{i}' for i in range(132)] + tasks.extend(nf.task(i, i, ['true'], sample=str(sample)) for i in ids) + tasks.append(nf.task(f'{sample}:compact', 'compact', ['true'], + sample=str(sample), dependencies=ids)) + files = nf.render_modules(tasks, self.root/'tasks.json') + self.assertLess(len(files['main.nf']), 10000) + self.assertNotIn('.out.', files['main.nf']) + self.assertEqual(files['main.nf'].count('include { SAMPLE_'), 49) + for i in range(49): + module = files[f'samples/SAMPLE_{i:04d}/main.nf'] + self.assertIn(f'workflow SAMPLE_{i:04d}', module) + self.assertIn('BATCH_0000.out.done_TASK_0000', module) + self.assertIn('BATCH_0002([upstream_0:', module) + cleanup = files[f'samples/SAMPLE_{i:04d}/modules/BATCH_0002.nf'] + self.assertIn('Channel.empty().mix(', cleanup) + self.assertIn('.collect()', cleanup) + self.assertNotIn('val dependency_1', cleanup) + self.assertIn(' take:\n upstream\n', cleanup) + # Never hide a cross-sample dependency behind independent workflows. + tasks[-1]['dependencies'].append('0:pgdb:0') + files = nf.render_modules(tasks, self.root/'tasks.json') + self.assertNotIn('samples/SAMPLE_0000/main.nf', files) + + def test_database_restart_restores_directories_without_invalidating_downloads(self): + from metapathways.nf_databases import plan + tasks = plan(self.root, ['swissprot', 'cazy'], 'fast') + nf.ordered(tasks) + directories = tasks[0] + (self.root / '.metapathways').mkdir() + subprocess.run(['bash', '-ec', directories['commands'][0]], check=True) + ready = Path(directories['outputs'][0]) + stamp = ready.stat().st_mtime_ns + (self.root / 'functional/formatted').rmdir() + subprocess.run(['bash', '-ec', directories['commands'][0]], check=True) + self.assertTrue((self.root / 'functional/formatted').is_dir()) + self.assertEqual(ready.stat().st_mtime_ns, stamp) + self.assertEqual(directories['status'], 'redo') + self.assertTrue({'release_silva', 'release_cazy'} <= {t['id'] for t in tasks}) + + def test_legacy_tool_output_is_streamed_and_returned(self): + from metapathways.sysutil import getstatusoutput + output = io.StringIO() + with patch.dict(os.environ, METAPATHWAYS_STREAM_TOOLS='1'), contextlib.redirect_stdout(output): + status, captured = getstatusoutput("printf 'stdout\\n'; printf 'stderr\\n' >&2; exit 3") + self.assertNotEqual(status, 0) + self.assertEqual(captured, 'stdout\nstderr') + self.assertEqual(output.getvalue(), 'stdout\nstderr\n') + + def test_annotation_searches_share_a_barrier(self): + from metapathways.context import Context + from metapathways import jobscreator + contexts = [] + for name in ('FILTER_AMINOS', 'FUNC_SEARCH:swissprot', 'FUNC_SEARCH:cazy', + 'PARSE_FUNC_SEARCH:swissprot', 'PARSE_FUNC_SEARCH:cazy', 'ANNOTATE_ORFS'): + c = Context() + c.name, c.status = name, 'yes' + c.outputs = {'result': str(self.root/name)} + contexts.append(c) + sample = SimpleNamespace(sample_name='sample', output_dir=str(self.root), + getContextBlocks=lambda: [contexts], writeParamsToRunLogs=lambda _: None) + with patch.object(jobscreator, 'JobCreator'): + tasks = nf.annotation_tasks({'sample': sample}, {}, + {'NUM_CPUS': 8, 'REFDBS': str(self.root)}, '16 GB') + self.assertEqual([t['cpus'] for t in tasks], [1, 8, 8, 1, 1, 1]) + self.assertEqual(tasks[1]['dependencies'], [tasks[0]['id']]) + self.assertEqual(tasks[2]['dependencies'], tasks[1]['dependencies']) + self.assertEqual(tasks[3]['dependencies'], [tasks[1]['id'], tasks[2]['id']]) + self.assertEqual(tasks[4]['dependencies'], tasks[3]['dependencies']) + self.assertEqual(tasks[5]['dependencies'], [tasks[3]['id'], tasks[4]['id']]) + + def test_all_thread_capable_stages_reserve_the_tool_budget(self): + from metapathways.context import Context + from metapathways import jobscreator + names = ('ORF_PREDICTION', 'SCAN_rRNA:barrnap', 'SCAN_tRNA', + 'FUNC_SEARCH:swissprot', 'COMPUTE_TPM', 'ANNOTATE_ORFS') + for threads in (1, 3, 8): + contexts = [] + for name in names: + c = Context() + c.name, c.status = name, 'yes' + c.inputs = {'fwd_fq': 'reads.fq'} + contexts.append(c) + sample = SimpleNamespace(sample_name='sample', output_dir=str(self.root), + getContextBlocks=lambda: [contexts], writeParamsToRunLogs=lambda _: None) + with self.subTest(threads=threads), patch.object(jobscreator, 'JobCreator'): + tasks = nf.annotation_tasks({'sample': sample}, {}, + {'NUM_CPUS': threads, 'REFDBS': str(self.root)}, '16 GB') + self.assertEqual([t['cpus'] for t in tasks], [threads]*5 + [1]) + config, limits = nf.configuration(tasks, self.args(max_cpus=16, max_memory='128 GB'), self.root) + self.assertEqual(limits['max_tasks'], 16) + self.assertIn('executor.cpus = 16', config) + + def fake_run(self, command, cwd, env, tasks, console): + work = Path(command[command.index('-work-dir')+1]) + (work / 'fixture').mkdir() + (work / 'fixture/.command.log').write_text('scheduler and tool output\n') + console.write('all terminal output\n') + self.assertTrue(env['CONDA_PKGS_DIRS'].startswith(str(self.root))) + for t in tasks: + Path(t['log']).parent.mkdir(parents=True, exist_ok=True) + Path(t['log']).write_text('tool output\n') + Path(t['invocation_receipt']).write_text(json.dumps(dict(status='SUCCESS'))) + + def test_tpm_resume_ignores_own_scratch_but_checks_inputs_and_results(self): + from metapathways.context import Context + from metapathways import jobscreator, nf_worker + for layout in ('paired', 'interleaved', 'single'): + with self.subTest(layout=layout): + root = self.root / layout + root.mkdir() + (root / 'run_statistics').mkdir() + scratch = root / 'bwa' + scratch.mkdir() + c = Context() + c.name, c.status = 'COMPUTE_TPM', 'yes' + c.inputs = {key: str(root / name) for key, name in + [('fwd_fq', 'R1.fq'), ('output_gff', 'genes.gff'), + ('output_fas', 'assembly.fa')]} + for value in c.inputs.values(): + Path(value).write_text('original input') + c.inputs['bwaFolder'] = str(scratch) + c.temps = {'rev_fq': None, 'inter': layout == 'interleaved'} + if layout == 'paired': + mate = root / 'R2.fq' + mate.write_text('original mate') + c.temps['rev_fq'] = str(mate) + c.outputs = {'stats_file': str(root / 'stats.txt')} + sample = SimpleNamespace(sample_name='sample', output_dir=str(root), + getContextBlocks=lambda: [[c]], writeParamsToRunLogs=lambda _: None) + with patch.object(jobscreator, 'JobCreator'): + t, = nf.annotation_tasks({'sample': sample}, {}, + {'NUM_CPUS': 4, 'REFDBS': str(root)}, '16 GB') + t['receipt'] = str(root / 'checkpoint.json') + + def run_stage(*_): + (scratch / 'sorted.bam').write_text('generated intermediate') + for value in t['outputs']: + p = Path(value) + p.parent.mkdir(parents=True, exist_ok=True) + p.write_text('result') + return (0, '') + + def execute(): + self.assertEqual(nf_worker.execute(t), 0) + return json.loads(Path('receipt.json').read_text())['status'] + + cwd = Path.cwd() + try: + os.chdir(root) + with patch('metapathways.execution.execute', side_effect=run_stage) as run, contextlib.redirect_stdout(io.StringIO()): + self.assertEqual(execute(), 'SUCCESS') + self.assertEqual(execute(), 'ALREADY_COMPUTED') + (scratch / 'sorted.bam').write_text('changed scratch') + (scratch / 'extra.txt').write_text('new scratch file') + self.assertEqual(execute(), 'ALREADY_COMPUTED') + self.assertEqual(run.call_count, 1) + for value in t['inputs']: + with Path(value).open('a') as handle: + handle.write('changed true input') + self.assertEqual(execute(), 'SUCCESS') + self.assertEqual(execute(), 'ALREADY_COMPUTED') + for value in t['outputs']: + Path(value).unlink() + self.assertEqual(execute(), 'SUCCESS') + finally: + os.chdir(cwd) + + @patch.object(nf.shutil, 'which', return_value='/bin/nextflow') + def test_default_cleanup_preserves_logs_and_results(self, _): + sentinel = self.root / 'precious.txt' + sentinel.write_text('keep') + with patch.object(nf, 'stream_run', side_effect=self.fake_run): + run = nf.launch([self.task], self.root, self.args(max_cpus=8, max_memory='64 GB'), 'test') + summary = json.loads((run/'summary.json').read_text()) + self.assertFalse(Path(summary['work_dir']).exists()) + self.assertFalse(Path(summary['conda_cache']).exists()) + self.assertEqual(sentinel.read_text(), 'keep') + self.assertEqual((run/'console.log').read_text(), 'all terminal output\n') + self.assertEqual(len(list((run/'nextflow_tasks').rglob('.command.log'))), 1) + + @patch.object(nf.shutil, 'which', return_value='/bin/nextflow') + def test_compact_exit_bundles_logs_on_success_and_failure(self, _): + for failure in (False, True): + with self.subTest(failure=failure): + root = self.root / str(failure) + def run(*args): + self.fake_run(*args) + if failure: + raise subprocess.CalledProcessError(1, ['nextflow']) + with patch.object(nf, 'stream_run', side_effect=run): + try: + nf.launch([self.task], root, + self.args(max_cpus=8, max_memory='64 GB', compact_results=True), 'test') + except subprocess.CalledProcessError: + self.assertTrue(failure) + log = next((root/'logs/test').iterdir()) + summary = json.loads((log/'summary.json').read_text()) + self.assertEqual(summary['tasks'][0]['status'], 'SUCCESS') + self.assertFalse((log/'tasks').exists()) + self.assertFalse(Path(summary['work_dir']).exists()) + with tarfile.open(log/'nextflow_tasks.tar.gz') as archive: + self.assertEqual(archive.extractfile('work/fixture/.command.log').read(), + b'scheduler and tool output\n') + with tarfile.open(log/'task_logs.tar.gz') as archive: + self.assertTrue(any(n.endswith('.json') for n in archive.getnames())) + + @patch.object(nf.shutil, 'which', return_value='/bin/nextflow') + def test_explicit_directories_preserved(self, _): + args = self.args(max_cpus=8, max_memory='64 GB', work_dir=str(self.root/'custom-work'), conda_cache=str(self.root/'custom-cache')) + with patch.object(nf, 'stream_run', side_effect=self.fake_run): + run = nf.launch([self.task], self.root, args, 'test') + summary = json.loads((run/'summary.json').read_text()) + self.assertTrue(Path(summary['work_dir']).exists()) + self.assertTrue(Path(summary['conda_cache']).exists()) + + @patch.object(nf.shutil, 'which', return_value='/bin/nextflow') + def test_failed_run_keeps_diagnostics_before_cleanup(self, _): + def fail(*args): + self.fake_run(*args) + raise subprocess.CalledProcessError(1, ['nextflow']) + with patch.object(nf, 'stream_run', side_effect=fail): + with self.assertRaises(subprocess.CalledProcessError): + nf.launch([self.task], self.root, self.args(max_cpus=8, max_memory='64 GB'), 'test') + run = next((self.root/'logs/test').iterdir()) + self.assertEqual(json.loads((run/'summary.json').read_text())['status'], 'FAILED') + self.assertEqual(len(list((run/'nextflow_tasks').rglob('.command.log'))), 1) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_pt_container.py b/tests/test_pt_container.py new file mode 100644 index 0000000..2c39936 --- /dev/null +++ b/tests/test_pt_container.py @@ -0,0 +1,387 @@ +"""Container lifecycle tests; licensed installation is a separate integration check.""" +import contextlib +import io +import json +import os +from pathlib import Path +import subprocess +import tempfile +import unittest +from unittest.mock import patch + +from metapathways import pt_container as pt + + +class ContainerTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix='mp pt test ') + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.installer = self.root / 'pathway-tools-29.5-install' + self.installer.write_bytes(b'fixture installer') + self.image = self.root / 'pt.sif' + patch_download = patch.object(pt, 'download_patches', return_value={'files': [], 'version': '29.5'}) + self.real_download_patches = pt.download_patches + self.download_patches = patch_download.start() + self.addCleanup(patch_download.stop) + + def fake_build(self, command, **kwargs): + self.assertIn('--fakeroot', command) + self.assertIn('-processors 2', command) + Path(command[-2]).write_bytes(b'fixture SIF') + return subprocess.CompletedProcess(command, 0) + + @patch.object(pt.shutil, 'which', return_value='/bin/apptainer') + def test_publish_after_validation(self, _): + with patch.object(pt.subprocess, 'run', side_effect=self.fake_build), patch.object(pt, 'validate', return_value='MP-PT-READY\n') as validate: + pt.build(self.installer, self.image, pt.digest(self.installer), 2) + validate.assert_called_once() + metadata = json.loads(Path(str(self.image) + '.json').read_text()) + self.assertEqual(metadata['image_sha256'], pt.digest(self.image)) + self.assertEqual(metadata['installer_sha256'], pt.digest(self.installer)) + self.assertEqual(metadata['official_patches']['version'], '29.5') + self.assertFalse(list(self.root.glob('.pt-build-*'))) + + @patch.object(pt.shutil, 'which', return_value='/bin/apptainer') + def test_failed_validation_preserves_previous_image(self, _): + self.image.write_bytes(b'previous valid image') + with patch.object(pt.subprocess, 'run', side_effect=self.fake_build), patch.object(pt, 'validate', side_effect=RuntimeError('startup failed')): + with self.assertRaisesRegex(RuntimeError, 'startup failed'): + pt.build(self.installer, self.image, pt.digest(self.installer), 2) + self.assertEqual(self.image.read_bytes(), b'previous valid image') + self.assertFalse(Path(str(self.image) + '.json').exists()) + self.assertFalse(list(self.root.glob('.pt-build-*'))) + + def test_changed_installer_rejected_before_execution(self): + with patch.object(pt.subprocess, 'run') as run: + with self.assertRaisesRegex(ValueError, 'Installer changed'): + pt.build(self.installer, self.image, 'wrong-checksum', 2) + run.assert_not_called() + + @patch.object(pt.shutil, 'which', return_value='/bin/apptainer') + def test_runtime_isolation(self, _): + cmd = pt.exec_command(self.image, self.root / 'private state', ['sh', '-c', 'echo hello']) + for flag in ['--containall', '--cleanenv']: + self.assertIn(flag, cmd) + self.assertIn(str(self.root / 'private state') + ':/data', cmd) + self.assertEqual(cmd[cmd.index('--home') + 1], str(self.root / 'private state') + ':/data') + self.assertEqual(cmd[cmd.index('--pwd') + 1], '/data') + scratch = Path(cmd[cmd.index('--workdir') + 1]) + self.assertEqual(scratch, self.root / 'private state' / 'container-work') + self.assertTrue(scratch.is_dir()) + other = pt.exec_command(self.image, self.root / 'other state', ['true']) + self.assertNotEqual(other[other.index('--workdir') + 1], str(scratch)) + self.assertEqual(cmd[-3:], ['sh', '-c', 'echo hello']) + + @patch.object(pt.shutil, 'which', return_value='/bin/apptainer') + def test_validation_requires_lisp_marker(self, _): + with patch.object(pt.subprocess, 'run', return_value=subprocess.CompletedProcess([], 0, 'no startup')): + with self.assertRaisesRegex(RuntimeError, 'validation marker'): + pt.validate(self.image, self.root / 'state') + + @patch.object(pt.shutil, 'which', return_value='/bin/apptainer') + def test_validation_preserves_startup_error(self, _): + with patch.object(pt.subprocess, 'run', return_value=subprocess.CompletedProcess([], 1, 'missing shared library')): + with self.assertRaisesRegex(RuntimeError, 'missing shared library'): + pt.validate(self.image, self.root / 'state') + + @patch.object(pt.shutil, 'which', return_value='/bin/apptainer') + def test_validation_marker_on_own_line_after_echo(self, _): + result = subprocess.CompletedProcess([], 0, 'CMD: (format t "~%MP-PT-READY~%")\nMP-PT-READY\n') + with patch.object(pt.subprocess, 'run', return_value=result) as run: + pt.validate(self.image, self.root / 'state') + self.assertIn('"~%MP-PT-READY~%"', run.call_args.args[0][-1]) + + def test_registry_and_override(self): + self.image.touch() + with patch.dict(os.environ, {'XDG_CONFIG_HOME': str(self.root)}, clear=True): + self.assertIsNone(pt.registered_image()) + pt.save_json(pt.registry_path(), {'image': str(self.image)}) + self.assertEqual(pt.registered_image(), str(self.image)) + with patch.dict(os.environ, {'METAPATHWAYS_PTOOLS_IMAGE': str(self.root / 'missing.sif')}): + with self.assertRaisesRegex(ValueError, 'image is missing'): + pt.registered_image() + + def test_dryrun_without_runtimes_never_registers(self): + with patch.dict(os.environ, {'XDG_CONFIG_HOME': str(self.root)}, clear=True), patch.object(pt.shutil, 'which', return_value=None), contextlib.redirect_stdout(io.StringIO()): + pt.main(['-i', str(self.installer), '-o', str(self.root / 'images'), '--dryrun']) + self.assertFalse(pt.registry_path().exists()) + plans = list(self.root.glob('images/logs/build_pt/*/tasks.json')) + self.assertEqual(len(plans), 1) + task = json.loads(plans[0].read_text())[0] + self.assertEqual(task['cpus'], 2) + self.assertFalse(task['adopt_existing']) + + def test_invalid_tag_rejected(self): + with self.assertRaisesRegex(ValueError, 'PGDB tag'): + pt.run_pgdb(self.image, self.root, self.root / 'out', "bad') (exit)") + + @patch.object(pt.shutil, 'which', return_value='/bin/apptainer') + def test_missing_sequence_source_stops_before_pathway_tools_and_retains_diagnostics(self, _): + inputs = self.root/'input' + inputs.mkdir() + (inputs/'0.pf').write_text('ID\tgene1\n//\n') + output = self.root/'failed-input' + with patch.object(pt.subprocess, 'run') as run: + with self.assertRaisesRegex(ValueError, 'sequence-backed Pathway Tools input'): + pt.run_pgdb(self.image, inputs, output, 'sample', sample_output=self.root/'sample') + run.assert_not_called() + record = json.loads(next(output.glob('diagnostics/*/execution.json')).read_text()) + self.assertEqual(record['stage'], 'input preparation') + self.assertEqual(record['status'], 'FAILED') + self.assertFalse(list(self.root.glob('.pt-run-*'))) + + @patch('metapathways.nextflow.local_capacity', return_value=(4, '32 GB')) + def test_build_with_mpdb_plans_export_after_image_without_running_it(self, _): + with patch.object(pt.nextflow, 'launch') as launch: + pt.main(['-i', str(self.installer), '-o', str(self.root/'images'), '-d', str(self.root/'MPDB'), '-a', 'blast', '--dryrun']) + tasks = launch.call_args.args[0] + self.assertEqual(len(tasks), 3) + self.assertEqual(tasks[2]['id'], 'screen_metacyc') + self.assertEqual(tasks[1]['dependencies'], [tasks[0]['id']]) + self.assertEqual(tasks[1]['inputs'], [tasks[0]['outputs'][0]]) + self.assertIn(str(self.root/'MPDB/functional/formatted/metacyc.pdb'), tasks[1]['outputs']) + self.download_patches.assert_not_called() + + def test_patch_snapshot_only_accepts_official_listing_filenames(self): + listing = b'patchbadbadarchive' + contents = [listing, b'official binary', b'official archive'] + with patch.object(pt, 'build_opener') as opener: + opener.return_value.open.side_effect = [io.BytesIO(x) for x in contents] + manifest = self.real_download_patches('29.5', self.root / 'patches') + self.assertEqual([x['name'] for x in manifest['files']], ['p123.fasl', 'p124.tar.gz']) + self.assertEqual(manifest['files'][0]['sha256'], pt.digest(self.root/'patches/files/p123.fasl')) + self.assertTrue(all(c.args[0].startswith(manifest['source']) for c in opener.return_value.open.call_args_list)) + + def test_failed_patch_download_does_not_build_or_publish(self): + self.download_patches.side_effect = RuntimeError('official feed unavailable') + with patch.object(pt.shutil, 'which', return_value='/bin/apptainer'), patch.object(pt.subprocess, 'run') as run: + with self.assertRaisesRegex(RuntimeError, 'official feed unavailable'): + pt.build(self.installer, self.image, pt.digest(self.installer), 2) + run.assert_not_called() + self.assertFalse(self.image.exists()) + + def test_empty_vendor_listing_and_redirect_are_rejected(self): + with patch.object(pt, 'build_opener') as opener: + opener.return_value.open.return_value = io.BytesIO(b'login required') + with self.assertRaisesRegex(RuntimeError, 'no recognized patch'): + self.real_download_patches('29.5', self.root / 'empty') + with self.assertRaisesRegex(ValueError, 'redirected'): + pt._NoRedirect().redirect_request(None, None, 302, '', {}, 'https://other/patch') + + def test_release_detection_is_explicit(self): + self.assertEqual(pt.installer_version(self.installer), '29.5') + self.assertEqual(pt.installer_version('renamed-installer', '29.5'), '29.5') + for name, version in [('renamed', None), (str(self.installer), '28.5'), ('renamed', '../29.5')]: + with self.assertRaises(ValueError): + pt.installer_version(name, version) + + @patch.object(pt.shutil, 'which', return_value='/bin/apptainer') + def test_validation_rejects_blast_warning_even_with_startup_marker(self, _): + result = subprocess.CompletedProcess([], 0, 'blastall or blastp could not be located\nMP-PT-READY\n') + with patch.object(pt.subprocess, 'run', return_value=result): + with self.assertRaisesRegex(RuntimeError, 'could not locate'): + pt.validate(self.image, self.root / 'state') + + @patch.object(pt.shutil, 'which', return_value='/bin/apptainer') + def test_pgdb_failure_preserves_private_error_log(self, _): + self.image.touch() + inputs = self.root / 'input' + inputs.mkdir() + (inputs / '0.pf').write_text('ID\tgene1\n') + output = self.root / 'failed' + def fail(command, *args, **kwargs): + state = Path(command[command.index('--home')+1].rsplit(':', 1)[0]) + (state / 'stage.txt').write_text('build\n') + (state / 'input/pathologic.log').write_text('Specific Lisp failure\n') + kb = state / 'ptools-local/pgdbs/user/samplecyc/1.0/kb' + kb.mkdir(parents=True) + (kb / 'samplebase.ocelot').write_text('saved checkpoint') + raise subprocess.CalledProcessError(255, command) + with patch.object(pt, '_run_pgdb_container', side_effect=fail), contextlib.redirect_stdout(io.StringIO()) as console: + with self.assertRaises(subprocess.CalledProcessError): + pt.run_pgdb(self.image, inputs, output, 'sample') + records = list(output.glob('diagnostics/*/execution.json')) + self.assertEqual(len(records), 1) + record = json.loads(records[0].read_text()) + self.assertEqual((record['status'], record['stage'], record['exit_code']), ('FAILED', 'build', 255)) + self.assertEqual((records[0].parent/'input/pathologic.log').read_text(), 'Specific Lisp failure\n') + self.assertIn('Specific Lisp failure', console.getvalue()) + self.assertEqual((Path(record['retained_pgdbs'])/'samplecyc/1.0/kb/samplebase.ocelot').read_text(), 'saved checkpoint') + self.assertTrue((records[0].parent/'input/0.pf').exists()) + self.assertFalse(list(self.root.glob('.pt-run-*'))) + self.assertFalse((inputs / 'pathologic.log').exists()) + + @patch.object(pt.shutil, 'which', return_value='/bin/apptainer') + def test_pgdb_success_preserves_log_and_publishes_archive(self, _): + self.image.touch() + inputs = self.root / 'input' + inputs.mkdir() + output = self.root / 'success' + def success(command, *args, **kwargs): + script = command[command.index('-ec') + 1] + self.assertIn('mkdir -p /data/blastdb /tmp/.X11-unix', script) + self.assertEqual(script.count('-nolisten local'), 2) + self.assertIn('-e /data/build-xvfb.log', script) + self.assertIn('-e /data/export-xvfb.log', script) + state = Path(command[command.index('--home')+1].rsplit(':', 1)[0]) + (state / 'build-xvfb.log').write_text('build display diagnostics\n') + (state / 'export-xvfb.log').write_text('export display diagnostics\n') + (state / 'stage.txt').write_text('archive\n') + (state / 'input/pathologic.log').write_text('Pathologic finished\n') + (state / 'output/samplecyc.tar.bz2').write_bytes(b'fixture archive') + with patch.object(pt, '_run_pgdb_container', side_effect=success): + pt.run_pgdb(self.image, inputs, output, 'sample') + record = next(output.glob('diagnostics/*/execution.json')) + self.assertEqual(json.loads(record.read_text())['status'], 'SUCCESS') + self.assertEqual((output/'samplecyc.tar.bz2').read_bytes(), b'fixture archive') + self.assertTrue((record.parent/'input/pathologic.log').is_file()) + self.assertEqual((record.parent/'build-xvfb.log').read_text(), 'build display diagnostics\n') + self.assertEqual((record.parent/'export-xvfb.log').read_text(), 'export display diagnostics\n') + self.assertFalse(list(self.root.glob('.pt-run-*'))) + + @patch.object(pt.shutil, 'which', return_value='/bin/apptainer') + def test_missing_archive_is_failure_with_diagnostics(self, _): + self.image.touch() + inputs = self.root / 'input' + inputs.mkdir() + output = self.root / 'no-archive' + with patch.object(pt, '_run_pgdb_container'), self.assertRaisesRegex(RuntimeError, 'no PGDB archive'): + pt.run_pgdb(self.image, inputs, output, 'sample') + record = next(output.glob('diagnostics/*/execution.json')) + self.assertEqual(json.loads(record.read_text())['status'], 'FAILED') + + def test_registered_image_plans_single_cpu_entities(self): + from metapathways import pipeline + self.image.touch() + (self.root / 'magsplitter/results/MAG_1').mkdir(parents=True) + (self.root / 'magsplitter/results/non_binned').mkdir() + with patch.object(pipeline.sys, 'argv', ['metapathways', 'ptools', '-o', str(self.root)]), patch.object(pt, 'registered_image', return_value=str(self.image)), patch.object(pt.nextflow, 'launch') as launch: + pipeline.ptools() + tasks = launch.call_args.args[0] + self.assertEqual(len(tasks), 2) + self.assertEqual([t['cpus'] for t in tasks], [1, 1]) + self.assertEqual([t['allow_failure'] for t in tasks], [False, True]) + self.assertTrue(all('--image' in t['commands'][0] for t in tasks)) + + def test_no_transport_and_single_entity_are_forwarded(self): + from metapathways import pipeline + self.image.touch() + (self.root / 'magsplitter/results/MAG_1').mkdir(parents=True) + with patch.object(pipeline.sys, 'argv', ['metapathways', 'ptools', '-o', str(self.root), + '--entity', 'community', '--no_transport_inference', '--taxprune', '--taxon_id', '131567']), \ + patch.object(pt, 'registered_image', return_value=str(self.image)), \ + patch.object(pt.nextflow, 'launch') as launch: + pipeline.ptools() + tasks = launch.call_args.args[0] + self.assertEqual(len(tasks), 1) + self.assertIn('--no_transport_inference', tasks[0]['commands'][0]) + self.assertIn('--taxon_id 131567', tasks[0]['commands'][0]) + self.assertIn('--taxprune', tasks[0]['commands'][0]) + + @patch.object(pt.shutil, 'which', return_value='/bin/apptainer') + def test_transport_switch_controls_tip_argument(self, _): + self.image.touch() + inputs = self.root/'switch-input' + inputs.mkdir() + for enabled in (True, False): + def run(command, *args, **kwargs): + self.assertEqual('-tip' in command, enabled) + state = Path(command[command.index('--home')+1].rsplit(':', 1)[0]) + (state/'output/samplecyc.tar.bz2').write_bytes(b'archive') + with patch.object(pt, '_run_pgdb_container', side_effect=run): + pt.run_pgdb(self.image, inputs, self.root/str(enabled), 'sample', transport_inference=enabled) + + def test_worker_rechecks_outputs_and_directory_inputs(self): + from metapathways import nf_worker as worker + source = self.root / 'source' + source.mkdir() + (source / 'input').write_text('first') + output = self.root / 'result' + t = pt.nextflow.task('fixture', 'fixture', ['mock-tool'], [str(source)], [str(output)], adopt_existing=False) + t['receipt'] = str(self.root / 'durable.json') + def run(*args, **kwargs): + output.write_text((source / 'input').read_text()) + previous_cwd = Path.cwd() + try: + os.chdir(self.root) + with patch.object(worker.subprocess, 'run', side_effect=run) as invoke: + self.assertEqual(worker.execute(t), 0) + self.assertEqual(worker.execute(t), 0) + self.assertEqual(invoke.call_count, 1) + (source / 'input').write_text('changed input') + self.assertEqual(worker.execute(t), 0) + self.assertEqual(output.read_text(), 'changed input') + self.assertEqual(invoke.call_count, 2) + output.unlink() + self.assertEqual(worker.execute(t), 0) + self.assertEqual(invoke.call_count, 3) + t['cache_version'] = 'private-fast-scratch-v1' + self.assertEqual(worker.execute(t), 0) + self.assertEqual(invoke.call_count, 4) + finally: + os.chdir(previous_cwd) + + def test_optional_failure_recorded_but_community_failure_fatal(self): + from metapathways import nf_worker as worker + t = pt.nextflow.task('fixture', 'fixture', ['mock-tool'], outputs=[str(self.root / 'missing')], adopt_existing=False) + t['receipt'] = str(self.root / 'durable.json') + previous_cwd = Path.cwd() + try: + os.chdir(self.root) + with patch.object(worker.subprocess, 'run', side_effect=RuntimeError('PGDB failed')), contextlib.redirect_stderr(io.StringIO()): + self.assertEqual(worker.execute(t), 1) + t['allow_failure'] = True + self.assertEqual(worker.execute(t), 0) + self.assertEqual(json.loads(Path(t['receipt']).read_text())['status'], 'FAILED') + finally: + os.chdir(previous_cwd) + + +if __name__ == '__main__': + unittest.main() + +class ContainerStartupRetryTests(unittest.TestCase): + def test_mount_failure_retries_and_records_attempts(self): + with tempfile.TemporaryDirectory() as directory: + state = Path(directory) + status = {} + failure = 'FATAL: container creation failed: mount hook function failure: mount /etc/hosts error' + with patch.object(pt, '_stream_container', side_effect=[(255, failure), (0, '')]), patch.object(pt.time, 'sleep') as sleep: + pt._run_pgdb_container(['fixture'], state, status) + self.assertEqual([a['exit_code'] for a in status['container_attempts']], [255, 0]) + sleep.assert_called_once_with(5) + + def test_persistent_mount_failure_stops_after_three_attempts(self): + with tempfile.TemporaryDirectory() as directory: + status = {} + with patch.object(pt, '_stream_container', return_value=(255, 'container creation failed: mount hook function failure')) as run, patch.object(pt.time, 'sleep') as sleep: + with self.assertRaises(subprocess.CalledProcessError): + pt._run_pgdb_container(['fixture'], Path(directory), status) + self.assertEqual(run.call_count, 3) + self.assertEqual([c.args[0] for c in sleep.call_args_list], [5, 15]) + self.assertEqual(len(status['container_attempts']), 3) + + def test_no_retry_after_shell_entry_or_for_unrelated_errors(self): + for stage, error in [('container setup', 'container creation failed: mount hook function failure'), + ('build', 'container creation failed: mount hook function failure'), + ('container startup', 'FATAL: image not found')]: + with self.subTest(stage=stage, error=error), tempfile.TemporaryDirectory() as directory: + state = Path(directory) + def fail(command, log): + (state/'stage.txt').write_text(stage) + return 255, error + with patch.object(pt, '_stream_container', side_effect=fail) as run, patch.object(pt.time, 'sleep') as sleep: + with self.assertRaises(subprocess.CalledProcessError): + pt._run_pgdb_container(['fixture'], state, {}) + self.assertEqual(run.call_count, 1) + sleep.assert_not_called() + + def test_output_is_streamed_and_saved(self): + import sys + with tempfile.TemporaryDirectory() as directory, contextlib.redirect_stdout(io.StringIO()) as console: + log = Path(directory)/'attempt.log' + code, tail = pt._stream_container([sys.executable, '-c', 'import sys; print("startup failure", file=sys.stderr); sys.exit(255)'], log) + self.assertEqual(code, 255) + self.assertIn('startup failure', tail) + self.assertEqual(log.read_text(), console.getvalue()) diff --git a/tests/test_pt_exports.py b/tests/test_pt_exports.py new file mode 100644 index 0000000..0b41445 --- /dev/null +++ b/tests/test_pt_exports.py @@ -0,0 +1,101 @@ +import ast +import glob +import os +from pathlib import Path +import tempfile +import unittest +from unittest.mock import Mock + +import pandas as pd + +from metapathways.pt_exports import ( + EMPTY_EXPORT_ERROR, PATHWAY_COLUMNS, REPORT_COLUMNS, + write_verified_empty_pathways, +) + + +class EmptyPathwayTests(unittest.TestCase): + def setUp(self): + temp = tempfile.TemporaryDirectory() + self.addCleanup(temp.cleanup) + self.root = Path(temp.name) + self.flat = self.root / '1.0/data' + self.flat.mkdir(parents=True) + self.reports = self.root / '1.0/reports' + self.reports.mkdir() + self.raw = self.flat / 'pathways.dat' + self.raw.write_text('# Export\n' + EMPTY_EXPORT_ERROR + '\n') + self.summary = self.reports / 'pathways-report_2026-10-06.txt' + self.summary.write_text('# Report\n' + ' | '.join(REPORT_COLUMNS) + '\n') + self.evidence = self.reports / 'pwy-evidence-list.dat' + self.evidence.write_text(';;; This file contains all inferred pathways and super-pathways\n') + self.out = self.root / 'sample_pwy.tsv' + + def test_verified_empty_export_preserves_source(self): + before = self.raw.read_bytes() + self.assertTrue(write_verified_empty_pathways(self.flat, self.out)) + self.assertEqual(self.raw.read_bytes(), before) + self.assertEqual(self.out.read_text(), '\t'.join(PATHWAY_COLUMNS) + '\n') + + def test_comment_only_export_also_requires_evidence(self): + self.raw.write_text('# No instances\n') + self.assertTrue(write_verified_empty_pathways(self.flat, self.out)) + + def test_other_errors_and_real_records_are_not_suppressed(self): + for text in ('Error: something else\n', 'UNIQUE-ID - PWY-1\n//\n', + EMPTY_EXPORT_ERROR + '\nUNIQUE-ID - PWY-1\n//\n'): + self.raw.write_text(text) + self.assertFalse(write_verified_empty_pathways(self.flat, self.out)) + self.assertFalse(self.out.exists()) + + def test_missing_or_conflicting_reports_fail(self): + original = self.summary.read_text() + for text in ('', original + 'pathway | PWY-1\n', 'incorrect header\n'): + self.summary.write_text(text) + with self.assertRaises(ValueError): + write_verified_empty_pathways(self.flat, self.out) + self.assertFalse(self.out.exists()) + self.summary.unlink() + with self.assertRaises(ValueError): + write_verified_empty_pathways(self.flat, self.out) + + def test_missing_or_nonempty_evidence_fails(self): + for text in ('', ';;; This file contains all inferred pathways and super-pathways\n(PWY-1 RXN-1)\n'): + self.evidence.write_text(text) + with self.assertRaises(ValueError): + write_verified_empty_pathways(self.flat, self.out) + self.evidence.unlink() + with self.assertRaises(FileNotFoundError): + write_verified_empty_pathways(self.flat, self.out) + self.assertFalse(self.out.exists()) + + def test_both_entrypoints_emit_empty_tables_without_calling_camelot(self): + # Load functions only: these legacy scripts parse CLI arguments at import. + repo = Path(__file__).resolve().parents[1] + (self.root / 'samplecyc.tar.bz2').touch() + annotations = self.root / 'results/annotation_table' + annotations.mkdir(parents=True) + pd.DataFrame([dict(orf_id='orf1', EC='1.2.3.4', RXN='RXN-1', + **{'ref dbname': 'db', 'target': 'protein', 'product': 'enzyme', + 'value': '1', 'trim_target': 'protein'})]).to_csv( + annotations / 'sample.EC_RXN_map.tsv', sep='\t', index=False) + for name in ('pgdb_build_wf.py', 'pgdb_build_single.py'): + with self.subTest(name=name): + tree = ast.parse((repo / 'dev' / name).read_text()) + functions = [node for node in tree.body if isinstance(node, ast.FunctionDef) + and node.name in ('extract_pwy', 'map_orfs2pwys')] + camelot = Mock(side_effect=AssertionError('Camelot should not parse empty exports')) + namespace = dict(os=os, glob=glob, pd=pd, make_camelot_file=camelot) + exec(compile(ast.Module(body=functions, type_ignores=[]), name, 'exec'), namespace) + namespace['extract_pwy'](str(self.root)) + namespace['map_orfs2pwys'](str(self.root), str(self.root)) + camelot.assert_not_called() + self.assertTrue(pd.read_csv(self.out, sep='\t').empty) + mapping = pd.read_csv(self.root / 'sample_pwy2orf.tsv', sep='\t') + self.assertTrue(mapping.empty) + self.assertIn('orf_id', mapping.columns) + self.assertIn('PWY_NAME', mapping.columns) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_pt_reactions.py b/tests/test_pt_reactions.py new file mode 100644 index 0000000..7a53504 --- /dev/null +++ b/tests/test_pt_reactions.py @@ -0,0 +1,65 @@ +import json +from pathlib import Path +import tempfile +import unittest + +from metapathways.pt_reactions import filter_reactions + + +class ReactionFilterTests(unittest.TestCase): + def test_only_blocked_assignments_removed_and_audited(self): + with tempfile.TemporaryDirectory() as temp: + root = Path(temp) + source = ('ID\tC137456-G1\nNAME\tC137456-G1\n' + 'FUNCTION\tsublancin 168 maturation protease / ABC efflux transporter\n' + 'METACYC\tTRANS-RXN8J2-121\nMETACYC\tSAFE-RXN\n' + 'EC\t3.4.21.1\nPRODUCT-TYPE\tP\nSTARTBASE\t3\nENDBASE\t242\n//\n') + original = root/'original.txt' + original.write_text(source) + for name in ('0.pf', 'contig_1.pf'): + (root/name).write_text(source) + audit = filter_reactions(root) + for name in ('0.pf', 'contig_1.pf'): + self.assertEqual((root/name).read_text(), source.replace('METACYC\tTRANS-RXN8J2-121\n', '')) + self.assertEqual(original.read_text(), source) + self.assertEqual(len(audit['removed']), 2) + self.assertTrue(all(r['feature_id'] == 'C137456-G1' for r in audit['removed'])) + self.assertEqual(json.loads((root/'ptools-reaction-filter.json').read_text()), audit) + + def test_unlisted_reaction_ids_are_not_substring_matches(self): + with tempfile.TemporaryDirectory() as temp: + root = Path(temp) + source = 'ID\tgene\nMETACYC\tTRANS-RXN8J2-1210\nMETACYC\tSAFE-RXN\n//\n' + (root/'0.pf').write_text(source) + self.assertEqual(filter_reactions(root)['removed'], []) + self.assertEqual((root/'0.pf').read_text(), source) + + def test_matching_db_list_and_stale_mapping_fallback(self): + from metapathways.pt_screen import digest + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + categories = root/'db/functional_categories' + categories.mkdir(parents=True) + mapping = categories/'MetaCyc-monomer-rxn-pairs.tsv' + mapping.write_text('MC\tRXN\nM\tCUSTOM\n') + image = root/'pt.sif' + image.write_bytes(b'test image') + inputs = root/'inputs' + inputs.mkdir() + (root/'metapathways_run_log.txt').write_text(f'Minimum Required Arguments:refdb_dir\t{root/"db"}\n') + compatibility = categories/'ptools_reaction_compatibility.json' + compatibility.write_text(json.dumps(dict(image_sha256=digest(image), mapping_sha256=digest(mapping), + reactions={'CUSTOM': {'reason': 'verified'}}))) + pf = inputs/'0.pf' + pf.write_text('ID\tG1\nMETACYC\tCUSTOM\n//\n') + audit = filter_reactions(inputs, image=image, sample_output=root) + self.assertEqual(audit['blacklist_source'], str(compatibility)) + self.assertEqual(len(audit['removed']), 1) + mapping.write_text('changed') + pf.write_text('ID\tG1\nMETACYC\tCUSTOM\n//\n') + self.assertEqual(filter_reactions(inputs, image=image, sample_output=root)['removed'], []) + self.assertIn('METACYC\tCUSTOM', pf.read_text()) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_pt_screen.py b/tests/test_pt_screen.py new file mode 100644 index 0000000..937b0ec --- /dev/null +++ b/tests/test_pt_screen.py @@ -0,0 +1,115 @@ +import argparse +import json +from pathlib import Path +import tempfile +import unittest + +from metapathways.pt_screen import Screen, write_inputs + + +class ScreenTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + self.root = Path(self.tmp.name) + mapping = self.root/'db/functional_categories' + mapping.mkdir(parents=True) + (mapping/'MetaCyc-monomer-rxn-pairs.tsv').write_text('MC\tRXN\nM1\tGOOD\nM2\tBAD\n') + image = self.root/'image.sif' + image.write_bytes(b'image') + self.args = argparse.Namespace(output_dir=str(self.root/'out'), image=str(image), + refdb_dir=str(self.root/'db'), reactions=None, batch_size=100, + confirm_runs=2, timeout=30, scratch_dir=None, max_tasks=1) + self.calls = [] + + def runner(self, image, directory, reactions, timeout, scratch): + self.calls.append(tuple(reactions)) + return {'status': 'FAIL' if 'BAD' in reactions else 'PASS'} + + def test_bisection_confirmation_and_resume(self): + screen = Screen(self.args, self.runner) + screen.run() + self.assertEqual(set(screen.candidates), {'BAD'}) + self.assertEqual(self.calls.count(('BAD',)), 2) + self.assertFalse(screen.interactions) + self.calls.clear() + resumed = Screen(self.args, self.runner) + resumed.run() + self.assertEqual(set(resumed.candidates), {'BAD'}) + self.assertEqual(self.calls, []) + + def test_batch_interaction_is_not_blacklisted(self): + def runner(*args): + return {'status': 'FAIL' if len(args[2]) > 1 else 'PASS'} + screen = Screen(self.args, runner) + screen.run() + self.assertFalse(screen.candidates) + self.assertEqual(len(screen.interactions), 1) + + def test_inconclusive_is_not_blacklisted(self): + screen = Screen(self.args, lambda *args: {'status': 'INCONCLUSIVE'}) + with self.assertRaises(RuntimeError): + screen.run() + self.assertFalse(screen.candidates) + + def test_failed_control_prevents_candidate(self): + screen = Screen(self.args, self.runner) + screen.attempt([], 'baseline') + screen.runner = lambda *args: {'status': 'FAIL'} + screen.isolate(['BAD']) + self.assertFalse(screen.candidates) + + def test_confirmation_must_repeat_failure(self): + screen = Screen(self.args, self.runner) + calls = 0 + def runner(*args): + nonlocal calls + calls += 1 + return {'status': 'FAIL' if calls == 1 else 'PASS'} + screen.runner = runner + screen.isolate(['BAD']) + self.assertFalse(screen.candidates) + + def test_interrupted_attempt_is_preserved_and_retried(self): + screen = Screen(self.args, self.runner) + def interrupt(*args): + (args[1]/'console.log').write_text('partial evidence') + raise KeyboardInterrupt() + screen.runner = interrupt + with self.assertRaises(KeyboardInterrupt): + screen.attempt(['BAD'], 'screen') + screen.runner = self.runner + self.assertEqual(screen.attempt(['BAD'], 'screen')['status'], 'FAIL') + preserved = list((screen.output/'attempts').glob('*-interrupted-*')) + self.assertEqual(len(preserved), 1) + self.assertEqual((preserved[0]/'console.log').read_text(), 'partial evidence') + + def test_inconclusive_receipt_retried_on_resume(self): + screen = Screen(self.args, lambda *args: {'status': 'INCONCLUSIVE'}) + screen.attempt(['BAD'], 'screen') + resumed = Screen(self.args, self.runner) + self.assertEqual(resumed.attempt(['BAD'], 'screen')['status'], 'FAIL') + self.assertEqual(self.calls, [('BAD',)]) + + def test_full_screen_publication(self): + self.args.publish = True + screen = Screen(self.args, self.runner) + screen.run() + file = Path(self.args.refdb_dir)/'functional_categories/ptools_reaction_compatibility.json' + data = json.loads(file.read_text()) + self.assertEqual(set(data['reactions']), {'BAD'}) + self.assertEqual(data['mapping_sha256'], screen.identity['mapping_sha256']) + + def test_changed_image_cannot_resume(self): + Screen(self.args, self.runner) + Path(self.args.image).write_bytes(b'new image') + with self.assertRaises(ValueError): + Screen(self.args, self.runner) + + def test_inputs_have_sequences_and_exact_reaction_ids(self): + directory = self.root/'input' + write_inputs(directory, ['BAD', 'GOOD'], 'MPscreen') + self.assertIn('METACYC\tBAD\n', (directory/'contig.pf').read_text()) + sequence = (directory/'contig.fasta').read_text().splitlines()[1] + self.assertEqual(len(sequence), 480) + self.assertIn('ENDBASE\t480', (directory/'contig.pf').read_text()) diff --git a/tests/test_pt_sequences.py b/tests/test_pt_sequences.py new file mode 100644 index 0000000..b2e8390 --- /dev/null +++ b/tests/test_pt_sequences.py @@ -0,0 +1,144 @@ +from pathlib import Path +import tempfile +import unittest + +from metapathways.pt_sequences import attach_sequences + + +class SequenceInputTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.base = self.root/'sample' + self.inputs = self.root/'private-input' + self.inputs.mkdir() + (self.base/'preprocessed').mkdir(parents=True) + (self.base/'preprocessed/sample.fasta').write_text('>contigA\nATGAAATAG\n>contigB\nTTATTTCAT\n') + table = self.base/'results/annotation_table/sample.ptinput.tsv' + table.parent.mkdir(parents=True) + table.write_text('orf_id\tseqname\tstart\tend\tstrand\nA-G1\tcontigA\t1\t9\t+\nB-G1\tcontigB\t1\t9\t-\n') + self.records = ['ID\tA-G1\nSTARTBASE\t1\nENDBASE\t9\nFUNCTION\tx\nPRODUCT-TYPE\tP\n//\n', + 'ID\tB-G1\nSTARTBASE\t9\nENDBASE\t1\nFUNCTION\ty\nPRODUCT-TYPE\tP\n//\n'] + (self.inputs/'0.pf').write_text(''.join(self.records)) + (self.inputs/'genetic-elements.dat').write_text('old compact manifest') + + def test_real_contigs_keep_feature_ids_coordinates_and_strands(self): + attach_sequences(self.inputs, self.base) + manifest = (self.inputs/'genetic-elements.dat').read_text() + self.assertEqual(manifest.count('SEQ-FILE\t'), 2) + self.assertEqual(set((self.inputs/'contig_1.pf').read_text().splitlines()), set(self.records[0].splitlines())) + self.assertEqual(set((self.inputs/'contig_2.pf').read_text().splitlines()), set(self.records[1].splitlines())) + self.assertEqual((self.inputs/'contig_1.fasta').read_text(), '>contig_1\nATGAAATAG\n') + self.assertEqual((self.inputs/'contig_2.fasta').read_text(), '>contig_2\nTTATTTCAT\n') + self.assertEqual((self.inputs/'0.pf').read_text(), ''.join(self.records)) + + def test_taxon_override_replaces_only_taxon_and_rejects_invalid_ids(self): + from metapathways.pt_sequences import set_organism_taxon + params = self.inputs/'organism-params.dat' + params.write_text('ID\tsample\nNCBI-TAXON-ID\t12908\nSTORAGE\tFILE\n') + set_organism_taxon(self.inputs, 131567) + self.assertEqual(params.read_text(), 'ID\tsample\nSTORAGE\tFILE\nNCBI-TAXON-ID\t131567\n') + for value in (0, -1, '131567', True): + with self.assertRaises(ValueError): + set_organism_taxon(self.inputs, value) + + def test_trna_name_formatting_preserves_ids_and_unknown_anticodons(self): + import json + from metapathways.pt_sequences import normalize_trna_name + for label in ('GluTTC', 'fMetCAT', 'LeuTAA', 'SeCTCA'): + name = 'C1094.tRNA2-' + label + expected = name[:-3] + '-' + name[-3:] + self.assertEqual(normalize_trna_name(name), expected) + self.assertEqual(normalize_trna_name(expected), expected) + for name in ('C1.tRNA1-UndetNNN', 'C1.tRNA1-GluNNC', 'customGluTTC'): + self.assertEqual(normalize_trna_name(name), name) + text = self.records[0].replace('FUNCTION\tx', 'NAME\tC1094.tRNA2-GluTTC').replace('PRODUCT-TYPE\tP', 'PRODUCT-TYPE\tTRNA') + (self.inputs/'0.pf').write_text(text) + attach_sequences(self.inputs, self.base) + written = (self.inputs/'contig_1.pf').read_text() + self.assertIn('ID\tA-G1\n', written) + self.assertIn('NAME\tC1094.tRNA2-Glu-TTC\n', written) + self.assertEqual((self.inputs/'0.pf').read_text(), text) + audit = json.loads((self.inputs/'sequence-input.json').read_text()) + self.assertEqual(audit['normalized_trna_names']['A-G1'], + {'original': 'C1094.tRNA2-GluTTC', 'staged': 'C1094.tRNA2-Glu-TTC'}) + + def test_prodigal_genetic_codes_are_preserved_per_contig(self): + folder = self.base/'orf_prediction' + folder.mkdir() + gff = folder/'sample.cds.gff' + gff.write_text('# Sequence Data: seqhdr="contigA"\n# Model Data: transl_table=4;uses_sd=0\n' + '# Sequence Data: seqhdr="contigB"\n# Model Data: transl_table=11;uses_sd=1\n') + attach_sequences(self.inputs, self.base) + manifest = (self.inputs/'genetic-elements.dat').read_text() + self.assertIn('NAME\tcontigA\nTYPE\t:CONTIG\nCODON-TABLE\t4', manifest) + self.assertIn('NAME\tcontigB\nTYPE\t:CONTIG\nCODON-TABLE\t11', manifest) + gff.write_text(gff.read_text() + '# Sequence Data: seqhdr="contigA"\n# Model Data: transl_table=11;\n') + with self.assertRaisesRegex(ValueError, 'Conflicting'): + attach_sequences(self.inputs, self.base) + + def test_new_pf_writer_normalizes_string_and_list_ec_values(self): + from io import StringIO + from metapathways.MetaPathways_create_genbank_ptinput import write_to_pf_file + for value in ('1.2.3.4,2.3.4.5', ['1.2.3.4,2.3.4.5', '1.2.3.4']): + handle = StringIO() + write_to_pf_file(str(self.inputs), 'A-G1', + dict(strand='+', start=1, end=9, feature='CDS', ec=value), handle, True) + self.assertEqual([line for line in handle.getvalue().splitlines() if line.startswith('EC\t')], + ['EC\t1.2.3.4', 'EC\t2.3.4.5']) + + def test_legacy_ec_lists_are_split_without_losing_provisional_values(self): + import json + text = self.records[0].replace('//', 'EC\t1.2.3.4, 2.3.4.5\nEC\t1.2.3.4|3.6.5.n1\n//') + (self.inputs/'0.pf').write_text(text) + attach_sequences(self.inputs, self.base) + written = (self.inputs/'contig_1.pf').read_text() + self.assertEqual([line for line in written.splitlines() if line.startswith('EC\t')], + ['EC\t1.2.3.4', 'EC\t2.3.4.5', 'EC\t3.6.5.n1']) + self.assertEqual((self.inputs/'0.pf').read_text(), text) + self.assertEqual(json.loads((self.inputs/'sequence-input.json').read_text())['normalized_ec_records'], 1) + + def test_mag_gets_only_the_contigs_its_records_reference(self): + (self.inputs/'0.pf').write_text(self.records[1]) + attach_sequences(self.inputs, self.base) + self.assertEqual((self.inputs/'genetic-elements.dat').read_text().count('SEQ-FILE\t'), 1) + self.assertIn('NAME\tcontigB', (self.inputs/'genetic-elements.dat').read_text()) + self.assertIn('TTATTTCAT', (self.inputs/'contig_1.fasta').read_text()) + + def test_missing_sequence_fails_before_replacing_manifest(self): + (self.base/'preprocessed/sample.fasta').write_text('>contigA\nATGAAATAG\n') + with self.assertRaisesRegex(ValueError, 'Missing Pathway Tools contig sequences: contigB'): + attach_sequences(self.inputs, self.base) + self.assertEqual((self.inputs/'genetic-elements.dat').read_text(), 'old compact manifest') + + def test_invalid_coordinates_fail_before_replacing_manifest(self): + table = self.base/'results/annotation_table/sample.ptinput.tsv' + table.write_text(table.read_text().replace('contigA\t1\t9', 'contigA\t1\t10')) + with self.assertRaisesRegex(ValueError, 'outside contigA'): + attach_sequences(self.inputs, self.base) + self.assertEqual((self.inputs/'genetic-elements.dat').read_text(), 'old compact manifest') + + def test_mag_representative_coordinates_are_restored_to_actual_member(self): + (self.inputs/'0.pf').write_text(self.records[1].replace('STARTBASE\t9', 'STARTBASE\t900')) + attach_sequences(self.inputs, self.base) + restored = (self.inputs/'contig_1.pf').read_text() + self.assertIn('STARTBASE\t9\n', restored) + self.assertIn('ENDBASE\t1\n', restored) + self.assertIn('"restored_coordinates": 1', (self.inputs/'sequence-input.json').read_text()) + + def test_unannotated_contig_is_not_required_or_exported(self): + with (self.base/'preprocessed/sample.fasta').open('a') as stream: + stream.write('>unannotated\nATGC\n') + attach_sequences(self.inputs, self.base) + self.assertNotIn('unannotated', (self.inputs/'genetic-elements.dat').read_text()) + + def test_unmapped_or_duplicate_features_fail(self): + for records in [self.records[0].replace('A-G1', 'unknown'), self.records[0]*2]: + (self.inputs/'0.pf').write_text(records) + with self.assertRaisesRegex(ValueError, 'Duplicate or unmapped'): + attach_sequences(self.inputs, self.base) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_pt_taxonomy.py b/tests/test_pt_taxonomy.py new file mode 100644 index 0000000..7011376 --- /dev/null +++ b/tests/test_pt_taxonomy.py @@ -0,0 +1,45 @@ +import argparse +import contextlib +import io +import unittest +from metapathways.pt_taxonomy import add_taxonomy_options, resolve_taxon +from metapathways import pipeline, analysis_workflow + + +class TaxonomyTests(unittest.TestCase): + def parser(self): + parser = argparse.ArgumentParser() + add_taxonomy_options(parser) + return parser + + def test_names_and_alias_resolve(self): + for name, taxon in [('all', 131567), ('bacteria', 2), ('archaea', 2157), + ('eukaryotes', 2759), ('euks', 2759)]: + with self.subTest(name=name): + self.assertEqual(resolve_taxon(self.parser().parse_args(['--taxonomic_scope', name])), taxon) + + def test_default_cellular_life_and_numeric_ids_work(self): + self.assertEqual(resolve_taxon(self.parser().parse_args([])), 131567) + self.assertEqual(resolve_taxon(self.parser().parse_args(['--taxon_id', '562'])), 562) + + def test_invalid_and_conflicting_options_fail_early(self): + for argv in [['--taxon_id', '0'], ['--taxon_id', '-1'], + ['--taxonomic_scope', 'prokaryotes'], ['--taxonomic_scope', 'unknown'], + ['--taxon_id', '2', '--taxonomic_scope', 'all']]: + with self.subTest(argv=argv), contextlib.redirect_stderr(io.StringIO()), self.assertRaises(SystemExit): + self.parser().parse_args(argv) + + def test_both_commands_expose_option(self): + args = pipeline.ptParser().parse_args(['ptools', '-o', '/tmp/example', '--taxonomic_scope', 'all']) + self.assertEqual(resolve_taxon(args), 131567) + args = analysis_workflow.parser().parse_args(['analysis_wf', '-i', '/tmp/in', '-o', '/tmp/out', + '-d', '/tmp/db', '--taxonomic_scope', 'euks']) + self.assertEqual(resolve_taxon(args), 2759) + + def test_pruning_defaults_and_opt_out(self): + parser = pipeline.ptParser() + self.assertTrue(parser.parse_args(['ptools', '-o', '/tmp/out']).taxprune) + self.assertFalse(parser.parse_args(['ptools', '-o', '/tmp/out', '--no_taxprune']).taxprune) + parser = analysis_workflow.parser() + self.assertTrue(parser.parse_args(['analysis_wf']).taxprune) + self.assertFalse(parser.parse_args(['analysis_wf', '--no_taxprune']).taxprune) diff --git a/tests/test_read_mapping.py b/tests/test_read_mapping.py new file mode 100644 index 0000000..fcb57c3 --- /dev/null +++ b/tests/test_read_mapping.py @@ -0,0 +1,190 @@ +"""Read-layout regressions, exercised without external bioinformatics tools.""" +import inspect +import os +import shlex +import tempfile +import unittest +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import patch + +from metapathways import MetaPathways_tpm as tpm +from metapathways import jobscreator + + +class ReadMappingTests(unittest.TestCase): + def test_distinct_paired_inputs(self): + self.assertEqual(tpm.coverm_read_arguments('R1.fq', 'R2.fq'), + ['-1', 'R1.fq', '-2', 'R2.fq']) + + def test_interleaved_with_missing_reverse(self): + for reverse in (None, '', 'None'): + with self.subTest(reverse=reverse): + self.assertEqual(tpm.coverm_read_arguments('reads.fq', reverse, True), + ['--interleaved', 'reads.fq']) + + def test_single_with_missing_reverse(self): + for reverse in (None, '', 'None'): + with self.subTest(reverse=reverse): + self.assertEqual(tpm.coverm_read_arguments('reads.fq', reverse), + ['--single', 'reads.fq']) + + def test_reject_conflicting_or_missing_inputs(self): + for args in [('reads.fq', 'reverse.fq', True), (None, 'reverse.fq', False), + ('None', None, True), ('', None, False), ('reads.fq', 'reads.fq', False)]: + with self.subTest(args=args), self.assertRaises(ValueError): + tpm.coverm_read_arguments(*args) + + def test_same_file_aliases_are_not_two_mates(self): + with tempfile.TemporaryDirectory() as tmp: + forward = Path(tmp) / 'reads.fq' + forward.touch() + for alias, link in [('symbolic.fq', os.symlink), ('hard.fq', os.link)]: + reverse = Path(tmp) / alias + link(forward, reverse) + with self.subTest(alias=alias), self.assertRaises(ValueError): + tpm.coverm_read_arguments(str(forward), str(reverse)) + + def test_main_dispatches_layout_and_stops_after_mapping_failure(self): + # Include the archived '-2 None --interleaved' invocation, and a stale + # BAM that must not be processed when CoverM fails. + for mode in ('paired', 'interleaved', 'single'): + with self.subTest(mode=mode), tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + for name in ['contigs.fa', 'annotation.gff', 'forward reads.fq', 'reverse reads.fq', 'stale.bam']: + (root / name).touch() + forward, reverse = str(root / 'forward reads.fq'), str(root / 'reverse reads.fq') + args = ['-c', str(root/'contigs.fa'), '-g', str(root/'annotation.gff'), + '-o', str(root/'counts.tsv'), '--stats', str(root/'mapping.log'), + '--bwaFolder', tmp, '--sample_name', 'sample', '--rpkmExec', 'coverm', + '--bwaExec', 'bwa', '-1', forward, '--num_threads', '8'] + if mode == 'paired': + args += ['-2', reverse] + elif mode == 'interleaved': + args += ['-2', 'None', '--interleaved'] + with patch.object(tpm.shutil, 'which', return_value='/mock/tool'), \ + patch.object(tpm, 'runRPKMCommand', return_value=(1, 'mapping failed')) as run, \ + patch.object(tpm, 'run_logged') as downstream, \ + patch.object(tpm.gutils, 'eprintf'): + self.assertEqual(tpm.main(args), 1) + self.assertEqual(downstream.call_count, 1) + self.assertEqual(downstream.call_args.args[0][0], 'gff2gtf.py') + cmd = shlex.split(run.call_args.kwargs['runcommand']) + self.assertEqual(cmd[cmd.index('-t')+1], '8') + if mode == 'paired': + self.assertEqual(cmd[cmd.index('-1')+1], forward) + self.assertEqual(cmd[cmd.index('-2')+1], reverse) + else: + flag = '--interleaved' if mode == 'interleaved' else '--single' + self.assertEqual(cmd[cmd.index(flag)+1], forward) + self.assertNotIn('-2', cmd) + self.assertIn('mapping failed', (root/'mapping.log').read_text()) + + def test_counting_layout_and_sort_threads(self): + for paired in (False, True): + with self.subTest(paired=paired), tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + for name in ('contigs.fa', 'annotation.gff', 'reads.fq', 'contigs.fa.reads.fq.bam', + 'sample.sorted.bam', 'sample.old.bam'): + (root / name).touch() + args = ['-c', str(root/'contigs.fa'), '-g', str(root/'annotation.gff'), + '-o', str(root/'counts.tsv'), '--stats', str(root/'mapping.log'), + '--bwaFolder', tmp, '--sample_name', 'sample', '--rpkmExec', 'coverm', + '--bwaExec', 'bwa', '-1', str(root/'reads.fq'), '--num_threads', '8'] + if paired: + args += ['--interleaved'] + with patch.object(tpm.shutil, 'which', return_value='/mock/tool'), \ + patch.object(tpm, 'runRPKMCommand', return_value=(0, '')), \ + patch.object(tpm, 'run_logged') as run: + self.assertEqual(tpm.main(args), 0) + commands = [call.args[0] for call in run.call_args_list] + sort = next(c for c in commands if c[0] == 'samtools') + counts = next(c for c in commands if c[0] == 'featureCounts') + self.assertEqual(sort[sort.index('-@')+1], '7') + self.assertEqual(sort[-1], str(root/'contigs.fa.reads.fq.bam')) + self.assertEqual(sort[sort.index('-o')+1], str(root/'sample.sorted.bam')) + self.assertEqual(counts[counts.index('-T')+1], '8') + self.assertEqual('-p' in counts, paired) + # Existing sorted/stale BAMs cannot substitute for missing CoverM output. + (root/'contigs.fa.reads.fq.bam').unlink() + with patch.object(tpm.shutil, 'which', return_value='/mock/tool'), \ + patch.object(tpm, 'runRPKMCommand', return_value=(0, '')), \ + patch.object(tpm, 'run_logged') as run: + with self.assertRaisesRegex(RuntimeError, 'expected BAM'): + tpm.main(args) + self.assertEqual(run.call_count, 1) # GFF preparation only + + def test_pipeline_does_not_serialize_missing_reverse(self): + # Access the class wrapped by the project's Singleton decorator. + cls = jobscreator.ContextCreator + creator = object.__new__(cls) + creator.configs = SimpleNamespace(RPKM_EXECUTABLE='coverm', BWA_EXECUTABLE='bwa', + NUM_CPUS='8', RPKM_CALCULATION='MetaPathways_tpm') + creator.params = SimpleNamespace(get=lambda *a: 'yes') + sample = SimpleNamespace(fq_files=[['reads.fq', 'None'], True], bwa_folder='/tmp/bwa', + genbank_dir='/tmp/genbank/', sample_name='sample', + preprocessed_dir='/tmp/preprocessed', output_results_rpkm_dir='/tmp/rpkm') + cmd = shlex.split(creator.create_rpkm_cmd(sample)[0].commands[0]) + self.assertEqual(cmd[cmd.index('--num_threads')+1], '8') + self.assertIn('--interleaved', cmd) + self.assertNotIn('-2', cmd) + self.assertNotIn('None', cmd) + sample.fq_files = [['R1.fq', 'R2.fq'], False] + cmd = shlex.split(creator.create_rpkm_cmd(sample)[0].commands[0]) + self.assertEqual(cmd[cmd.index('-1')+1], 'R1.fq') + self.assertEqual(cmd[cmd.index('-2')+1], 'R2.fq') + self.assertNotIn('--interleaved', cmd) + + def test_orf_and_trna_commands_pass_configured_threads(self): + cls = jobscreator.ContextCreator + creator = object.__new__(cls) + creator.params = SimpleNamespace(get=lambda *a: 'yes') + sample = SimpleNamespace(sample_name='sample', preprocessed_dir='/tmp/input', + orf_prediction_dir='/tmp/orfs/', output_results_tRNA_dir='/tmp/trna') + for threads in (1, 3, 8): + creator.configs = SimpleNamespace(NUM_CPUS=threads, + ORF_PREDICTION='orf', PRODIGAL_EXECUTABLE='pprodigal', + SCAN_tRNA='trna', SCAN_tRNA_EXECUTABLE='ptRNAscan.py') + for method, flag in ((creator.create_orf_prediction_cmd, '--nthreads'), + (creator.create_tRNA_scan_statistics, '-t')): + with self.subTest(threads=threads, flag=flag): + cmd = shlex.split(method(sample)[0].commands[0]) + self.assertEqual(cmd[cmd.index(flag)+1], str(threads)) + + def test_barrnap_and_rrna_blast_use_the_same_budget(self): + cls = jobscreator.ContextCreator + creator = object.__new__(cls) + creator.params = SimpleNamespace(get=lambda group, key, **kw: + ['ssu'] if key == 'rRNA_refdbs' else 'yes') + creator.configs = SimpleNamespace(NUM_CPUS=3, PARSE_FUNC_SEARCH='parse', + SCAN_rRNA='rrna', REFDBS='/tmp/db') + sample = SimpleNamespace(sample_name='sample', preprocessed_dir='/tmp/input', + orf_prediction_dir='/tmp/orfs/', blast_results_dir='/tmp/blast', + output_results_rRNA_dir='/tmp/rrna/') + with patch.object(jobscreator.shutil, 'which', side_effect=lambda name: name): + contexts = creator.create_scan_rRNA_seqs_cmd(sample) + cmd = shlex.split(contexts[0].commands[0]) + self.assertEqual(cmd[cmd.index('--threads')+1], '3') + cmd = shlex.split(contexts[1].commands[1]) + self.assertEqual(cmd[cmd.index('-num_threads')+1], '3') + + def test_trnascan_workers_do_not_create_nested_thread_pools(self): + import importlib.util + spec = importlib.util.spec_from_file_location('ptrna_test', + Path(__file__).resolve().parents[1] / 'dev/ptRNAscan.py') + module = importlib.util.module_from_spec(spec) + # This worker test does not parse FASTA or require pyfastx. + with patch.dict('sys.modules', pyfastx=SimpleNamespace()): + spec.loader.exec_module(module) + args = SimpleNamespace(**dict.fromkeys(('bacterial', 'archaeal', 'mito', + 'general', 'genomic', 'eukaryotic', 'infernal', 'max', 'legacy', + 'cove', 'nopseudo', 'quiet'), False)) + module.file_parts_queue.put(('chunk.fasta', {'-o': 'chunk.txt'})) + with patch.object(module.subprocess, 'Popen') as run: + module.run_tRNAscan(0, args) + cmd = run.call_args.args[0] + self.assertEqual(cmd[cmd.index('--thread')+1], '1') + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_reporting.py b/tests/test_reporting.py new file mode 100644 index 0000000..6d21745 --- /dev/null +++ b/tests/test_reporting.py @@ -0,0 +1,217 @@ +import csv +import io +import json +from pathlib import Path +import sqlite3 +import tempfile +import threading +import unittest +from urllib.request import urlopen, Request +from urllib.error import HTTPError +from urllib.parse import urlencode + +from metapathways.reporting import build_report +from metapathways.report_server import ReportServer, query, csv_value + + +class ReportTests(unittest.TestCase): + def setUp(self): + self.temp=tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root=Path(self.temp.name) + for sample in ('alpha','beta'): + self.fixture(sample) + self.reports=build_report(self.root) + self.db=sqlite3.connect(self.reports/'results.sqlite') + self.db.row_factory=sqlite3.Row + self.addCleanup(self.db.close) + + def test_taxonomy_is_joined_by_sample_orf_database_and_target(self): + self.write('alpha', 'results/annotation_table/test.annotation_taxonomy.tsv', + 'orf_id\treference_db\ttarget\ttaxid\ttaxonomy\tlca_taxonomy\n' + 'G1\tswissprot\tACC1\t3\tSpecies A\tBacteria\n' + 'G1\tmetacyc\tWRONG_TARGET\t5\tArchaea\tArchaea\n') + build_report(self.root) + with sqlite3.connect(self.reports/'results.sqlite') as db: + actual = db.execute("SELECT reference_db,taxonomy,lca_taxonomy FROM annotation_explorer WHERE sample_id='alpha' ORDER BY reference_db").fetchall() + self.assertEqual(actual, [('metacyc','Not computed','Not computed'), ('swissprot','Species A','Bacteria')]) + self.assertEqual(db.execute("SELECT DISTINCT taxonomy FROM annotation_explorer WHERE sample_id='beta'").fetchall(), [('Not computed',)]) + + def test_wide_abundance_preserves_values_nulls_and_sample_identity(self): + for sample in ('alpha','beta'): + prefix = f'{sample}.fasta/{sample}.fastq.gz ' + headers = ['Length','Read Count','Mean','Variance','Trimmed Mean','RPKM','TPM','Extra'] + self.write(sample, 'results/rpkm/test.contig_counts.tsv', + 'Contig\t'+'\t'.join(prefix+h for h in headers)+'\nC1\t200\t0\t1.5\t2.5\t1.2\t3.4\t4.5\t9\n') + build_report(self.root) + with sqlite3.connect(self.reports/'results.sqlite') as db: + db.row_factory = sqlite3.Row + self.assertEqual(db.execute('SELECT COUNT(*) FROM abundance_explorer').fetchone()[0],4) + row = db.execute("SELECT * FROM abundance_explorer WHERE sample_id='alpha' AND feature_type='contig'").fetchone() + self.assertEqual((row['length_bp'],row['count'],row['mean_coverage'],row['coverage_variance'],row['trimmed_mean_coverage'],row['rpkm'],row['tpm']), (200,0,1.5,2.5,1.2,3.4,4.5)) + orf = db.execute("SELECT * FROM abundance_explorer WHERE sample_id='alpha' AND feature_type='orf'").fetchone() + self.assertEqual((orf['orf_id'],orf['contig_id'],orf['count']),('G1','C1',10)) + self.assertIsNone(orf['mean_coverage']) + self.assertIsNone(orf['length_bp']) # Absent in this legacy fixture; do not invent a length. + self.assertEqual(db.execute("SELECT value FROM abundance WHERE sample_id='alpha' AND measurement LIKE '% Extra'").fetchone()[0],9) + spec = {'table':'abundance_explorer','filters':[{'column':'count','op':'ge','value':10}], 'columns':['sample_id','feature_id','count']} + columns, result = query(db, spec, export=True) + self.assertEqual(columns, spec['columns']) + self.assertEqual(len(list(result)),2) + + def test_report_footer_run_details_and_sample_default(self): + self.write('alpha', 'logs/run/example/summary.json', json.dumps({ + 'mp_version':'3.5.1', 'status':'SUCCESS', 'resources':{'executor':'slurm'}, 'tasks':[]})) + build_report(self.root) + meta = json.loads((self.reports/'schema.json').read_text()) + self.assertEqual(meta['run_details'], {'mp_version':'3.5.1','status':'SUCCESS','executor':'slurm','command':'run'}) + self.assertIn('report_mp_version', meta) + page = (self.reports/'EDA_portal.html').read_text() + self.assertIn('issues/new/choose', page) + self.assertNotIn('Source abundance values are not recomputed', page) + self.assertIn('id="runDetails"', page) + with sqlite3.connect(self.reports/'results.sqlite') as db: + db.row_factory = sqlite3.Row + self.assertEqual(query(db, {})['columns'], ['sample_id','output_path']) + js = (self.reports/'portal.js').read_text() + self.assertIn("params.get('table')||'samples'", js) + + def test_expected_rna_is_not_a_missing_protein_annotation(self): + self.write('alpha', 'results/rpkm/test.orf_counts.tsv', + 'Gene_ID\tCount\tRPKM\tTPM\tfeature\tseqname\n' + 'G1\t10\t1\t2\tCDS\tC1\n' + 'RNA1\t5\t1\t2\ttRNA\tC1\n' + 'RNA2\t6\t1\t2\trRNA\tC1\n' + 'MISSING_CDS\t7\t1\t2\tCDS\tC1\n') + build_report(self.root) + with sqlite3.connect(self.reports/'results.sqlite') as db: + messages = [r[0] for r in db.execute("SELECT message FROM issues WHERE sample_id='alpha'")] + self.assertFalse(any('original read mapping' in message for message in messages)) + self.assertEqual([m for m in messages if 'lack primary annotations' in m], + ['1 referenced ORF identifiers lack primary annotations; placeholders preserve these relationships.']) + self.assertEqual(db.execute("SELECT COUNT(*) FROM abundance_explorer WHERE sample_id='alpha'").fetchone()[0],4) + self.assertEqual(db.execute("SELECT COUNT(*) FROM orfs WHERE sample_id='alpha' AND orf_id IN ('RNA1','RNA2')").fetchone()[0],2) + + def write(self, sample, relative, value): + path=self.root/sample/relative + path.parent.mkdir(parents=True,exist_ok=True) + path.write_text(value) + + def fixture(self,sample): + self.write(sample,'preprocessed/test.mapping.txt','C1\toriginal1\t200\n') + self.write(sample,'results/annotation_table/test.functional_and_taxonomic_table.txt', + 'ORF_ID\tORF_length\tstart\tend\tContig_Name\tContig_length\tstrand\ttarget\tproduct\ttaxonomy\n' + f'G1\t90\t1\t90\tC1\t200\t+\tACC1\t{sample} kinase\tBacteria; Test\n' + 'G2\t90\t100\t190\tC1\t200\t-\tACC2\t\troot\n') + self.write(sample,'results/annotation_table/test.1.txt', + 'orf_id\tref dbname\ttarget\tproduct\tvalue\nG1\tswissprot\tACC1\tkinase\t12\n\tmetacyc\tACC2\tkinase\t5\n') + self.write(sample,'results/annotation_table/test.EC_RXN_map.tsv', + 'orf_id\tref dbname\ttarget\tproduct\tvalue\tEC\tRXN\nG1\tswissprot\tACC1\tkinase\t12\t2.7.7.2|2.7.1.26\tRXN-1\nG1\tmetacyc\tACC2\tkinase\t5\t2.7.7.2\tRXN-2\n') + self.write(sample,'ptools/orf_map.txt','G1\tG2\n') + self.write(sample,'magsplitter/contig_to_mag.tsv','original1\tMAG1\n') + self.write(sample,'magsplitter/results/MAG1/0.pf','ID\tG1\nNAME\tG1\n//\n') + self.write(sample,'magsplitter/results/MAG_failed/0.pf','ID\tG2\n//\n') + header='SAMPLE\tPWY_NAME\tPWY_COMMON_NAME\tPWY_SCORE\tNUM_REACTIONS\tNUM_COVERED_REACTIONS\tORF_COUNT\tORFS\n' + self.write(sample,'results/pgdb/community/tag_pwy.tsv',header+'tag\tPWY-1\tTest pathway\t0.5\t4\t2\t2\tG1,G2\n') + self.write(sample,'results/pgdb/MAGs/MAG1/tag_pwy.tsv',header+'MAG1\tPWY-1\tTest pathway\t0.5\t4\t2\t1\tG1\n') + self.write(sample,'results/rpkm/test.orf_counts.tsv','Gene_ID\tCount\tRPKM\tTPM\nG1\t10\t1.2\t3.4\n') + self.write(sample,'logs/ptools/run1/summary.json',json.dumps({'tasks':[{'task':'pgdb-MAG_failed','label':'MAG_failed','status':'FAILED','error':'Expected optional failure'}]})) + + def test_join_keys_and_no_annotation_cartesian_product(self): + self.assertFalse(self.db.execute('PRAGMA foreign_key_check').fetchall()) + self.assertEqual(query(self.db,{'table':'orf_explorer'})['total'],4) + self.assertEqual(query(self.db,{'table':'annotation_explorer'})['total'],4) + self.assertEqual(query(self.db,{'table':'pathway_gene_explorer'})['total'],6) + result=query(self.db,{'table':'pathway_gene_explorer','filters':[{'column':'sample_id','op':'eq','value':'alpha'},{'column':'entity_id','op':'eq','value':'MAG1'}]}) + self.assertEqual(result['total'],1) + self.assertEqual(result['rows'][0]['orf_id'],'G1') + self.assertEqual(query(self.db,{'table':'mag_gene_explorer'})['total'],4) + self.assertEqual(query(self.db,{'table':'mag_orf_explorer'})['total'],4) + failed=query(self.db,{'table':'entities','filters':[{'column':'entity_id','op':'eq','value':'MAG_failed'}]}) + self.assertEqual(failed['rows'][0]['last_task_status'],'FAILED') + self.assertEqual(failed['rows'][0]['pathway_status'],'unavailable') + + def test_combined_workflow_statuses_preserve_sample_identity(self): + self.write('', 'logs/analysis_wf/run2/summary.json', json.dumps({'tasks': [ + {'task': 'alpha:pgdb:MAG1', 'label': 'alpha:pgdb:MAG1', 'sample': 'alpha', + 'entity': 'MAG1', 'status': 'SKIPPED'}, + {'task': 'beta:pgdb:MAG1', 'label': 'beta:pgdb:MAG1', 'sample': 'beta', + 'entity': 'MAG1', 'status': 'SUCCESS'}]})) + build_report(self.root) + with sqlite3.connect(self.reports/'results.sqlite') as db: + self.assertEqual(db.execute("SELECT sample_id,last_task_status,pathway_status FROM entities WHERE entity_id='MAG1' ORDER BY sample_id").fetchall(), + [('alpha', 'SKIPPED', 'unavailable'), ('beta', 'SUCCESS', 'available')]) + self.assertEqual(db.execute("SELECT sample_id FROM pathways WHERE entity_id='MAG1'").fetchall(), [('beta',)]) + + def test_related_filters_preserve_rows_and_same_annotation(self): + spec={'table':'orf_explorer','related':{'pathway_id':'PWY-1','entity_id':'MAG1','ec':'2.7.7.2','reference_db':'swissprot'}} + self.assertEqual(query(self.db,spec)['total'],2) + spec['related']['reaction']='RXN-2' + self.assertEqual(query(self.db,spec)['total'],0) + + def test_projection_numeric_filter_and_literal_search(self): + result=query(self.db,{'table':'annotation_explorer','search':'kinase','columns':['orf_id','score'], + 'filters':[{'column':'score','op':'ge','value':'10'}]}) + self.assertEqual(result['total'],2) + self.assertEqual(set(result['rows'][0]),{'orf_id','score'}) + self.assertEqual(query(self.db,{'table':'orf_explorer','search':'%'})['total'],0) + self.assertEqual(query(self.db,{'table':'orf_explorer','search':''})['total'],2) + + def test_query_validation_and_read_only_csv(self): + for spec in ({'table':'orfs; DROP TABLE samples'}, {'columns':['";DROP TABLE samples']}, + {'filters':[{'column':'orf_id','op':'raw','value':'1=1'}]}, {'limit':0}): + with self.assertRaises(ValueError): query(self.db,spec) + columns, cursor=query(self.db,{'table':'orf_explorer','columns':['sample_id','orf_id'],'limit':1},export=True) + self.assertEqual(columns,['sample_id','orf_id']) + self.assertEqual(len(list(cursor)),4) + self.assertEqual(csv_value('=SUM(A1)'),"'=SUM(A1)") + self.assertEqual(csv_value(-3.4),-3.4) + + def test_missing_annotations_retained_and_report_refresh_does_not_touch_sources(self): + source=self.root/'alpha/results/annotation_table/test.functional_and_taxonomic_table.txt' + source.unlink() + extra=self.reports/'user-notes.txt'; extra.write_text('retain') + build_report(self.root) + with sqlite3.connect(self.reports/'results.sqlite') as db: + self.assertEqual(db.execute("SELECT COUNT(*) FROM orfs WHERE sample_id='alpha' AND annotation_present=0").fetchone()[0],2) + self.assertGreater(db.execute('SELECT COUNT(*) FROM issues').fetchone()[0],0) + self.assertEqual(extra.read_text(),'retain') + self.assertIn('EDA_portal.html',(self.reports/'MP_run_report.html').read_text()) + + def test_mag_period_normalization_matches_splitter_directory(self): + self.write('alpha','magsplitter/contig_to_mag.tsv','original1\tMAG.1\n') + build_report(self.root) + with sqlite3.connect(self.reports/'results.sqlite') as db: + row=db.execute("SELECT entity_id,original_mag_id FROM contig_mags WHERE sample_id='alpha'").fetchone() + self.assertEqual(row,('MAG_1','MAG.1')) + self.assertEqual(db.execute("SELECT COUNT(*) FROM mag_orf_explorer WHERE sample_id='alpha' AND entity_id='MAG_1'").fetchone()[0],2) + + def test_compact_hit_fallback_fills_blank_orf_identifiers(self): + (self.root/'alpha/results/annotation_table/test.EC_RXN_map.tsv').unlink() + build_report(self.root) + with sqlite3.connect(self.reports/'results.sqlite') as db: + self.assertEqual(db.execute("SELECT COUNT(*) FROM annotations WHERE sample_id='alpha' AND orf_id='G1'").fetchone()[0],2) + + def test_http_query_export_and_path_restrictions(self): + server=ReportServer(self.root) + self.addCleanup(server.server_close) + worker=threading.Thread(target=server.serve_forever,daemon=True);worker.start() + self.addCleanup(server.shutdown) + prefix=server.url.rsplit('/reports/',1)[0] + spec={'table':'orf_explorer','columns':['sample_id','orf_id'],'filters':[{'column':'sample_id','op':'eq','value':'alpha'}],'limit':1} + with urlopen(prefix+'/api/query?'+urlencode({'spec':json.dumps(spec)})) as response: + data=json.load(response) + self.assertEqual(data['total'],2) + self.assertEqual(len(data['rows']),1) + with urlopen(prefix+'/api/export?'+urlencode({'spec':json.dumps(spec)})) as response: + records=list(csv.reader(io.StringIO(response.read().decode()))) + self.assertEqual(len(records),3) + for path in ('/%2e%2e/etc/passwd','/.secret','/unknown'): + with self.assertRaises(HTTPError):urlopen(prefix+path) + with self.assertRaises(HTTPError): + urlopen(Request(prefix+'/api/meta',headers={'Origin':'https://example.com'})) + with urlopen(server.url) as response: + self.assertIn(b'Explore and export',response.read()) + + +if __name__=='__main__':unittest.main() diff --git a/tests/test_rna_abundance.py b/tests/test_rna_abundance.py new file mode 100644 index 0000000..03dd5b5 --- /dev/null +++ b/tests/test_rna_abundance.py @@ -0,0 +1,96 @@ +"""RNA locus identity, annotation buffer boundaries and abundance regressions.""" +import importlib.util +import io +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + + +def helper(name): + spec = importlib.util.spec_from_file_location(name, Path(__file__).resolve().parents[1] / 'dev' / (name + '.py')) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +class RnaAbundanceTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + + def put(self, name, text): + p = self.root / name + p.write_text(text) + return p + + def test_rna_emitted_once_across_buffers_and_on_rna_only_contigs(self): + from metapathways import MetaPathways_annotate_fast as annot + cds = self.put('cds.gff', ''.join(f's-C1\tprodigal\tCDS\t{i}\t{i+2}\t1\t+\t0\tID=s-C1-G{i}\n' for i in (1, 4))) + rrna = self.put('rrna.gff', ''.join(f'{contig}\tbarrnap\trRNA\t{start}\t{start+9}\t1\t+\t.\tName=5S_rRNA;product=5S ribosomal RNA\n' for contig, start in [('s-C1', 10), ('s-C1', 30), ('s-C2', 10)])) + trna = self.put('trna.gff', 's-C1\ttRNAscan\ttRNA\t50\t59\t1\t+\t.\tID=s-C1.trna1;Name=s-C1.tRNA1-Ala;isotype=Ala;anticodon=CGC\n') + original = annot.GffFileParser.__init__ + def small_buffer(obj, *args, **kwargs): + original(obj, *args, **kwargs) + obj.Size = 1 + output = self.root / 'annot.gff' + with patch.object(annot.GffFileParser, '__init__', small_buffer): + annot.create_annotation({}, {}, str(cds), ['stats'], str(rrna), ['stats'], str(trna), + str(output), str(self.root/'comparison'), {'s-C1': 100, 's-C2': 100}, [], + '>s-C1\nACGT\n>s-C2\nACGT\n', 's') + rows = [line.split('\t') for line in output.read_text().splitlines() if '\t' in line] + self.assertEqual(len(rows), 4) + ids = [row[8].split(';')[0] for row in rows] + self.assertEqual(len(set(ids)), 4) + self.assertEqual(sum(row[0] == 's-C2' for row in rows), 1) + + def fixtures(self, duplicate=False, zero=False): + counts = self.put('counts.tsv', 'Geneid\tChr\tStart\tEnd\tStrand\tLength\tsample.bam\nG1\tC1\t1\t100\t+\t100\t' + ('0' if zero else '10') + '\nG2\tC1\t201\t400\t+\t200\t' + ('0' if zero else '10') + '\n') + gtf = 'C1\tx\tCDS\t1\t100\t.\t+\t0\tgene_id "G1";\nC1\tx\trRNA\t201\t400\t.\t+\t.\tgene_id "G2";\n' + if duplicate: + gtf += 'C1\tx\trRNA\t501\t700\t.\t+\t.\tgene_id "G2";\n' + return counts, self.put('input.gtf', gtf) + + def test_abundance_uses_featurecounts_lengths_and_one_to_one_join(self): + abund = helper('abund_calc') + counts, gtf = self.fixtures() + legacy = self.put('legacy.tsv', 'G1\t999\nG2\t999\n') + result = abund.abundance_table(counts, gtf, legacy) + self.assertEqual(result.Length.tolist(), [100, 200]) + self.assertEqual(result.Count.tolist(), [10, 10]) + self.assertAlmostEqual(result.TPM.sum(), 1e6) + self.assertAlmostEqual(result.TPM.iloc[0] / result.TPM.iloc[1], 2) + self.assertAlmostEqual(result.RPKM.iloc[0], 5000000.) + self.assertAlmostEqual(result.RPKM.iloc[1], 2500000.) + self.assertFalse(result.Gene_ID.duplicated().any()) + + def test_duplicate_loci_rejected_instead_of_silently_pooled(self): + counts, gtf = self.fixtures(duplicate=True) + with self.assertRaisesRegex(ValueError, 'duplicate gene IDs'): + helper('abund_calc').abundance_table(counts, gtf) + + def test_zero_counts_are_zero_not_nan(self): + counts, gtf = self.fixtures(zero=True) + result = helper('abund_calc').abundance_table(counts, gtf) + self.assertEqual(result.TPM.tolist(), [0, 0]) + self.assertEqual(result.RPKM.tolist(), [0, 0]) + + def test_gff_conversion_rejects_duplicate_ids_before_mapping(self): + gff = self.put('bad.gff', 'C1\tx\trRNA\t1\t100\t.\t+\t.\tID=RNA1\nC1\tx\trRNA\t201\t300\t.\t+\t.\tID=RNA1\n') + with self.assertRaisesRegex(ValueError, 'Duplicate gene IDs'): + helper('gff2gtf').gff_to_gtf(gff, self.root/'bad.gtf', ['rRNA']) + + def test_child_stderr_is_logged_and_failure_propagates(self): + from metapathways.MetaPathways_tpm import run_logged + log = io.StringIO() + with self.assertRaises(subprocess.CalledProcessError) as error: + run_logged([sys.executable, '-c', 'import sys; print("specific failure", file=sys.stderr); sys.exit(7)'], log) + self.assertEqual(error.exception.returncode, 7) + self.assertIn('specific failure', log.getvalue()) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_rrna_query_features.py b/tests/test_rrna_query_features.py new file mode 100644 index 0000000..9287a40 --- /dev/null +++ b/tests/test_rrna_query_features.py @@ -0,0 +1,46 @@ +import tempfile +import unittest +from unittest.mock import patch +from pathlib import Path +from metapathways.MetaPathways_rRNA_stats_calculator import process_blastout_file, append_taxonomic_information, main + +class RrnaFeatureTests(unittest.TestCase): + def test_coordinate_and_legacy_headers_resolve_subunit(self): + with tempfile.TemporaryDirectory() as directory: + d = Path(directory) + queries = ['ctg:0-250(+)', 'ctg:299-599(-)', '16S_rRNA::old:0-250(+)', 'old23_23S_rRNA'] + (d/'query.fna').write_text(''.join('>'+q+'\n'+'A'*300+'\n' for q in queries)) + (d/'query.gff').write_text('ctg\tbarrnap\trRNA\t1\t250\t.\t+\t.\tName=16S_rRNA\nctg\tbarrnap\trRNA\t300\t599\t.\t-\t.\tName=23S_rRNA\n') + (d/'blast.tsv').write_text(''.join(q+'\tref\t99\t240\t0\t0\t1\t240\t1\t240\t1e-40\t200\n' for q in queries)) + for subunit, expected in [('16S',queries[:1]+queries[2:3]), ('23S',[queries[1],queries[3]])]: + table = {} + process_blastout_file(d/'blast.tsv','test',table,subunit,d/'query.fna',query_gff=d/'query.gff') + self.assertEqual(set(table),set(expected)) + + def test_inclusive_alignment_length_and_reverse_coordinates(self): + with tempfile.TemporaryDirectory() as directory: + db = Path(directory) / 'silva.fasta' + db.write_text('>ref Bacteria;Test species\nAAAA\n') + table = {'forward': [99, 1e-40, 200, 'ref', 1, 180], + 'reverse': [99, 1e-40, 200, 'ref', 180, 1], + 'short': [99, 1e-40, 200, 'ref', 1, 179]} + append_taxonomic_information(str(db), table, dict(length=180, similarity=20, evalue=1e-6, bitscore=50)) + self.assertEqual(table['forward'][6], 'Bacteria;Test species') + self.assertEqual(table['reverse'][6], 'Bacteria;Test species') + self.assertEqual(table['short'][6], '-') + + def test_cli_writes_taxonomy_for_coordinate_only_query(self): + with tempfile.TemporaryDirectory() as directory: + d = Path(directory) + (d/'query.fna').write_text('>ctg:0-250(+)\n'+'A'*250+'\n') + (d/'query.gff').write_text('ctg\tbarrnap\trRNA\t1\t250\t.\t+\t.\tName=16S_rRNA\n') + (d/'blast.tsv').write_text('ctg:0-250(+)\tref\t99\t240\t0\t0\t1\t240\t1\t240\t1e-40\t200\n') + (d/'silva.fasta').write_text('>ref Bacteria;Test species\n'+'A'*250+'\n') + args=['-q',str(d/'query.fna'),'--query-gff',str(d/'query.gff'),'-r','16S','-i',str(d/'blast.tsv'),'-d',str(d/'silva.fasta'),'-o',str(d/'stats.tsv')] + with patch('metapathways.MetaPathways_rRNA_stats_calculator.runBlastCommandrRNA', return_value=0): + main(args) + text=(d/'stats.tsv').read_text() + self.assertIn('#Number of rRNA sequences detected:\t1', text) + self.assertIn('Bacteria;Test species', text) + +if __name__ == '__main__': unittest.main() diff --git a/tests/test_test_data.py b/tests/test_test_data.py new file mode 100644 index 0000000..bd24233 --- /dev/null +++ b/tests/test_test_data.py @@ -0,0 +1,53 @@ +import tempfile +from pathlib import Path +import unittest +from metapathways.test_data import prepare, SEEDS + + +class TestTests(unittest.TestCase): + def test_bundled_data_copies_idempotently_without_indexes(self): + with tempfile.TemporaryDirectory() as directory: + root = prepare(directory) + self.assertEqual(len(list((root/'cami-test/inputs').rglob('*.gz'))), 9) + self.assertEqual(len(list((root/'cami-test/inputs/mag_maps').glob('*.tsv'))), 3) + self.assertEqual(len(list((root/'MPDB').rglob('*.*'))), 1) + self.assertTrue(all((root/'MPDB'/name).is_file() for name in SEEDS)) + file = root/'MPDB/functional/swissprot_test' + mtime = file.stat().st_mtime_ns + prepare(root) + self.assertEqual(file.stat().st_mtime_ns, mtime) + + def test_modified_data_is_preserved_before_any_copy(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + (root/'cami-test').mkdir() + file = root/'cami-test/all.tsv' + file.write_text('my manifest') + with self.assertRaisesRegex(ValueError, 'differs'): + prepare(root) + self.assertEqual(file.read_text(), 'my manifest') + self.assertFalse((root/'MPDB').exists()) + + def test_reference_symlinks_are_rejected(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + (root/'external').mkdir() + (root/'MPDB').symlink_to(root/'external', target_is_directory=True) + with self.assertRaisesRegex(ValueError, 'symbolic link'): + prepare(root) + self.assertEqual(list((root/'external').iterdir()), []) + + def test_build_db_uses_explicit_writable_destination(self): + from unittest.mock import patch + from metapathways import pipeline + with tempfile.TemporaryDirectory() as directory: + root = prepare(directory) + destination = str(root/'MPDB') + with patch('sys.argv', ['metapathways', 'build_db', '--test', '-d', destination]), \ + patch('metapathways.nextflow.launch') as launch: + pipeline.build_db() + tasks, output, args, command = launch.call_args.args + self.assertEqual(output, destination) + self.assertEqual(command, 'build_db') + self.assertTrue(any(destination in x for t in tasks for x in t['inputs'])) + self.assertFalse(any('/regtests/' in x for t in tasks for x in t['outputs']))