diff --git a/PROTOCOLS.yaml b/PROTOCOLS.yaml index f9837bc..9c6188c 100644 --- a/PROTOCOLS.yaml +++ b/PROTOCOLS.yaml @@ -1,5 +1,201 @@ spec_version: 1.0.0 repository: waldronlab/ai-agent-protocols -generated_at: '2026-08-18T13:24:45Z' -protocols: [] +generated_at: '2026-08-18T13:25:22Z' +protocols: +- name: humann4-augmented-clustering + description: Predict ORFs and perform augmented UniRef-compatible protein clustering. + version: 1.0.0 + authors: + - name: Levi Waldron + orcid: 0000-0003-2725-0694 + date: '2026-08-08' + status: draft + license: CC-BY-4.0 + type: atomic + protocol_doi: ~ + repository_doi: ~ + publication_doi: ~ + citation: 10.1016/j.cell.2019.01.001 + upstream_repositories: + - https://github.com/biobakery/humann + - https://github.com/biobakery/metaphlan + database_urls: + - http://huttenhower.sph.harvard.edu/humann_data/uniprot/uniref_annotated/ + - ftp://ftp.uniprot.org/pub/databases/uniprot/uniref/ + protocols_used: [] + key_packages: [] + category: metagenomics + tags: + - humann + - chocophlan + - database + - uniref + - clustering + - mmseqs2 + - diamond + protocol_url: https://raw.githubusercontent.com/waldronlab/ai-agent-protocols/main/protocols/humann4-augmented-clustering/protocol.md +- name: humann4-chocophlan-build + description: Construct the HUMAnN 4 ChocoPhlAn nucleotide pangenome database. + version: 1.0.0 + authors: + - name: Levi Waldron + orcid: 0000-0003-2725-0694 + date: '2026-08-08' + status: draft + license: CC-BY-4.0 + type: atomic + protocol_doi: ~ + repository_doi: ~ + publication_doi: ~ + citation: 10.7554/eLife.65088 + upstream_repositories: + - https://github.com/biobakery/humann + - https://github.com/biobakery/metaphlan + database_urls: + - http://huttenhower.sph.harvard.edu/humann_data/chocophlan/ + - http://cmprod1.cibio.unitn.it/databases/chocophlan/ + protocols_used: [] + key_packages: [] + category: metagenomics + tags: + - humann + - chocophlan + - database + - pangenome + - bowtie2 + protocol_url: https://raw.githubusercontent.com/waldronlab/ai-agent-protocols/main/protocols/humann4-chocophlan-build/protocol.md +- name: humann4-database-build + description: End-to-end composite pipeline for constructing HUMAnN 4 reference databases + from MetaPhlAn 4.2 SGBs. + version: 1.0.0 + authors: + - name: Levi Waldron + orcid: 0000-0003-2725-0694 + date: '2026-08-18' + status: draft + license: CC-BY-4.0 + type: composite + protocol_doi: ~ + repository_doi: ~ + publication_doi: ~ + citation: 10.7554/eLife.65088 + upstream_repositories: + - https://github.com/biobakery/humann + - https://github.com/biobakery/metaphlan + database_urls: + - http://huttenhower.sph.harvard.edu/humann_data/chocophlan/ + - http://huttenhower.sph.harvard.edu/humann_data/uniprot/uniref_annotated/ + - http://huttenhower.sph.harvard.edu/humann_data/mapping/ + protocols_used: + - name: humann4-sgb-aggregation + repository: waldronlab/ai-agent-protocols + version: 1.0.0 + - name: humann4-augmented-clustering + repository: waldronlab/ai-agent-protocols + version: 1.0.0 + - name: humann4-chocophlan-build + repository: waldronlab/ai-agent-protocols + version: 1.0.0 + - name: humann4-translated-search-build + repository: waldronlab/ai-agent-protocols + version: 1.0.0 + - name: humann4-utility-mapping + repository: waldronlab/ai-agent-protocols + version: 1.0.0 + key_packages: [] + category: metagenomics + tags: + - humann + - chocophlan + - database + - pipeline + - composite + protocol_url: https://raw.githubusercontent.com/waldronlab/ai-agent-protocols/main/protocols/humann4-database-build/protocol.md +- name: humann4-sgb-aggregation + description: Aggregate and subsample isolate genomes and MAGs for MetaPhlAn 4.2 + SGBs. + version: 1.0.0 + authors: + - name: Levi Waldron + orcid: 0000-0003-2725-0694 + date: '2026-08-08' + status: draft + license: CC-BY-4.0 + type: atomic + protocol_doi: ~ + repository_doi: ~ + publication_doi: ~ + citation: 10.1016/j.cell.2019.01.001 + upstream_repositories: + - https://github.com/biobakery/metaphlan + - https://github.com/biobakery/panphlan + database_urls: + - http://cmprod1.cibio.unitn.it/databases/MetaPhlAn/ + - http://cmprod1.cibio.unitn.it/databases/PanPhlAn/ + protocols_used: [] + key_packages: [] + category: metagenomics + tags: + - humann + - chocophlan + - database + - sgb + - metaphlan + protocol_url: https://raw.githubusercontent.com/waldronlab/ai-agent-protocols/main/protocols/humann4-sgb-aggregation/protocol.md +- name: humann4-translated-search-build + description: Compile the HUMAnN 4 translated search database. + version: 1.0.0 + authors: + - name: Levi Waldron + orcid: 0000-0003-2725-0694 + date: '2026-08-08' + status: draft + license: CC-BY-4.0 + type: atomic + protocol_doi: ~ + repository_doi: ~ + publication_doi: ~ + citation: 10.7554/eLife.65088 + upstream_repositories: https://github.com/biobakery/humann + database_urls: http://huttenhower.sph.harvard.edu/humann_data/uniprot/uniref_annotated/ + protocols_used: [] + key_packages: [] + category: metagenomics + tags: + - humann + - database + - translated-search + - diamond + - uniref + protocol_url: https://raw.githubusercontent.com/waldronlab/ai-agent-protocols/main/protocols/humann4-translated-search-build/protocol.md +- name: humann4-utility-mapping + description: Generate HUMAnN 4 utility mapping databases for functional annotation. + version: 1.0.0 + authors: + - name: Levi Waldron + orcid: 0000-0003-2725-0694 + date: '2026-08-08' + status: draft + license: CC-BY-4.0 + type: atomic + protocol_doi: ~ + repository_doi: ~ + publication_doi: ~ + citation: 10.7554/eLife.65088 + upstream_repositories: + - https://github.com/biobakery/humann + - https://github.com/eggnogdb/eggnog-mapper + database_urls: + - http://huttenhower.sph.harvard.edu/humann_data/mapping/ + - http://huttenhower.sph.harvard.edu/humann_data/legacy_dbs/ + protocols_used: [] + key_packages: [] + category: metagenomics + tags: + - humann + - database + - functional-annotation + - eggnog-mapper + - uniprot + protocol_url: https://raw.githubusercontent.com/waldronlab/ai-agent-protocols/main/protocols/humann4-utility-mapping/protocol.md diff --git a/README.md b/README.md index 861b796..94839e6 100644 --- a/README.md +++ b/README.md @@ -2,6 +2,26 @@ This repository serves as both the central federation registry and a host for AI agent-compatible scientific protocols for the Waldron Lab. +## What is an AI Agent Protocol? + +An "AI Agent Protocol" is related to but different than an [AI Skill](https://github.com/bioconductor/ai-agent-skills). + +* **AI Skill**: A specific capability given to an AI agent (e.g., how to query a specific biological database, or how to use a particular R package). Skills teach the AI *how* to perform specific actions. +* **AI Agent Protocol**: A scientific workflow, experimental plan, or analytical pipeline designed to be executed by or in collaboration with an AI agent. Protocols in this registry are **designed around provenance to published methods**. Protocols: + - are formal records providing citation both to primary scientific literature and to publication of protocols (#4). Atomic protocols have a single purpose with a single citation to primary literature; composite protocols may be composed of multiple atomic protocols. + - will record **human reviews** (#2) + - will support **formal unit tests/benchmarks** (#3) to verify correct execution by different AI agents and models. + +## Utility and Core Use Cases + +Some likely use cases include: + +1. **Constraining Coding Agents to Established Methods**: Forces AI agents to adhere strictly to vetted, peer-reviewed analytical protocols rather than drifting, inventing parameters, or inventing plausible but untested methodology during automated script generation. Protocols are expected to create more uniform behavior by different AI agents and models. +2. **Cross-Language and Pipeline Translation**: Serves as an unambiguous English-language specification for translating computational workflows across programming languages and pipeline frameworks (e.g., Nextflow ↔ Snakemake, R ↔ Python) without losing domain-specific logic or parameter integrity. +3. **Discrepancy Auditing (Paper vs. Code vs. Protocol)**: Acts as an explicit benchmark to systematically detect inconsistencies between high-level descriptions in published manuscript Methods sections, formal protocol documentation, and actual codebase implementations. Protocols should be easier for people with domain expertise to review than codebase or even Methods sections which are less structured, can be split across main manuscript and supplementary materials, and may lack necessary details for full implementation. +4. **Filling the Methodological Reproducibility Gap**: Provides the granular operational, environment, and parameter-level details that traditional journal Methods sections often omit, facilitating computational reproducibility with less susceptbility to bitrot or dependency issues. +5. **A federated registry of AI agent-compatible protocols**: This repository serves as a central registry for AI agent-compatible protocols, designed to allow researchers to independently create their own protocol repositories and federate them into this central registry. + ## License This repository is dual-licensed: diff --git a/protocols/humann4-augmented-clustering/protocol.md b/protocols/humann4-augmented-clustering/protocol.md new file mode 100644 index 0000000..f46d034 --- /dev/null +++ b/protocols/humann4-augmented-clustering/protocol.md @@ -0,0 +1,75 @@ +--- +name: humann4-augmented-clustering +description: Predict ORFs and perform augmented UniRef-compatible protein clustering. +version: 1.0.0 +authors: + - name: Levi Waldron + orcid: 0000-0003-2725-0694 +date: 2026-08-08 +status: draft +license: CC-BY-4.0 +type: atomic + +protocol_doi: ~ +repository_doi: ~ +publication_doi: ~ + +citation: "10.1016/j.cell.2019.01.001" + +upstream_repositories: + - "https://github.com/biobakery/humann" + - "https://github.com/biobakery/metaphlan" + +database_urls: + - "http://huttenhower.sph.harvard.edu/humann_data/uniprot/uniref_annotated/" + - "ftp://ftp.uniprot.org/pub/databases/uniprot/uniref/" + +protocols_used: [] +key_packages: [] +category: metagenomics +tags: [humann, chocophlan, database, uniref, clustering, mmseqs2, diamond] +--- + +# HUMAnN 4 Augmented Protein Clustering + +This protocol outlines how to process the representative genomes for MetaPhlAn 4.2 SGBs to predict proteins and cluster them in compatibility with the UniRef90/UniRef50 ontologies. + +## Materials + +- **Software & Repositories:** + - `Prodigal` ([hyattpd/Prodigal](https://github.com/hyattpd/Prodigal)) — Fast prokaryotic gene recognition and ORF prediction. + - `DIAMOND` ([bbuchfink/diamond](https://github.com/bbuchfink/diamond)) — Accelerated protein alignment against UniRef. + - `MMseqs2` ([soedinglab/MMseqs2](https://github.com/soedinglab/MMseqs2)) — Ultra-fast de novo protein clustering. + - `humann` ([biobakery/humann](https://github.com/biobakery/humann)) — Upstream database utilities and configurations. +- **Databases & Reference Data:** + - Previous UniRef Database Builds ([HUMAnN Data Server](http://huttenhower.sph.harvard.edu/humann_data/uniprot/uniref_annotated/)) — Baseline UniRef50/UniRef90 releases used in HUMAnN 3. + - UniProt UniRef Releases ([UniProt FTP](ftp://ftp.uniprot.org/pub/databases/uniprot/uniref/)) — Official UniRef90 and UniRef50 FASTA releases. + +## Steps + +### Step 1: Open Reading Frame (ORF) Prediction + +Run `Prodigal` on all representative SGB genomes (as output by `humann4-sgb-aggregation`) to predict ORFs. Extract both the nucleotide and translated amino acid sequences. + +### Step 2: Intermediate UniRef Alignment + +Use `DIAMOND` to align all predicted protein sequences against the standard `UniRef90` and `UniRef50` databases. + +For any ORF that matches an existing UniRef90 cluster (meeting the minimum thresholds of ≥90% identity and ≥80% coverage), assign it to that established cluster. + +### Step 3: De Novo Clustering of Unmapped ORFs + +For ORFs that fail to map to standard UniRef clusters, perform *de novo* clustering using `MMseqs2`. Cluster these sequences at a 90% identity threshold to create novel protein families. + +Name these newly generated clusters using the convention: `SGB_Ref90_XXXX` (where XXXX is a unique identifier). + +### Step 4: Generate ORF Mapping File + +Compile a comprehensive TSV mapping file linking every original ORF ID (from all SGBs) to its assigned `UniRef90` ID or its novel `SGB_Ref90` ID. + +## Notes + +- **HPC Cluster Scalability:** Predicting ORFs and performing all-against-UniRef alignment across millions of SGB genes requires high-memory cluster nodes and multi-threaded MMseqs2/DIAMOND jobs. +- This protocol bridges the gap between characterized UniProt proteins and novel ORFs discovered via large-scale MAG assembly in MetaPhlAn 4.2. + + diff --git a/protocols/humann4-chocophlan-build/protocol.md b/protocols/humann4-chocophlan-build/protocol.md new file mode 100644 index 0000000..702d3f4 --- /dev/null +++ b/protocols/humann4-chocophlan-build/protocol.md @@ -0,0 +1,67 @@ +--- +name: humann4-chocophlan-build +description: Construct the HUMAnN 4 ChocoPhlAn nucleotide pangenome database. +version: 1.0.0 +authors: + - name: Levi Waldron + orcid: 0000-0003-2725-0694 +date: 2026-08-08 +status: draft +license: CC-BY-4.0 +type: atomic + +protocol_doi: ~ +repository_doi: ~ +publication_doi: ~ + +citation: "10.7554/eLife.65088" + +upstream_repositories: + - "https://github.com/biobakery/humann" + - "https://github.com/biobakery/metaphlan" + +database_urls: + - "http://huttenhower.sph.harvard.edu/humann_data/chocophlan/" + - "http://cmprod1.cibio.unitn.it/databases/chocophlan/" + +protocols_used: [] +key_packages: [] +category: metagenomics +tags: [humann, chocophlan, database, pangenome, bowtie2] +--- + +# HUMAnN 4 ChocoPhlAn Build + +This protocol constructs the ChocoPhlAn 4 nucleotide pangenome database, formatting it for `bowtie2` mapping in the HUMAnN 4 pipeline. + +## Materials + +- **Software & Repositories:** + - `bowtie2` ([BenLangmead/bowtie2](https://github.com/BenLangmead/bowtie2)) — Ultrafast nucleotide aligner and index builder. + - `humann` ([biobakery/humann](https://github.com/biobakery/humann)) — Primary consumer and database management scripts (`humann/tools/`). +- **Databases & Reference Data:** + - Previous ChocoPhlAn Database Builds ([HUMAnN Data Server](http://huttenhower.sph.harvard.edu/humann_data/chocophlan/)) — Full and demo nucleotide pangenome releases for HUMAnN 3. + - Segata Lab ChocoPhlAn Database ([Segata Lab Server](http://cmprod1.cibio.unitn.it/databases/chocophlan/)) — Upstream microbial pangenome repository. + +## Steps + +### Step 1: Pan-proteome Compilation + +For each MetaPhlAn 4.2 SGB, identify its pan-proteome: the comprehensive set of all unique `UniRef90` and `SGB_Ref90` protein families present across the SGB's representative genomes. + +### Step 2: Nucleotide Sequence Selection + +For every protein family present in a given SGB's pan-proteome, select one representative nucleotide ORF sequence from the SGB's member genomes. + +### Step 3: FASTA Concatenation and Indexing + +Concatenate the representative nucleotide sequences into a single SGB-specific pangenome FASTA file. + +Build a `bowtie2` index for each SGB pangenome FASTA to finalize the ChocoPhlAn database structure. + +## Notes + +- **Run-Time Dynamic Indexing:** The resulting `bowtie2` indexes are the primary mapping targets for the initial nucleotide-level search step in HUMAnN. At runtime, HUMAnN extracts and combines only the pangenomes of species identified in the sample's MetaPhlAn profile. +- Databases generated by this protocol can be installed locally or configured using `humann_config --update database_folders chocophlan `. + + diff --git a/protocols/humann4-database-build/protocol.md b/protocols/humann4-database-build/protocol.md new file mode 100644 index 0000000..e3ee2bb --- /dev/null +++ b/protocols/humann4-database-build/protocol.md @@ -0,0 +1,84 @@ +--- +name: humann4-database-build +description: End-to-end composite pipeline for constructing HUMAnN 4 reference databases from MetaPhlAn 4.2 SGBs. +version: 1.0.0 +authors: + - name: Levi Waldron + orcid: 0000-0003-2725-0694 +date: 2026-08-18 +status: draft +license: CC-BY-4.0 +type: composite + +protocol_doi: ~ +repository_doi: ~ +publication_doi: ~ + +citation: "10.7554/eLife.65088" + +upstream_repositories: + - "https://github.com/biobakery/humann" + - "https://github.com/biobakery/metaphlan" + +database_urls: + - "http://huttenhower.sph.harvard.edu/humann_data/chocophlan/" + - "http://huttenhower.sph.harvard.edu/humann_data/uniprot/uniref_annotated/" + - "http://huttenhower.sph.harvard.edu/humann_data/mapping/" + +protocols_used: + - name: humann4-sgb-aggregation + repository: waldronlab/ai-agent-protocols + version: 1.0.0 + - name: humann4-augmented-clustering + repository: waldronlab/ai-agent-protocols + version: 1.0.0 + - name: humann4-chocophlan-build + repository: waldronlab/ai-agent-protocols + version: 1.0.0 + - name: humann4-translated-search-build + repository: waldronlab/ai-agent-protocols + version: 1.0.0 + - name: humann4-utility-mapping + repository: waldronlab/ai-agent-protocols + version: 1.0.0 + +key_packages: [] +category: metagenomics +tags: [humann, chocophlan, database, pipeline, composite] +--- + +# HUMAnN 4 Reference Database Construction Pipeline + +This composite protocol coordinates the complete end-to-end generation of reference databases required by HUMAnN 4, incorporating Species-level Genome Bins (SGBs) from MetaPhlAn 4.2. + +## Materials + +- **Software & Repositories:** + - `humann` ([biobakery/humann](https://github.com/biobakery/humann)) — Downstream functional profiler and database management commands. + - `metaphlan` ([biobakery/metaphlan](https://github.com/biobakery/metaphlan)) — Species-level genome bin taxonomy definitions. +- **Databases & Baseline Reference Data:** + - HUMAnN 3 Reference Data Archive ([HUMAnN Server](http://huttenhower.sph.harvard.edu/humann_data/)) — Previous versions of ChocoPhlAn, UniRef DIAMOND indexes, and utility mappings. + +## Steps + +This composite workflow executes five constituent atomic protocols in sequence: + +### Step 1: SGB Genome Aggregation & Subsampling +Execute `humann4-sgb-aggregation` to retrieve representative isolate genomes and MAGs for all MetaPhlAn 4.2 SGBs, subsampling overrepresented SGBs (max 100 genomes) via Mash distance. + +### Step 2: Augmented UniRef Protein Clustering +Execute `humann4-augmented-clustering` to predict ORFs with Prodigal, map known sequences to UniRef90/UniRef50 via DIAMOND, and cluster novel unmapped ORFs into `SGB_Ref90_XXXX` families using MMseqs2. + +### Step 3: ChocoPhlAn Nucleotide Pangenome Database Build +Execute `humann4-chocophlan-build` to extract representative nucleotide sequences for each SGB pan-proteome and compile SGB-specific Bowtie2 indexes. + +### Step 4: Translated Search Database Build +Execute `humann4-translated-search-build` to combine official UniRef90 with novel SGB clusters and generate the comprehensive DIAMOND index (`augmented_uniref90.dmnd`). + +### Step 5: Functional Utility Mapping Compilation +Execute `humann4-utility-mapping` to extract UniProtKB annotations for standard clusters, generate eggNOG-mapper orthology/pathway predictions for novel SGB clusters, and compile unified TSV cross-reference mapping tables (KO, EC, GO, Pfam, MetaCyc). + +## Notes + +- **HPC Execution:** This end-to-end workflow is designed for high-performance computing clusters with SLURM/PBS orchestration or workflow engines (e.g., Snakemake/Nextflow). +- **Provenance & Citations:** Executing this composite pipeline incorporates primary literature methodology citations from Pasolli et al. (2019) (*Cell*) and Beghini et al. (2021) (*eLife*). diff --git a/protocols/humann4-sgb-aggregation/protocol.md b/protocols/humann4-sgb-aggregation/protocol.md new file mode 100644 index 0000000..b09c653 --- /dev/null +++ b/protocols/humann4-sgb-aggregation/protocol.md @@ -0,0 +1,65 @@ +--- +name: humann4-sgb-aggregation +description: Aggregate and subsample isolate genomes and MAGs for MetaPhlAn 4.2 SGBs. +version: 1.0.0 +authors: + - name: Levi Waldron + orcid: 0000-0003-2725-0694 +date: 2026-08-08 +status: draft +license: CC-BY-4.0 +type: atomic + +protocol_doi: ~ +repository_doi: ~ +publication_doi: ~ + +citation: "10.1016/j.cell.2019.01.001" + +upstream_repositories: + - "https://github.com/biobakery/metaphlan" + - "https://github.com/biobakery/panphlan" + +database_urls: + - "http://cmprod1.cibio.unitn.it/databases/MetaPhlAn/" + - "http://cmprod1.cibio.unitn.it/databases/PanPhlAn/" + +protocols_used: [] +key_packages: [] +category: metagenomics +tags: [humann, chocophlan, database, sgb, metaphlan] +--- + +# HUMAnN 4 SGB Aggregation + +This protocol details the retrieval and aggregation of Species-level Genome Bins (SGBs) for compatibility with MetaPhlAn 4.2. + +## Materials + +- **Software & Repositories:** + - `Mash` ([marbl/Mash](https://github.com/marbl/Mash)) — Fast genome distance estimation and min-hashing. + - `metaphlan` ([biobakery/metaphlan](https://github.com/biobakery/metaphlan)) — Species-level genome bin (SGB) taxonomy and marker definitions. + - `panphlan` ([biobakery/panphlan](https://github.com/biobakery/panphlan)) — Reference implementation for Mash-distance subsampling logic. +- **Databases & Reference Data:** + - MetaPhlAn 4.2 SGB Genome Catalog ([Segata Lab Database Server](http://cmprod1.cibio.unitn.it/databases/MetaPhlAn/)) — Reference isolate genomes and MAGs. + +## Steps + +### Step 1: Retrieve SGB Genomes + +Download all representative isolate genomes and Metagenome-Assembled Genomes (MAGs) corresponding to the MetaPhlAn 4.2 SGB definitions. Ensure all FASTA files are quality filtered (e.g. using CheckM). + +### Step 2: Subsample Overrepresented SGBs + +For SGBs containing more than 100 member genomes, subsample the set down to a maximum of 100 representative genomes. +This is done to limit the computational scale of subsequent clustering steps. +Use `Mash` to sketch and calculate pairwise distances between all genomes in the SGB. +Select a representative subset that maximizes the Mash distances to preserve the maximum genomic diversity within the SGB. + +## Notes + +- **Computational Context:** This is the first step in the HUMAnN 4 database generation pipeline. End-to-end execution across tens of thousands of SGBs is typically executed in parallel on high-performance computing (HPC) clusters. +- SGBs with fewer than 100 members do not require subsampling. + + + diff --git a/protocols/humann4-translated-search-build/protocol.md b/protocols/humann4-translated-search-build/protocol.md new file mode 100644 index 0000000..d2d22db --- /dev/null +++ b/protocols/humann4-translated-search-build/protocol.md @@ -0,0 +1,60 @@ +--- +name: humann4-translated-search-build +description: Compile the HUMAnN 4 translated search database. +version: 1.0.0 +authors: + - name: Levi Waldron + orcid: 0000-0003-2725-0694 +date: 2026-08-08 +status: draft +license: CC-BY-4.0 +type: atomic + +protocol_doi: ~ +repository_doi: ~ +publication_doi: ~ + +citation: "10.7554/eLife.65088" + +upstream_repositories: + - "https://github.com/biobakery/humann" + +database_urls: + - "http://huttenhower.sph.harvard.edu/humann_data/uniprot/uniref_annotated/" + +protocols_used: [] +key_packages: [] +category: metagenomics +tags: [humann, database, translated-search, diamond, uniref] +--- + +# HUMAnN 4 Translated Search Build + +This protocol compiles the comprehensive translated search database utilized by HUMAnN 4 for alignment of reads that fail to map during the nucleotide pangenome search. + +## Materials + +- **Software & Repositories:** + - `DIAMOND` ([bbuchfink/diamond](https://github.com/bbuchfink/diamond)) — High-throughput translated DNA/protein aligner. + - `humann` ([biobakery/humann](https://github.com/biobakery/humann)) — Primary consumer and configuration scripts. +- **Databases & Reference Data:** + - Previous UniRef DIAMOND Databases ([HUMAnN Data Server](http://huttenhower.sph.harvard.edu/humann_data/uniprot/uniref_annotated/)) — Pre-indexed DIAMOND reference databases for HUMAnN 3 (`uniref90_diamond`, `uniref50_diamond`). + +## Steps + +### Step 1: Database Concatenation + +Merge the official UniRef90 FASTA database with the newly generated FASTA containing representative sequences of all novel `SGB_Ref90` clusters. + +This results in a single, comprehensive protein database encompassing both characterized UniProt sequences and novel MetaPhlAn 4.2 SGB ORFs. + +### Step 2: DIAMOND Indexing + +Build a `DIAMOND` index (e.g., `augmented_uniref90.dmnd`) from the merged FASTA file. + +## Notes + +- **Fallback Translated Search:** This database serves as the fallback translated search target in the HUMAnN pipeline, allowing functional profiling of reads that do not map to the specific ChocoPhlAn pangenomes of identified species. +- Configured in HUMAnN via `humann_config --update database_folders protein `. + + diff --git a/protocols/humann4-utility-mapping/protocol.md b/protocols/humann4-utility-mapping/protocol.md new file mode 100644 index 0000000..a3c3990 --- /dev/null +++ b/protocols/humann4-utility-mapping/protocol.md @@ -0,0 +1,71 @@ +--- +name: humann4-utility-mapping +description: Generate HUMAnN 4 utility mapping databases for functional annotation. +version: 1.0.0 +authors: + - name: Levi Waldron + orcid: 0000-0003-2725-0694 +date: 2026-08-08 +status: draft +license: CC-BY-4.0 +type: atomic + +protocol_doi: ~ +repository_doi: ~ +publication_doi: ~ + +citation: "10.7554/eLife.65088" + +upstream_repositories: + - "https://github.com/biobakery/humann" + - "https://github.com/eggnogdb/eggnog-mapper" + +database_urls: + - "http://huttenhower.sph.harvard.edu/humann_data/mapping/" + - "http://huttenhower.sph.harvard.edu/humann_data/legacy_dbs/" + +protocols_used: [] +key_packages: [] +category: metagenomics +tags: [humann, database, functional-annotation, eggnog-mapper, uniprot] +--- + +# HUMAnN 4 Utility Mapping Generation + +This protocol outlines the generation of the TSV mapping files used by HUMAnN 4 to translate protein cluster abundances into functional ontology abundances (e.g. KEGG Orthology, GO terms, EC numbers). + +## Materials + +- **Software & Repositories:** + - `eggNOG-mapper` ([eggnogdb/eggnog-mapper](https://github.com/eggnogdb/eggnog-mapper)) — Fast functional annotation and orthology assignment. + - `humann` ([biobakery/humann](https://github.com/biobakery/humann)) — Downstream regrouping/renaming utilities (`humann_regroup_table`, `humann_rename_table`). +- **Databases & Reference Data:** + - Previous Utility Mapping Databases ([HUMAnN Data Server](http://huttenhower.sph.harvard.edu/humann_data/mapping/)) — Baseline mapping TSVs used in HUMAnN 3 (mapping UniRef50/UniRef90 to KO, EC, GO, Pfam, MetaCyc). + - UniProtKB Complete Knowledgebase XML ([UniProt FTP](ftp://ftp.uniprot.org/pub/databases/uniprot/current_release/knowledgebase/complete/)) — Source annotations for standard UniRef clusters. + +## Steps + +### Step 1: Standard Cluster Annotation via Lookup + +Parse the UniProtKB XML to extract functional annotations (GO, KEGG KO, Pfam, EC, and EggNOG) for all standard UniRef90 clusters. +Direct execution of InterProScan is avoided here due to computational complexity; mappings are derived directly from the pre-computed cross-references in UniProt. + +### Step 2: Primary Orthology Generation for Novel Clusters + +Deploy `eggNOG-mapper` on the representative sequences of the novel `SGB_Ref90` clusters. +This provides automated mapping to KEGG Orthology (KO), KEGG BRITE, COGs, and GO categories based on orthologous group placement. + +### Step 3: Targeted Biochemical Annotation (Optional) + +If highly specific biochemical annotation is required, deploy targeted pipelines (such as `dbCAN2` for carbohydrate-active enzymes, or `antiSMASH` for biosynthetic gene clusters) on the novel SGB sequence catalogs. + +### Step 4: Compile HUMAnN Mapping TSVs + +Combine the parsed UniProt annotations (Step 1) with the eggNOG-mapper predictions (Step 2/3) into the unified TSV format required by HUMAnN. +Generate separate mapping files for each ontology (e.g. `map_uniref90_to_ko.txt`, `map_uniref90_to_ec.txt`), seamlessly integrating both standard and novel clusters. + +## Notes + +- **Integration with bioBakery:** This is the final step in the HUMAnN 4 database generation workflow. The resulting TSV tables are placed in the HUMAnN utility mapping directory (`humann_config --update database_folders utility_mapping `) and are utilized by downstream utilities like `humann_regroup_table`. + +