diff --git a/.Rbuildignore b/.Rbuildignore index f2e9c0d..eb7a420 100644 --- a/.Rbuildignore +++ b/.Rbuildignore @@ -5,7 +5,6 @@ ^CRAN-SUBMISSION$ ^cran-comments\.md$ ^inst/extdata/testing -^TODO.md$ ^.*\.DS_Store$ .DS_Store .venv_altdoc @@ -13,8 +12,8 @@ ^altdoc$ ^_pkgdown\.yml$ ^pkgdown$ -(^|/)wip-.*$ ^.venv_altdoc$ ^edit-bench$ ^immdata-test$ ^immundata-quick-start$ +^dev$ diff --git a/.github/ISSUE_TEMPLATE/2-enhancement.yml b/.github/ISSUE_TEMPLATE/2-enhancement.yml index 82a8626..7b7e380 100644 --- a/.github/ISSUE_TEMPLATE/2-enhancement.yml +++ b/.github/ISSUE_TEMPLATE/2-enhancement.yml @@ -3,17 +3,6 @@ description: "Propose a new feature or an improvement." labels: - enhancement body: -- type: dropdown - id: kind - attributes: - label: What kind of feature would you like to request? - options: - - 'New data processing, filtering, wrangling methods you are missing' - - 'Request enhancement to input/output - format processing, reading data, writing data to the disk' - - 'Improve existing functions, e.g., add additional function parameters, change how a function works, change default values' - - 'Other - please describe it in the ticket body' - validations: - required: true - type: textarea id: description attributes: diff --git a/.github/ISSUE_TEMPLATE/3-documentation.yml b/.github/ISSUE_TEMPLATE/3-documentation.yml index e8a6f63..45f2450 100644 --- a/.github/ISSUE_TEMPLATE/3-documentation.yml +++ b/.github/ISSUE_TEMPLATE/3-documentation.yml @@ -3,19 +3,10 @@ description: "Propose a tutorial for a specific use case or new documentation." labels: - documentation body: -- type: dropdown - id: kind - attributes: - label: What kind of documentation would you like to request? - options: - - 'New tutorial for a specific use case' - - 'Improved documentation or error message' - - 'Other - please describe it in the ticket body' - validations: - required: true - type: textarea id: description attributes: - label: Please describe your idea or what you want to achieve + label: What type of documentation would you like to see on the website? + description: Interested in a tutorial? More details on how a specific method works? More examples? Something else entirely? validations: required: true diff --git a/.github/ISSUE_TEMPLATE/4-question.yml b/.github/ISSUE_TEMPLATE/4-question.yml index 818fc6d..34a41f3 100644 --- a/.github/ISSUE_TEMPLATE/4-question.yml +++ b/.github/ISSUE_TEMPLATE/4-question.yml @@ -6,7 +6,7 @@ body: - type: textarea id: description attributes: - label: Question - description: Describe what you want to know. + label: Your question + description: Describe what you want to know. E.g., how to analyse a specific data, how to do X, when a method will be available, etc. validations: required: true diff --git a/.github/workflows/altdoc-dev.yaml b/.github/workflows/altdoc-dev.yaml new file mode 100644 index 0000000..c178d77 --- /dev/null +++ b/.github/workflows/altdoc-dev.yaml @@ -0,0 +1,66 @@ +name: Build development documentation + +on: + push: + branches: [dev] + workflow_dispatch: + +concurrency: + group: altdoc-pages + cancel-in-progress: false + +permissions: read-all + +jobs: + altdoc-dev: + if: github.ref == 'refs/heads/dev' + runs-on: ubuntu-latest + env: + GITHUB_PAT: ${{ secrets.GITHUB_TOKEN }} + R_KEEP_PKG_SOURCE: yes + permissions: + contents: write + + steps: + - uses: actions/checkout@v4 + + - uses: r-lib/actions/setup-pandoc@v2 + + - uses: quarto-dev/quarto-actions/setup@v2 + + - uses: r-lib/actions/setup-r@v2 + with: + use-public-rspm: true + + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + cache: pip + cache-dependency-path: altdoc/requirements.txt + + - name: Install MkDocs + run: | + python -m venv .venv_altdoc + .venv_altdoc/bin/python -m pip install --requirement altdoc/requirements.txt + + - uses: r-lib/actions/setup-r-dependencies@v2 + with: + extra-packages: local::. + needs: website + + - name: Configure development preview + run: | + sed -i 's#https://immunomind.github.io/immundata/#https://immunomind.github.io/immundata/dev/#g' altdoc/mkdocs.yml altdoc/pkgdown.yml README.md + sed -i 's/^site_name: immundata$/site_name: immundata dev/' altdoc/mkdocs.yml + + - name: Build site + run: altdoc::render_docs(freeze = FALSE, parallel = FALSE, verbose = TRUE) + shell: Rscript {0} + + - name: Deploy development preview + uses: JamesIves/github-pages-deploy-action@v4 + with: + clean: true + branch: gh-pages + folder: docs + target-folder: dev diff --git a/.github/workflows/altdoc.yaml b/.github/workflows/altdoc.yaml new file mode 100644 index 0000000..e441685 --- /dev/null +++ b/.github/workflows/altdoc.yaml @@ -0,0 +1,68 @@ +name: Build documentation + +on: + push: + branches: [main, master] + pull_request: + release: + types: [published] + workflow_dispatch: + +concurrency: + group: ${{ github.event_name == 'pull_request' && format('altdoc-pr-{0}', github.event.pull_request.number) || 'altdoc-pages' }} + cancel-in-progress: false + +permissions: read-all + +jobs: + altdoc: + runs-on: ubuntu-latest + env: + GITHUB_PAT: ${{ secrets.GITHUB_TOKEN }} + R_KEEP_PKG_SOURCE: yes + permissions: + contents: write + + steps: + - uses: actions/checkout@v4 + + - uses: r-lib/actions/setup-pandoc@v2 + + - uses: quarto-dev/quarto-actions/setup@v2 + + - uses: r-lib/actions/setup-r@v2 + with: + use-public-rspm: true + + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + cache: pip + cache-dependency-path: altdoc/requirements.txt + + - name: Install MkDocs + run: | + python -m venv .venv_altdoc + .venv_altdoc/bin/python -m pip install --requirement altdoc/requirements.txt + + - uses: r-lib/actions/setup-r-dependencies@v2 + with: + extra-packages: local::. + needs: website + + - name: Build site + run: altdoc::render_docs(freeze = FALSE, parallel = FALSE, verbose = TRUE) + shell: Rscript {0} + + - name: Deploy to GitHub Pages + if: >- + github.event_name != 'pull_request' && + (github.ref == 'refs/heads/main' || + github.ref == 'refs/heads/master' || + startsWith(github.ref, 'refs/tags/')) + uses: JamesIves/github-pages-deploy-action@v4 + with: + clean: true + clean-exclude: dev + branch: gh-pages + folder: docs diff --git a/.github/workflows/pkgdown.yaml b/.github/workflows/pkgdown.yaml deleted file mode 100644 index bfc9f4d..0000000 --- a/.github/workflows/pkgdown.yaml +++ /dev/null @@ -1,49 +0,0 @@ -# Workflow derived from https://github.com/r-lib/actions/tree/v2/examples -# Need help debugging build failures? Start at https://github.com/r-lib/actions#where-to-find-help -on: - push: - branches: [main, master] - pull_request: - release: - types: [published] - workflow_dispatch: - -name: pkgdown.yaml - -permissions: read-all - -jobs: - pkgdown: - runs-on: ubuntu-latest - # Only restrict concurrency for non-PR jobs - concurrency: - group: pkgdown-${{ github.event_name != 'pull_request' || github.run_id }} - env: - GITHUB_PAT: ${{ secrets.GITHUB_TOKEN }} - permissions: - contents: write - steps: - - uses: actions/checkout@v4 - - - uses: r-lib/actions/setup-pandoc@v2 - - - uses: r-lib/actions/setup-r@v2 - with: - use-public-rspm: true - - - uses: r-lib/actions/setup-r-dependencies@v2 - with: - extra-packages: any::pkgdown, local::. - needs: website - - - name: Build site - run: pkgdown::build_site_github_pages(new_process = FALSE, install = FALSE) - shell: Rscript {0} - - - name: Deploy to GitHub pages 🚀 - if: github.event_name != 'pull_request' - uses: JamesIves/github-pages-deploy-action@v4.5.0 - with: - clean: false - branch: gh-pages - folder: docs diff --git a/.gitignore b/.gitignore index ea8ebf0..b3f2a51 100644 --- a/.gitignore +++ b/.gitignore @@ -15,3 +15,5 @@ immdata-test altdoc/freeze.rds docs wip-* +annotations.parquet +metadata.json diff --git a/DESCRIPTION b/DESCRIPTION index e3a88c3..48dd802 100644 --- a/DESCRIPTION +++ b/DESCRIPTION @@ -1,6 +1,6 @@ Package: immundata Title: A Unified Data Layer for Large-Scale Single-Cell, Spatial and Bulk Immunomics -Version: 0.0.5 +Version: 0.1.0 Authors@R: person("Vadim I.", "Nazarov", , "support@immunomind.com", role = c("aut", "cre"), comment = c(ORCID = "0000-0003-3659-2709")) @@ -14,18 +14,16 @@ URL: https://immunomind.github.io/docs/, https://github.com/immunomind/immundata BugReports: https://github.com/immunomind/immundata/issues Encoding: UTF-8 Roxygen: list(markdown = TRUE) -RoxygenNote: 7.3.3 Depends: R (>= 4.1.0), - dplyr, - duckplyr (>= 1.1.0) + dplyr (>= 1.2.1), + duckplyr (>= 1.2.1) Imports: checkmate, cli, dbplyr, - ggplot2, glue, - jsonlite (>= 2.0.0), + jsonlite, lifecycle, R6, readr, @@ -34,8 +32,11 @@ Imports: tools, utils Suggests: + duckdb, rmarkdown, testthat (>= 3.0.0), Seurat +Config/Needs/website: altdoc Config/testthat/edition: 3 Config/testthat/parallel: true +Config/roxygen2/version: 8.1.0 diff --git a/NAMESPACE b/NAMESPACE index 790a485..b256de3 100644 --- a/NAMESPACE +++ b/NAMESPACE @@ -1,5 +1,10 @@ # Generated by roxygen2: do not edit by hand +S3method(base::`dimnames<-`,ImmunData) +S3method(base::`names<-`,ImmunData) +S3method(dimnames,ImmunData) +S3method(dplyr::collect,ImmunData) +S3method(dplyr::compute,ImmunData) S3method(dplyr::count,ImmunData) S3method(dplyr::filter,ImmunData) S3method(dplyr::mutate,ImmunData) @@ -7,19 +12,21 @@ S3method(print,ImmunData) export(ImmunData) export(agg_receptors) export(agg_repertoires) +export(agg_strata) export(annotate) +export(annotate_anndata) export(annotate_barcodes) export(annotate_chains) export(annotate_immundata) export(annotate_receptors) export(annotate_seurat) export(assert_receptor_schema) +export(downsample_immundata) export(filter_barcodes) export(filter_immundata) export(filter_receptors) export(from_immunarch) export(get_test_idata) -export(get_test_immundata) export(imd_drop_cols) export(imd_files) export(imd_meta_schema) @@ -38,84 +45,101 @@ export(make_receptor_schema) export(make_seq_options) export(mutate_immundata) export(read_immundata) -export(read_metadata) +export(read_manifest) export(read_repertoires) +export(rename_strata) export(test_receptor_schema) export(write_immundata) -import(rlang) importFrom(R6,R6Class) -importFrom(checkmate,assert) -importFrom(checkmate,assertCharacter) -importFrom(checkmate,assertR6) -importFrom(checkmate,assert_character) -importFrom(checkmate,assert_choice) -importFrom(checkmate,assert_class) -importFrom(checkmate,assert_data_frame) -importFrom(checkmate,assert_directory_exists) -importFrom(checkmate,assert_file_exists) -importFrom(checkmate,assert_logical) -importFrom(checkmate,checkCharacter) -importFrom(checkmate,checkChoice) -importFrom(checkmate,checkR6) -importFrom(checkmate,check_character) -importFrom(checkmate,check_choice) -importFrom(checkmate,check_class) -importFrom(checkmate,check_data_frame) -importFrom(checkmate,check_function) -importFrom(checkmate,check_list) -importFrom(checkmate,check_scalar_na) -importFrom(checkmate,check_subset) -importFrom(checkmate,check_true) -importFrom(checkmate,makeAssertion) -importFrom(checkmate,test_character) -importFrom(checkmate,test_data_frame) -importFrom(checkmate,test_file_exists) -importFrom(checkmate,test_function) -importFrom(checkmate,test_list) -importFrom(checkmate,test_subset) -importFrom(cli,cli_abort) -importFrom(cli,cli_alert_danger) -importFrom(cli,cli_alert_info) -importFrom(cli,cli_alert_success) -importFrom(cli,cli_alert_warning) -importFrom(cli,cli_end) -importFrom(cli,cli_li) -importFrom(cli,cli_ol) -importFrom(cli,cli_ul) -importFrom(dplyr,all_of) -importFrom(dplyr,any_of) -importFrom(dplyr,arrange_at) -importFrom(dplyr,collect) -importFrom(dplyr,compute) -importFrom(dplyr,count) -importFrom(dplyr,distinct) -importFrom(dplyr,filter) -importFrom(dplyr,full_join) -importFrom(dplyr,left_join) -importFrom(dplyr,mutate) -importFrom(dplyr,n) -importFrom(dplyr,pull) -importFrom(dplyr,relocate) -importFrom(dplyr,rename) -importFrom(dplyr,right_join) -importFrom(dplyr,row_number) -importFrom(dplyr,select) -importFrom(dplyr,semi_join) -importFrom(dplyr,summarise) -importFrom(duckplyr,as_duckdb_tibble) -importFrom(duckplyr,as_tbl) -importFrom(duckplyr,compute_parquet) -importFrom(duckplyr,duckdb_tibble) -importFrom(duckplyr,read_csv_duckdb) -importFrom(duckplyr,read_parquet_duckdb) -importFrom(ggplot2,autoplot) +importFrom(checkmate, + assert, + assertCharacter, + assertR6, + assert_character, + assert_choice, + assert_class, + assert_data_frame, + assert_directory_exists, + assert_file_exists, + assert_logical, + assert_string, + checkCharacter, + checkChoice, + checkR6, + check_character, + check_choice, + check_class, + check_data_frame, + check_function, + check_list, + check_scalar_na, + check_subset, + check_true, + makeAssertion, + test_character, + test_data_frame, + test_file_exists, + test_function, + test_list, + test_subset +) +importFrom(cli, + cli_abort, + cli_alert_danger, + cli_alert_info, + cli_alert_success, + cli_alert_warning, + cli_end, + cli_li, + cli_ol, + cli_ul +) +importFrom(dplyr, + all_of, + any_of, + arrange, + collect, + compute, + count, + distinct, + filter, + full_join, + inner_join, + left_join, + mutate, + n, + pull, + relocate, + rename, + right_join, + row_number, + select, + semi_join, + summarise, + union_all +) +importFrom(duckplyr, + as_duckdb_tibble, + as_tbl, + compute_parquet, + duckdb_tibble, + read_csv_duckdb, + read_parquet_duckdb +) importFrom(glue,glue) +importFrom(jsonlite,unbox) importFrom(lifecycle,deprecated) importFrom(readr,read_delim) -importFrom(rlang,sym) -importFrom(rlang,syms) +importFrom(rlang, + ":=", + .data, + sym, + syms +) importFrom(tibble,as_tibble) importFrom(tools,file_ext) -importFrom(utils,packageDescription) -importFrom(utils,packageVersion) -importFrom(utils,tail) +importFrom(utils, + packageDescription, + packageVersion, + tail +) diff --git a/NEWS.md b/NEWS.md new file mode 100644 index 0000000..ecf5513 --- /dev/null +++ b/NEWS.md @@ -0,0 +1,94 @@ +# immundata 0.1.0 + +This release introduces manifests as the input-file annotation interface and +makes repertoire, strata, and provenance state more explicit and reliable. + +## Breaking changes + +* Renamed the input repertoire metadata interface to avoid confusion with the + `metadata.json` snapshot file. `read_metadata()` is replaced by + `read_manifest()`. In `read_repertoires()`, use `manifest`, + `manifest_file_col`, and `path = ""` instead of `metadata`, + `metadata_file_col`, and `path = ""`. The default manifest file + column is now `"file"` rather than `"File"`. +* `read_repertoires()` now uses `repertoire_schema = ""` by default. This + creates one repertoire per input file, or one per manifest row when paths are + supplied by a manifest. Set `repertoire_schema = NULL` to retain the previous + behavior of leaving repertoires undefined. +* `agg_strata()` now uses the argument names `schema` and `prefix` instead of + `by` and `strata_name_prefix`. +* Removed the `ImmunData$metadata` accessor. Use `idata$repertoires` for the + repertoire definitions and summaries, and use manifests for annotations + associated with input repertoire files. +* For extension developers, `imd_schema("metadata_filename")` is now + `imd_schema("manifest_filename")`, and the unused `imd_files()$receptors` + entry has been removed. + +## New features and improvements + +* `read_repertoires()` now works approximately 60 times faster by + combining CSV, TSV, and compressed text inputs into + one temporary Parquet file before processing by default. This avoids repeated + text scans in downstream duckplyr queries while retaining original input + paths in provenance. Use `prematerialize = FALSE` to disable it or + `prematerialize_folder` to select the temporary storage directory. I recommend you + to use it pretty much always. +* Added `read_manifest()` for CSV, TSV, TXT, and in-memory manifests. It infers + common delimiters, resolves file-relative paths, validates file availability, + and adds normalized source paths for joining to repertoire data. The special + `repertoire_schema = ""` value defines repertoires from all manifest + columns. +* Promoted strata to first-class `ImmunData` state. Objects now expose + `schema_strata` and a `$strata` table; `agg_strata()` and `rename_strata()` + update this state, and snapshots persist and restore it. +* Added grouped mutation through `.by` in `mutate_immundata()` and + `dplyr::mutate()` methods for `ImmunData`, including a duckplyr-compatible + fallback for grouped summary expressions. +* Added `conflicts = c("error", "replace")` to the annotation functions. + Existing annotation columns are protected by default, while intentional + replacement is allowed for columns that do not define core `ImmunData` + state. +* `mutate()`, `compute()`, and annotation operations now preserve repertoire, + strata, and provenance state when the biological grouping has not changed. + Filtering and downsampling rebuild affected repertoire and strata summaries + and retain existing stratum labels when possible. +* Added consistent progress control to manifest reading, repertoire ingestion + and aggregation, and snapshot reading and writing. Use `verbose = FALSE` for + individual calls or `options(immundata.verbose = FALSE)` globally. +* Snapshot metadata now stores repertoire and strata definitions and validates + them against the Parquet annotation columns when loading. Older metadata + formats remain readable and are upgraded in memory when necessary. +* Provenance now includes derived artifact locations (`artifacts_root` and + `artifacts_path`) associated with the project home and current snapshot. + +## Bug fixes + +* Corrected repertoire-level cell and receptor counts for paired-chain data and + prevented chain rows from inflating `n_barcodes`, `n_receptors`, receptor + proportions, and repertoire-occurrence counts. +* Made `imd_repertoire_id` assignment deterministic by ordering repertoire + schema values before assigning identifiers. +* Scoped single-cell barcodes by source filename during chain selection and + pairing, preventing identical barcode strings from different input files from + being treated as the same cell. +* Sequence filters now retain every chain belonging to a matched receptor, + including exact, regular-expression, Hamming, and Levenshtein matching. + Distance calculations no longer use the k-mer prefilter, and temporary DuckDB + table names are unique across repeated operations. +* Fixed downsampling with DuckDB 1.5 and later, preserved annotation columns in + bulk count mode, retained provenance and strata state, and made `n = 1` mean + an absolute sampling depth of one. +* Hardened ingestion against duplicate manifest paths, negative bulk counts, + missing argument columns, and collisions between custom and canonical locus + columns. +* Prevented `mutate()` and annotation replacement from overwriting system, + receptor-schema, repertoire-schema, or strata-defining columns. +* Fixed Windows path handling in the test and example infrastructure. + +## Documentation and maintenance + +* Reworked the package documentation around biological units, lazy duckplyr + workflows, ingestion, aggregation, filtering, annotation, snapshots, and + provenance, and moved website generation to altdoc. +* Removed the unused ggplot2 dependency and raised the minimum supported dplyr + version to 1.2.1. diff --git a/R/core_immundata.R b/R/core_immundata.R index bcd194f..dbbf595 100644 --- a/R/core_immundata.R +++ b/R/core_immundata.R @@ -1,18 +1,133 @@ -#' @title ImmunData: A Unified Structure for Immune Receptor Repertoire Data +#' @title ImmunData: A data structure for storing adaptive immune receptor repertoire data #' #' @description -#' `ImmunData` is an abstract R6 class for managing and transforming immune receptor repertoire data. -#' It supports flexible backends (e.g., Arrow, DuckDB, dbplyr) and lazy evaluation, -#' and provides tools for filtering, aggregation, and receptor-to-repertoire mapping. +#' `ImmunData` stores adaptive immune receptor repertoire (AIRR) data and the rules +#' used to turn observed sequences or cells into data units for analysis. Think AnnData +#' or SeuratObject, but for immune repertoires. #' -#' @seealso [read_repertoires()], [read_immundata()] +#' You work with an `ImmunData` object after importing bulk or single-cell AIRR-seq +#' data. The major idea behind `ImmunData` is that because sequencing provides only +#' information about sequences and, for single-cell data, cell +#' barcodes, the your responsibility is to determine, which sequences you want to treat as the +#' same receptor, repertoire, or stratum (group of repertoires). You define these analysis units with +#' schemas. A schema is a stored set of column names and chain-selection rules +#' that tells `ImmunData` how to group observations. Those definitions are kept +#' inside `ImmunData` to ensure that downstream functions count, filter, and compare +#' the same units consistently. Repertoire and strata schemas can be changed later to re-aggregate +#' repertoires differently, e.g., merge receptors from different clusters into +#' per-patient clusters. Receptor schema is fixed once and for all, so if you want +#' to work with a different receptor definition, e.g., use "CDR3aa + V gene" instead of +#' just "CDR3aa" as a definiton for a unique receptor, you will need to create +#' a separate `ImmunData` object. +#' +#' `ImmunData` is immutable, meaning that functions that transform an `ImmunData` +#' object return a new object, and the original object is not changed. Due to multiple +#' optimisations on the backend, it does not mean that you re-create the whole +#' dataset each time you run a, let's stay, a filter. However, it does affect analysis workflow +#' significantly. You can read about it more on the website and in tutorials. +#' +#' @section From observed data to analysis units: +#' +#' `ImmunData` connects observed records to user-defined analysis units: +#' +#' * A **chain observation** is an observed receptor-chain sequence, such as a +#' TRA, TRB, or IGH sequence. Chain observations form the main table. +#' * A **barcode** is an observed identifier for a cell in single-cell data. It +#' links chains found in the same cell. +#' * A **receptor** is a virtual analysis unit that you define. For example, you +#' may define it by CDR3 sequence alone, by CDR3 and V gene, or as a paired +#' TRA-TRB receptor. The receptor schema records which chain features and loci +#' must match for observations to receive the same receptor identifier. +#' * A **repertoire** is a virtual collection of receptors that you define from +#' annotation columns. For example, one repertoire may contain all receptors +#' from one sample, or from one donor at one time point. +#' * A **stratum** is a virtual collection of repertoires for a comparison. For +#' example, one stratum may contain all repertoires from one treatment arm. +#' +#' These definitions do not change the observed sequences. They determine how +#' observations are grouped and counted during analysis. The resulting +#' hierarchy is `chain observations and barcodes -> receptors -> repertoires -> +#' strata`. +#' +#' @section Inspect and transform an object: +#' +#' Print an object for a compact overview. Use `$receptors` for the receptor +#' table, `$repertoires` for one summary row per repertoire, and `$strata` for +#' one row per stratum. Most analysis functions accept the complete +#' `ImmunData` object directly. +#' +#' Common transformations include: +#' +#' * [filter_immundata()] to keep selected chains, cells, or receptors; +#' * [mutate_immundata()] to calculate annotation columns; +#' * [annotate()] to add external biological information; +#' * [agg_repertoires()] to define repertoires; and +#' * [agg_strata()] to group repertoires into strata. +#' +#' @section Create an object: +#' +#' Create an `ImmunData` object with [read_repertoires()], or reopen a saved +#' object with [read_immundata()]. Do not call the `$new()` constructor in +#' analysis code. Direct construction is reserved for package developers. +#' +#' @section Lazy data and storage: +#' +#' The chain-level table uses duckplyr and can remain on disk. Filtering, +#' mutation, and aggregation stay lazy when possible, so large datasets do not +#' need to be loaded fully into R memory. Downstream analysis functions in the +#' `immunarch` package are designed to accept lazy `ImmunData` objects. Pass the +#' object directly; you usually do not need to call [dplyr::collect()]. Collect +#' data only when another function explicitly requires an in-memory data frame +#' or when you want to inspect a small table in R. +#' +#' Objects created by [read_repertoires()] are backed by files in their output +#' folder. Keep that folder while you use the object. Use [write_immundata()] to +#' save a transformed object and [read_immundata()] to reopen it. +#' +#' @seealso [read_repertoires()], [read_immundata()], [write_immundata()], +#' [agg_repertoires()], [agg_strata()], [filter_immundata()], +#' [mutate_immundata()], [annotate()] +#' +#' @examples +#' library(immundata) +#' library(dplyr) +#' +#' options(immundata.verbose = FALSE) +#' +#' # Load the small dataset included with immundata, then define one repertoire +#' # for each treatment-response group. +#' idata <- get_test_idata() |> +#' agg_repertoires(schema = "Response") +#' +#' idata$repertoires |> +#' select(Response, n_barcodes, n_receptors) |> +#' arrange(Response) +#' # Expected result: +#' # Response n_barcodes n_receptors +#' # FR 955 871 +#' # PR 947 867 +#' +#' # Under the current receptor definition, the full-response (FR) repertoire +#' # contains 955 chain observations grouped into 871 receptor units. +#' +#' # Keep only the full-response repertoire. filter() returns a new object. +#' fr_only <- idata |> +#' filter(Response == "FR") +#' +#' tibble( +#' original_repertoires = nrow(idata$repertoires), +#' filtered_repertoires = nrow(fr_only$repertoires) +#' ) +#' # Expected result: +#' # original_repertoires filtered_repertoires +#' # 2 1 +#' # The original object still contains both repertoires. #' #' @concept core_immundata #' @export ImmunData <- R6Class( "ImmunData", private = list( - # .annotations A barcode-level table that links each barcode (i.e., cell ID) # to a receptor. It can also store cell-level metadata such as # sample ID, donor, or tissue source. This table is **not aggregated** and @@ -20,32 +135,57 @@ ImmunData <- R6Class( .annotations = NULL, # .repertoire_table A duckplyr table with repertoire names and receptor counts. - .repertoire_table = NULL + .repertoire_table = NULL, + + # .strata_table A duckplyr table with one row per stratum. + .strata_table = NULL, + + # .provenance Internal snapshot/provenance metadata used by IO helpers. + .provenance = NULL ), public = list( - - #' @field schema_receptor A named list describing how to interpret receptor-level data. - #' This includes the fields used for aggregation (e.g., `CDR3`, `V_gene`, `J_gene`), - #' and optionally unique identifiers for each receptor row. Used to ensure consistency - #' across processing steps. + #' @field schema_receptor A named list defining the virtual receptor unit. + #' The `features` element names the chain columns used to group + #' observations, such as CDR3 sequence and V gene. The `chains` element + #' selects one chain or a paired set of chains. schema_receptor = NULL, - #' @field schema_repertoire A named list defining how barcodes or annotations should be - #' grouped into repertoires. This may include sample-level metadata (e.g., `sample_id`, - #' `donor_id`) used to define unique repertoires. + #' @field schema_repertoire A character vector naming annotation columns + #' whose unique combinations define one repertoire, such as `sample_id` + #' or `c("donor_id", "timepoint")`. It is `NULL` when repertoires have + #' not been defined. schema_repertoire = NULL, - #' @description Creates a new `ImmunData` object. - #' This constructor expects receptor-level and barcode-level data, - #' along with a receptor schema defining aggregation and identity fields. + #' @field schema_strata A character vector naming repertoire-level columns + #' whose unique combinations define one stratum, such as `treatment`. It + #' is `NULL` when strata have not been defined. + schema_strata = NULL, + + #' @description Low-level constructor for package developers. Analysis code + #' must create an `ImmunData` object with [read_repertoires()] or reopen one + #' with [read_immundata()]. #' - #' @param schema A character vector specifying the receptor schema (e.g., aggregate fields, ID columns). - #' @param annotations A cell/barcode-level dataset mapping barcodes to receptor rows. - #' @param repertoires A repertoire table, created inside the body of [agg_repertoires]. + #' @param schema A character vector or named list. A character vector names + #' the features used to define a chain-agnostic receptor. A named list is + #' created by [make_receptor_schema()] and can also select receptor chains. + #' @param annotations A duckplyr table. It contains retained chain + #' observations, receptor identifiers, and biological annotations. + #' @param repertoires A data frame or `NULL`. It contains one row per + #' repertoire and its summary statistics and is usually created by + #' [agg_repertoires()]. + #' @param provenance A list or `NULL`. It contains internal storage and + #' snapshot history. + #' @param strata A data frame or `NULL`. It contains one row per stratum, + #' its label, and the repertoire-level columns that define it. initialize = function(schema, annotations, - repertoires = NULL) { - checkmate::check_data_frame(annotations) + repertoires = NULL, + provenance = NULL, + strata = NULL) { + checkmate::assert_data_frame(annotations) + checkmate::assert_data_frame(repertoires, null.ok = TRUE) + checkmate::assert_data_frame(strata, null.ok = TRUE) + checkmate::assert_list(provenance, null.ok = TRUE) if (checkmate::test_character(schema)) { schema <- make_receptor_schema(features = schema, chains = NULL) @@ -53,19 +193,74 @@ ImmunData <- R6Class( private$.annotations <- annotations self$schema_receptor <- schema + private$.provenance <- if (is.null(provenance)) { + NULL + } else { + normalize_provenance(provenance) + } if (!is.null(repertoires)) { - self$schema_repertoire <- setdiff(colnames(repertoires), c(imd_schema()$repertoire, imd_schema()$n_receptors, imd_schema()$n_barcodes, imd_schema()$n_cells)) + self$schema_repertoire <- setdiff( + colnames(repertoires), + c( + imd_schema()$repertoire, + imd_schema()$strata, + imd_schema()$strata_name, + imd_schema()$n_receptors, + imd_schema()$n_barcodes, + imd_schema()$n_cells + ) + ) private$.repertoire_table <- repertoires } + + if (!is.null(strata)) { + if (is.null(repertoires)) { + cli::cli_abort("A {.field strata} table requires a non-null {.field repertoires} table.") + } + + internal_strata_columns <- c( + imd_schema("strata"), + imd_schema("strata_name") + ) + missing_internal_columns <- setdiff(internal_strata_columns, colnames(strata)) + if (length(missing_internal_columns) > 0) { + cli::cli_abort( + "Strata table is missing required column(s): [{missing_internal_columns}]." + ) + } + + self$schema_strata <- setdiff(colnames(strata), internal_strata_columns) + if (length(self$schema_strata) == 0) { + cli::cli_abort("Strata table must contain at least one strata schema column.") + } + + missing_repertoire_schema <- setdiff( + self$schema_strata, + self$schema_repertoire + ) + if (length(missing_repertoire_schema) > 0) { + cli::cli_abort( + "Strata schema column(s) [{missing_repertoire_schema}] are not part of the inferred repertoire schema." + ) + } + + missing_repertoire_columns <- setdiff(colnames(strata), colnames(repertoires)) + if (length(missing_repertoire_columns) > 0) { + cli::cli_abort( + "Strata column(s) [{missing_repertoire_columns}] are missing from {.field repertoires}." + ) + } + + private$.strata_table <- strata + } } ), active = list( - - #' @field receptors Accessor for the dynamically-created table with receptors. + #' @field receptors A derived duckplyr table of distinct receptors. For a + #' paired receptor, the selected chain features are shown side by side. receptors = function() { receptor_id_col <- imd_schema("receptor") - barcode_col <- imd_schema("barcode") locus_col <- imd_schema("locus") features <- imd_receptor_features(self$schema_receptor) chains <- imd_receptor_chains(self$schema_receptor) @@ -74,21 +269,36 @@ ImmunData <- R6Class( receptor_data <- private$.annotations |> select(all_of(c( receptor_id_col, - barcode_col, features, locus_col - ))) + ))) |> + distinct() - locus_1 <- chains[1] - locus_2 <- chains[2] + if (!grepl("\\|", chains[2])) { + locus_1 <- chains[1] + locus_2 <- chains[2] - receptor_data |> - filter(!!rlang::sym(locus_col) == locus_1) |> - full_join( - receptor_data |> - filter(!!rlang::sym(locus_col) == locus_2), - by = c(receptor_id_col, barcode_col) - ) + receptor_data |> + filter(!!rlang::sym(locus_col) == locus_1) |> + full_join( + receptor_data |> + filter(!!rlang::sym(locus_col) == locus_2), + by = receptor_id_col + ) + } else { + relaxed_chain_alternatives <- trimws(unlist(strsplit(chains[2], "\\|"))) + locus_1 <- chains[1] + locus_2 <- relaxed_chain_alternatives[1] + locus_3 <- relaxed_chain_alternatives[2] + + receptor_data |> + filter(!!rlang::sym(locus_col) == locus_1) |> + full_join( + receptor_data |> + filter(!!rlang::sym(locus_col) %in% c(locus_2, locus_3)), + by = receptor_id_col + ) + } } else { private$.annotations |> select({{ receptor_id_col }}, all_of(features)) |> @@ -96,33 +306,104 @@ ImmunData <- R6Class( } }, - #' @field annotations Accessor for the annotation-level table (`.annotations`). + #' @field annotations The lazy duckplyr table of retained chain + #' observations and their biological annotations. For most tasks, pass the + #' complete `ImmunData` object to a transformation function or use + #' `collect(idata)` to inspect this table in memory. annotations = function() { private$.annotations }, - #' @field repertoires Get a table of repertoires and their basic statistics. + #' @field repertoires A small table with one row per repertoire, the columns + #' that define it, and summary statistics such as `n_barcodes` and + #' `n_receptors`. It is `NULL` when repertoires have not been defined. repertoires = function() { # TODO: cache repertoire table to memory if not very big? if (!is.null(private$.repertoire_table)) { - private$.repertoire_table |> - collect() |> - arrange_at(vars(1)) + repertoire_table <- private$.repertoire_table |> + collect() + + if (imd_schema("repertoire") %in% colnames(repertoire_table)) { + repertoire_table <- repertoire_table |> + arrange(.data[[imd_schema("repertoire")]]) + } + + repertoire_table } else { NULL } }, - #' @field metadata Get a table of repertoires without their basic statistics. - metadata = function() { - if (!is.null(private$.repertoire_table)) { - private$.repertoire_table |> - select(c(imd_schema("repertoire"), self$schema_repertoire)) |> - collect() |> - arrange_at(vars(1)) + #' @field strata A small table with one row per stratum, its label, and the + #' repertoire-level columns that define it. It is `NULL` when strata have + #' not been defined. + strata = function() { + if (!is.null(private$.strata_table)) { + strata_table <- private$.strata_table |> + collect() + + if (imd_schema("strata") %in% colnames(strata_table)) { + strata_table <- strata_table |> + arrange(.data[[imd_schema("strata")]]) + } + + strata_table } else { NULL } + }, + + #' @field provenance Read-only named list describing the snapshot origin + #' and storage context carried by this object. Retrieve the complete list with + #' `idata$provenance`, or one field with, for example, + #' `idata$provenance$current_path`. The fields are: + #' + #' * `home_path`: project home used for managed snapshots and artifacts. + #' The original ingestion snapshot is stored directly in this folder; + #' it is `NULL` for an object with no persisted home. + #' * `current_path`: exact folder of the most recently loaded or written + #' snapshot. Transformations preserve this source path until the + #' transformed object is written as another snapshot; it is `NULL` for + #' an object that has never been loaded from or written to disk. + #' * `snapshot_root`: derived managed-snapshot root, + #' `home_path/snapshots`, or `NULL` when `home_path` is `NULL`. + #' * `artifacts_root`: derived project-level root for optional external + #' tool outputs, `home_path/artifacts`, or `NULL` when `home_path` is + #' `NULL`. + #' * `artifacts_path`: derived namespace for artifacts associated with the + #' most recently loaded or written snapshot. It is + #' `artifacts_root/root` for the original + #' ingestion, `artifacts_root//vNNN` for a managed snapshot, and + #' `artifacts_root/by-id/` for a detached explicit snapshot. + #' External tools can append `/` and create that directory; + #' artifact contents are not part of `ImmunData`. Write a transformed + #' object as a new snapshot before storing artifacts that should be + #' associated with the transformed data. + #' * `snapshot_id`: unique identifier generated when the snapshot is + #' written; `NULL` for an in-memory object that has never been written. + #' * `lineage`: ordered list of ingestion and snapshot events leading to + #' the current snapshot. + #' + #' The accessor is read-only; assigning to `idata$provenance` is an error. + provenance = function(value) { + if (missing(value)) { + return(get_provenance(self)) + } + + cli::cli_abort("`provenance` is read-only and cannot be assigned directly.") } ) ) + +clone_with_annotations <- function(idata, annotations) { + checkmate::assert_r6(idata, "ImmunData") + checkmate::assert_data_frame(annotations) + + ImmunData$new( + schema = idata$schema_receptor, + annotations = annotations, + repertoires = idata$repertoires, + strata = idata$strata, + provenance = get_provenance(idata) + ) +} diff --git a/R/globals.R b/R/globals.R index fc5344a..abfb54e 100644 --- a/R/globals.R +++ b/R/globals.R @@ -16,12 +16,11 @@ utils::globalVariables(c("dd", "meta", "n_cells", "n_barcodes", "p", "tmp_recept #' - `cell`: Column name for cell barcode IDs. #' - `receptor`: Column name for receptor unique identifiers. #' - `repertoire`: Column name for repertoire group IDs. -#' - `metadata_filename`: Column name for metadata files (internal). +#' - `manifest_filename`: Column name for manifest file paths (internal). #' - `count`: Column name for receptor count per group. -#' - `filename`: Original column name used in user metadata. #' - `files`: Default file names used to store structured Immundata: -#' - `receptors`: File name for receptor-level data (`receptors.parquet`). -#' - `annotations`: File name for annotation-level data (`annotations.parquet`). +#' - `metadata`: File name for schemas and small summary tables (`metadata.json`). +#' - `annotations`: File name for chain-level data (`annotations.parquet`). #' #' @keywords internal IMD_GLOBALS <- list( @@ -32,7 +31,9 @@ IMD_GLOBALS <- list( chain = "imd_chain_id", group = "imd_group_id", repertoire = "imd_repertoire_id", - metadata_filename = "imd_filename", + strata = "imd_strata_id", + strata_name = "strata_name", + manifest_filename = "imd_filename", count = "imd_count", receptor_count = "imd_count", chain_count = "imd_n_chains", @@ -41,7 +42,6 @@ IMD_GLOBALS <- list( n_barcodes = "n_barcodes", n_cells = "n_cells", n_repertoires = "n_repertoires", - filename = "filename", locus = "locus", sim_exact = "imd_sim_exact_", sim_regex = "imd_sim_regex_", @@ -49,13 +49,20 @@ IMD_GLOBALS <- list( sim_lev = "imd_sim_lev_" ), meta_schema = list( - version = "version", - receptor_schema = "receptor_schema", - repertoire_schema = "repertoire_schema" + format_version = "format_version", + package_version = "package_version", + schema_receptor = "schema_receptor", + schema_repertoire = "schema_repertoire", + schema_strata = "schema_strata", + repertoires = "repertoires", + producer = "producer", + snapshot_id = "snapshot_id", + lineage = "lineage", + provenance = "provenance", + extensions = "extensions" ), files = list( metadata = "metadata.json", - receptors = "receptors.parquet", annotations = "annotations.parquet" ), rename_cols = list( @@ -121,19 +128,43 @@ IMD_GLOBALS <- list( ) ) -#' @title Get Immundata internal schema field names +#' @title Get a standard ImmunData column name #' #' @description -#' Returns the standardized field names used across Immundata objects and processing functions, -#' as defined in `IMD_GLOBALS$schema`. These include column names for cell ids or barcodes, receptors, -#' repertoires, and related metadata. +#' Use `imd_schema()` when code needs the standard column name for an +#' `ImmunData` identifier or calculated value, such as the cell barcode, +#' receptor identifier, repertoire identifier, count, or proportion. #' -#' @param key Character which field to return. -#' @param format Character what format to load - "airr" or "10x". -#' @param schema Receptor schema from [make_receptor_schema()]. +#' Use this helper in reusable analysis code or package extensions instead of +#' writing an internal name such as `"imd_barcode"` directly. It only returns +#' names; it does not inspect or change an [ImmunData] object. #' -#' @concept schema +#' @param key A character string or `NULL`. One schema key, for example +#' `"barcode"`, `"receptor"`, `"repertoire"`, `"count"`, or +#' `"proportion"`. Use `NULL`, the default, to return all available keys and +#' column names. +#' +#' @return If `key` is supplied, one character string containing the standard +#' column name. If `key = NULL`, a named list of all schema keys and column +#' names. +#' +#' @seealso [make_receptor_schema()], [imd_rename_cols()], [ImmunData] +#' +#' @examples +#' imd_schema("barcode") +#' # Expected result: "imd_barcode" #' +#' imd_schema("receptor") +#' # Expected result: "imd_receptor_id" +#' +#' # Use a returned name for programmatic selection. +#' barcode_column <- imd_schema("barcode") +#' get_test_idata() |> +#' dplyr::collect() |> +#' dplyr::select(dplyr::all_of(barcode_column)) |> +#' head(2) +#' +#' @concept schema #' @export imd_schema <- function(key = NULL) { if (is.null(key)) { @@ -144,7 +175,55 @@ imd_schema <- function(key = NULL) { } } -#' @rdname imd_schema +#' @title Developer helpers for ImmunData schemas and storage +#' +#' @description +#' These helpers expose package constants for extension developers. They are not +#' needed for routine biological analysis. Use [imd_schema()] for a standard +#' column name and [make_receptor_schema()] to define biological receptors. +#' +#' The functions remain exported for compatibility with packages that extend +#' `immundata`, but their values describe implementation details and may grow as +#' the storage format develops. +#' +#' @param key A character string or `NULL`. For `imd_schema_sym()`, one schema +#' key accepted by [imd_schema()]. Use `NULL`, the default, to return the +#' complete named schema list. +#' @param format A character string. For `imd_repertoire_schema()`, the preset +#' name. Currently only `"airr"` is accepted. +#' @param schema A receptor-schema list. For `imd_receptor_features()` and +#' `imd_receptor_chains()`, a schema created by [make_receptor_schema()]. +#' +#' @return +#' * `imd_schema_sym()` returns an rlang symbol for one standard column. With +#' `key = NULL`, it returns the complete named schema list. +#' * `imd_meta_schema()` returns a named list of fields used in +#' `metadata.json`. +#' * `imd_files()` returns a named list of standard snapshot file names. +#' * `imd_repertoire_schema()` returns the configured preset for `format`, or +#' `NULL` when no preset is configured. +#' * `imd_receptor_features()` returns the character vector in +#' `schema$features`. +#' * `imd_receptor_chains()` returns the character vector in `schema$chains`, or +#' `NULL` for a chain-agnostic schema. +#' +#' @examples +#' schema <- make_receptor_schema( +#' features = c("junction_aa", "v_call"), +#' chains = c("TRA", "TRB") +#' ) +#' +#' imd_receptor_features(schema) +#' # Expected result: c("junction_aa", "v_call") +#' +#' imd_receptor_chains(schema) +#' # Expected result: c("TRA", "TRB") +#' +#' imd_files() +#' # Lists the standard metadata and Parquet file names. +#' +#' @keywords internal +#' @concept schema #' @export imd_schema_sym <- function(key = NULL) { if (is.null(key)) { @@ -155,20 +234,60 @@ imd_schema_sym <- function(key = NULL) { } } -#' @rdname imd_schema +#' @rdname imd_schema_sym #' @export imd_meta_schema <- function() { # TODO: pass value to the function IMD_GLOBALS$meta_schema } -#' @rdname imd_schema +#' @rdname imd_schema_sym #' @export imd_files <- function() { IMD_GLOBALS$files } -#' @rdname imd_schema +#' @title Get input-column presets +#' +#' @description +#' Use these helpers to inspect or customize the column renaming and removal +#' presets used by [read_repertoires()]. +#' +#' `imd_rename_cols()` returns mappings from standard output names to source +#' names. `imd_drop_cols()` returns technical columns that can usually be +#' removed before receptors are defined. These functions return definitions +#' only; they do not change input files or an [ImmunData] object. +#' +#' @param format A character string. The input format preset. For +#' `imd_rename_cols()`, use `"default"` or `"10x"`; the default is +#' `"default"`. For `imd_drop_cols()`, use `"universal"`, `"airr"`, or +#' `"10x"`; the default is `"airr"`. +#' +#' @return `imd_rename_cols()` returns a named character vector in the form +#' `c(new_name = "source_name")`. `imd_drop_cols()` returns a character +#' vector of source columns to remove. +#' +#' @seealso [read_repertoires()], [make_default_preprocessing()], [imd_schema()] +#' +#' @examples +#' imd_rename_cols("10x") +#' # Includes c(v_call = "v_gene", locus = "chain"). +#' +#' head(imd_drop_cols("10x"), 3) +#' # Expected result: +#' # "full_length" "is_cell" "contig_id" +#' +#' # Keep the 10x `contig_id` column while dropping the other default columns. +#' columns_to_drop <- setdiff(imd_drop_cols("10x"), "contig_id") +#' custom_preprocessing <- list( +#' exclude_columns = make_exclude_columns(columns_to_drop), +#' filter_nonproductive = make_productive_filter( +#' truthy = c("TRUE", "true", "1") +#' ) +#' ) +#' +#' @concept ingestion +#' @rdname imd_input_columns #' @export imd_rename_cols <- function(format = "default") { checkmate::assert_character(format) @@ -177,7 +296,7 @@ imd_rename_cols <- function(format = "default") { IMD_GLOBALS$rename_cols[[format]] } -#' @rdname imd_schema +#' @rdname imd_input_columns #' @export imd_drop_cols <- function(format = "airr") { checkmate::assert_character(format) @@ -186,7 +305,7 @@ imd_drop_cols <- function(format = "airr") { IMD_GLOBALS$drop_cols[[format]] } -#' @rdname imd_schema +#' @rdname imd_schema_sym #' @export imd_repertoire_schema <- function(format = "airr") { checkmate::assert_character(format) @@ -195,13 +314,13 @@ imd_repertoire_schema <- function(format = "airr") { IMD_GLOBALS$agg_schema$repertoires[[format]] } -#' @rdname imd_schema +#' @rdname imd_schema_sym #' @export imd_receptor_features <- function(schema) { schema[["features"]] } -#' @rdname imd_schema +#' @rdname imd_schema_sym #' @export imd_receptor_chains <- function(schema) { schema[["chains"]] diff --git a/R/immundata-package.R b/R/immundata-package.R index 8a5ef7d..ad0d315 100644 --- a/R/immundata-package.R +++ b/R/immundata-package.R @@ -2,7 +2,6 @@ "_PACKAGE" ## usethis namespace: start -#' @import rlang #' @importFrom checkmate assert #' @importFrom checkmate assert_character #' @importFrom checkmate assert_choice @@ -11,6 +10,7 @@ #' @importFrom checkmate assert_directory_exists #' @importFrom checkmate assert_file_exists #' @importFrom checkmate assert_logical +#' @importFrom checkmate assert_string #' @importFrom checkmate assertCharacter #' @importFrom checkmate assertR6 #' @importFrom checkmate check_character @@ -43,13 +43,14 @@ #' @importFrom cli cli_ul #' @importFrom dplyr all_of #' @importFrom dplyr any_of -#' @importFrom dplyr arrange_at +#' @importFrom dplyr arrange #' @importFrom dplyr collect #' @importFrom dplyr compute #' @importFrom dplyr count #' @importFrom dplyr distinct #' @importFrom dplyr filter #' @importFrom dplyr full_join +#' @importFrom dplyr inner_join #' @importFrom dplyr left_join #' @importFrom dplyr mutate #' @importFrom dplyr n @@ -61,19 +62,22 @@ #' @importFrom dplyr select #' @importFrom dplyr semi_join #' @importFrom dplyr summarise +#' @importFrom dplyr union_all #' @importFrom duckplyr as_duckdb_tibble #' @importFrom duckplyr as_tbl #' @importFrom duckplyr compute_parquet #' @importFrom duckplyr duckdb_tibble #' @importFrom duckplyr read_csv_duckdb #' @importFrom duckplyr read_parquet_duckdb -#' @importFrom ggplot2 autoplot #' @importFrom glue glue +#' @importFrom jsonlite unbox #' @importFrom lifecycle deprecated #' @importFrom R6 R6Class #' @importFrom readr read_delim +#' @importFrom rlang .data #' @importFrom rlang sym #' @importFrom rlang syms +#' @importFrom rlang := #' @importFrom tibble as_tibble #' @importFrom tools file_ext #' @importFrom utils packageDescription diff --git a/R/import-standalone-purrr.R b/R/import-standalone-purrr.R deleted file mode 100644 index 4ec36e1..0000000 --- a/R/import-standalone-purrr.R +++ /dev/null @@ -1,244 +0,0 @@ -# Standalone file: do not edit by hand -# Source: -# ---------------------------------------------------------------------- -# -# --- -# repo: r-lib/rlang -# file: standalone-purrr.R -# last-updated: 2023-02-23 -# license: https://unlicense.org -# imports: rlang -# --- -# -# This file provides a minimal shim to provide a purrr-like API on top of -# base R functions. They are not drop-in replacements but allow a similar style -# of programming. -# -# ## Changelog -# -# 2023-02-23: -# * Added `list_c()` -# -# 2022-06-07: -# * `transpose()` is now more consistent with purrr when inner names -# are not congruent (#1346). -# -# 2021-12-15: -# * `transpose()` now supports empty lists. -# -# 2021-05-21: -# * Fixed "object `x` not found" error in `imap()` (@mgirlich) -# -# 2020-04-14: -# * Removed `pluck*()` functions -# * Removed `*_cpl()` functions -# * Used `as_function()` to allow use of `~` -# * Used `.` prefix for helpers -# -# nocov start - -map <- function(.x, .f, ...) { - .f <- as_function(.f, env = global_env()) - lapply(.x, .f, ...) -} -walk <- function(.x, .f, ...) { - map(.x, .f, ...) - invisible(.x) -} - -map_lgl <- function(.x, .f, ...) { - .rlang_purrr_map_mold(.x, .f, logical(1), ...) -} -map_int <- function(.x, .f, ...) { - .rlang_purrr_map_mold(.x, .f, integer(1), ...) -} -map_dbl <- function(.x, .f, ...) { - .rlang_purrr_map_mold(.x, .f, double(1), ...) -} -map_chr <- function(.x, .f, ...) { - .rlang_purrr_map_mold(.x, .f, character(1), ...) -} -.rlang_purrr_map_mold <- function(.x, .f, .mold, ...) { - .f <- as_function(.f, env = global_env()) - out <- vapply(.x, .f, .mold, ..., USE.NAMES = FALSE) - names(out) <- names(.x) - out -} - -map2 <- function(.x, .y, .f, ...) { - .f <- as_function(.f, env = global_env()) - out <- mapply(.f, .x, .y, MoreArgs = list(...), SIMPLIFY = FALSE) - if (length(out) == length(.x)) { - set_names(out, names(.x)) - } else { - set_names(out, NULL) - } -} -map2_lgl <- function(.x, .y, .f, ...) { - as.vector(map2(.x, .y, .f, ...), "logical") -} -map2_int <- function(.x, .y, .f, ...) { - as.vector(map2(.x, .y, .f, ...), "integer") -} -map2_dbl <- function(.x, .y, .f, ...) { - as.vector(map2(.x, .y, .f, ...), "double") -} -map2_chr <- function(.x, .y, .f, ...) { - as.vector(map2(.x, .y, .f, ...), "character") -} -imap <- function(.x, .f, ...) { - map2(.x, names(.x) %||% seq_along(.x), .f, ...) -} - -pmap <- function(.l, .f, ...) { - .f <- as.function(.f) - args <- .rlang_purrr_args_recycle(.l) - do.call("mapply", c( - FUN = list(quote(.f)), - args, MoreArgs = quote(list(...)), - SIMPLIFY = FALSE, USE.NAMES = FALSE - )) -} -.rlang_purrr_args_recycle <- function(args) { - lengths <- map_int(args, length) - n <- max(lengths) - - stopifnot(all(lengths == 1L | lengths == n)) - to_recycle <- lengths == 1L - args[to_recycle] <- map(args[to_recycle], function(x) rep.int(x, n)) - - args -} - -keep <- function(.x, .f, ...) { - .x[.rlang_purrr_probe(.x, .f, ...)] -} -discard <- function(.x, .p, ...) { - sel <- .rlang_purrr_probe(.x, .p, ...) - .x[is.na(sel) | !sel] -} -map_if <- function(.x, .p, .f, ...) { - matches <- .rlang_purrr_probe(.x, .p) - .x[matches] <- map(.x[matches], .f, ...) - .x -} -.rlang_purrr_probe <- function(.x, .p, ...) { - if (is_logical(.p)) { - stopifnot(length(.p) == length(.x)) - .p - } else { - .p <- as_function(.p, env = global_env()) - map_lgl(.x, .p, ...) - } -} - -compact <- function(.x) { - Filter(length, .x) -} - -transpose <- function(.l) { - if (!length(.l)) { - return(.l) - } - - inner_names <- names(.l[[1]]) - - if (is.null(inner_names)) { - fields <- seq_along(.l[[1]]) - } else { - fields <- set_names(inner_names) - .l <- map(.l, function(x) { - if (is.null(names(x))) { - set_names(x, inner_names) - } else { - x - } - }) - } - - # This way missing fields are subsetted as `NULL` instead of causing - # an error - .l <- map(.l, as.list) - - map(fields, function(i) { - map(.l, .subset2, i) - }) -} - -every <- function(.x, .p, ...) { - .p <- as_function(.p, env = global_env()) - - for (i in seq_along(.x)) { - if (!rlang::is_true(.p(.x[[i]], ...))) { - return(FALSE) - } - } - TRUE -} -some <- function(.x, .p, ...) { - .p <- as_function(.p, env = global_env()) - - for (i in seq_along(.x)) { - if (rlang::is_true(.p(.x[[i]], ...))) { - return(TRUE) - } - } - FALSE -} -negate <- function(.p) { - .p <- as_function(.p, env = global_env()) - function(...) !.p(...) -} - -reduce <- function(.x, .f, ..., .init) { - f <- function(x, y) .f(x, y, ...) - Reduce(f, .x, init = .init) -} -reduce_right <- function(.x, .f, ..., .init) { - f <- function(x, y) .f(y, x, ...) - Reduce(f, .x, init = .init, right = TRUE) -} -accumulate <- function(.x, .f, ..., .init) { - f <- function(x, y) .f(x, y, ...) - Reduce(f, .x, init = .init, accumulate = TRUE) -} -accumulate_right <- function(.x, .f, ..., .init) { - f <- function(x, y) .f(y, x, ...) - Reduce(f, .x, init = .init, right = TRUE, accumulate = TRUE) -} - -detect <- function(.x, .f, ..., .right = FALSE, .p = is_true) { - .p <- as_function(.p, env = global_env()) - .f <- as_function(.f, env = global_env()) - - for (i in .rlang_purrr_index(.x, .right)) { - if (.p(.f(.x[[i]], ...))) { - return(.x[[i]]) - } - } - NULL -} -detect_index <- function(.x, .f, ..., .right = FALSE, .p = is_true) { - .p <- as_function(.p, env = global_env()) - .f <- as_function(.f, env = global_env()) - - for (i in .rlang_purrr_index(.x, .right)) { - if (.p(.f(.x[[i]], ...))) { - return(i) - } - } - 0L -} -.rlang_purrr_index <- function(x, right = FALSE) { - idx <- seq_along(x) - if (right) { - idx <- rev(idx) - } - idx -} - -list_c <- function(x) { - inject(c(!!!x)) -} - -# nocov end diff --git a/R/io_annotations_write.R b/R/io_annotations_write.R deleted file mode 100644 index 35aeed2..0000000 --- a/R/io_annotations_write.R +++ /dev/null @@ -1,4 +0,0 @@ -#' Write immune receptor annotations to single-cell object metadata - Seurat, AnnData, etc. -# write_annotations <- function(idata, target, cells = NA) { -# -# } diff --git a/R/io_immundata_conversion.R b/R/io_immundata_conversion.R index afa210c..bfb2d0d 100644 --- a/R/io_immundata_conversion.R +++ b/R/io_immundata_conversion.R @@ -3,7 +3,7 @@ #' @description #' The `from_immunarch()` function takes an **immunarch** object (as returned by #' `immunarch::repLoad()`), writes each repertoire to a TSV file with an added -#' `filename` column in a specified folder, and then imports those files into +#' internal filename column in a specified folder, and then imports those files into #' an **ImmunData** object via `read_repertoires()`. #' #' @param imm A list returned by `immunarch::repLoad()`, typically containing: @@ -56,8 +56,8 @@ from_immunarch <- function( names(rep_list) <- paste0("repertoire_", seq_along(rep_list)) } - # write each repertoire with a 'filename' column - immundata_filename_col <- IMD_GLOBALS$schema$filename + # write each repertoire and track its path through the internal manifest column + immundata_filename_col <- IMD_GLOBALS$schema$manifest_filename file_paths <- c() for (nm in names(rep_list)) { df <- rep_list[[nm]] @@ -77,12 +77,12 @@ from_immunarch <- function( } names(file_paths) <- names(rep_list) - # if metadata present, add 'filename' column and pass it in - metadata_df <- NULL + # If immunarch metadata is present, add the internal manifest join column. + manifest_df <- NULL if (!is.null(imm$meta)) { - metadata_df <- imm$meta - if ("Sample" %in% colnames(metadata_df)) { - metadata_df[[immundata_filename_col]] <- normalizePath(file_paths[metadata_df$Sample]) + manifest_df <- imm$meta + if ("Sample" %in% colnames(manifest_df)) { + manifest_df[[immundata_filename_col]] <- normalizePath(file_paths[manifest_df$Sample]) } else { cli_abort("No `Sample` in the metadata object. Please create the column with repertoires names from `$data`") } @@ -95,7 +95,7 @@ from_immunarch <- function( read_repertoires( path = unname(file_paths), schema = schema, - metadata = metadata_df, + manifest = manifest_df, output_folder = output_folder, repertoire_schema = "repertoire_id" ) diff --git a/R/io_immundata_read.R b/R/io_immundata_read.R index 9e378d4..5c22c21 100644 --- a/R/io_immundata_read.R +++ b/R/io_immundata_read.R @@ -1,104 +1,208 @@ -#' @title Load a saved ImmunData from disk +#' @title Load an ImmunData object from disk #' #' @description -#' Reconstructs an `ImmunData` object from files previously saved to a directory -#' by [write_immundata()] or the internal saving step of [read_repertoires()]. -#' It reads the `annotations.parquet` file for the main data and `metadata.json` -#' to retrieve the necessary receptor and repertoire schemas. -#' -#' @param path Character(1). Path to the **directory** containing the saved -#' `ImmunData` files (`annotations.parquet` and `metadata.json`). -#' @param prudence Character(1). Controls strictness of type inference when -#' reading the Parquet file, passed to `duckplyr::read_parquet_duckdb()`. -#' Default `"stingy"` likely implies stricter type checking or safer inference. -#' @param verbose Logical(1). If `TRUE` (default), prints informative messages -#' using `cli` during loading. Set to `FALSE` for quiet operation. +#' Continue an analysis later by reopening an [ImmunData] dataset saved on disk. +#' Use `read_immundata()` after restarting R, in another script, or when another +#' person gives you a dataset created by [write_immundata()] or +#' [read_repertoires()]. It is that simple, just don't forget to save the +#' `ImmunData` object first! +#' +#' The unit restored retains all information: chain rows, +#' cell and receptor identifiers, repertoire and stratum definitions, and +#' provenance. The function does not change these biological units or the saved +#' files. It returns a new [ImmunData] object. +#' +#' @param path A character string. Path to a saved dataset directory. The +#' directory must contain `annotations.parquet` and `metadata.json`. When +#' `tag` is supplied, use the project home directory that contains the +#' `snapshots` directory. Read more about snapshots on the website. +#' @param tag A character string or `NULL`. Snapshot tag to read from +#' `path/snapshots//vNNN`. If `NULL`, the default, `path` itself is read. +#' @param version A non-negative integer or `NULL`. Snapshot version within +#' `tag`. For example, `1` reads `v001`. If `NULL`, the default, the latest +#' available version for the tag is read. `version` can only be used with +#' `tag`. +#' @param prudence A character string. Memory protection used while reading the +#' Parquet data. This controls whether duckplyr may convert an intermediate +#' result from DuckDB-managed memory to an R data frame: `"stingy"`, the +#' default here, never permits conversion; `"thrifty"` permits up to 1 million +#' table cells (rows multiplied by columns); and `"lavish"` permits conversion +#' regardless of size. Here, "table cells" does not mean biological cells. +#' Passed to [duckplyr::read_parquet_duckdb()]. +#' @param verbose A logical value. Whether to print progress and summary +#' messages. Defaults to `getOption("immundata.verbose", TRUE)`. #' #' @details -#' This function expects a directory structure created by [write_immundata()], -#' containing at least: -#' - `annotations.parquet`: The main annotation data table. -#' - `metadata.json`: Contains package version, receptor schema, and optionally -#' repertoire schema. -#' -#' The loading process involves: -#' 1. Checking that the specified `path` is a directory and contains the -#' required `annotations.parquet` and `metadata.json` files. -#' 2. Reading `metadata.json` using `jsonlite::read_json()`. -#' 3. Reading `annotations.parquet` using `duckplyr::read_parquet_duckdb()` with -#' the specified `prudence` level. -#' 4. Extracting the `receptor_schema` and `repertoire_schema` from the loaded -#' metadata. -#' 5. Instantiating a new `ImmunData` object using the loaded `annotations` data -#' and the `receptor_schema`. -#' 6. If a non-empty `repertoire_schema` was found in the metadata, it calls -#' [agg_repertoires()] on the newly created object to recalculate and -#' attach repertoire-level information based on that schema. -#' -#' @return A new `ImmunData` object reconstructed from the saved files. If -#' repertoire information was saved, it will be recalculated and included. -#' -#' @seealso [write_immundata()] for saving `ImmunData` objects, -#' [read_repertoires()] for the primary data loading pipeline, [ImmunData] class, -#' [agg_repertoires()] for repertoire definition. +#' Read either a dataset directory directly or a versioned snapshot within its +#' project home. +#' +#' @section Choose the saved state: +#' +#' To reopen a dataset saved directly in a folder, supply that folder as `path` +#' and leave `tag` and `version` as `NULL`. +#' +#' To reopen a managed snapshot, supply the project home as `path` and its tag. +#' By default, the latest version for that tag is read. Supply `version` when +#' you need an exact earlier state. +#' +#' @section Backend and serialized data: +#' +#' `annotations.parquet` stores the retained chain-level annotation table. +#' It is reopened as a lazy duckplyr table, so the complete table does not need +#' to be loaded into R memory. `metadata.json` stores the format and package +#' versions, receptor, repertoire, and stratum schemas, the repertoire +#' table, the snapshot identifier, lineage events, and provenance paths. +#' +#' Receptor and stratum views are reconstructed from this serialized state; they +#' are not stored as separate files. Please also mind, that the saved files +#' is an ImmunData-specific serialization, not an RDS file. +#' +#' @return A new, disk-backed [ImmunData] object representing the selected saved +#' state. Its provenance records the directory that was read. +#' +#' @seealso [write_immundata()] for saving an analysis, [read_repertoires()] for +#' importing AIRR-seq files, [ImmunData] #' #' @concept ingestion #' @export #' #' @examples -#' \dontrun{ -#' # Assume 'my_idata' is an ImmunData object created previously -#' # my_idata <- read_repertoires(...) +#' library(immundata) +#' library(dplyr) +#' +#' options(immundata.verbose = FALSE) +#' +#' # Create a project home and save a filtered biological state as a snapshot +#' idata <- get_test_idata() +#' project_dir <- tempfile("immundata-project-") #' -#' # Define a temporary directory for saving -#' save_dir <- tempfile("saved_immundata_") +#' project_idata <- write_immundata( +#' idata, +#' output_folder = project_dir, +#' rehome = TRUE +#' ) #' -#' # Save the ImmunData object -#' write_immundata(my_idata, save_dir) +#' fr_response <- project_idata |> +#' filter(Response == "FR") #' -#' # --- Later, in a new session or script --- +#' write_immundata(fr_response, tag = "fr-response") #' -#' # Load the ImmunData object back from the directory -#' loaded_idata <- read_immundata(save_dir) +#' # Read the exact first version of this snapshot +#' continued_fr <- read_immundata( +#' project_dir, +#' tag = "fr-response", +#' version = 1 +#' ) #' -#' # Verify the loaded object -#' print(loaded_idata) -#' # compare_methods(my_idata$annotations, loaded_idata$annotations) # If available +#' continued_fr |> +#' collect() |> +#' summarise( +#' n_chains = n(), +#' n_receptors = n_distinct(imd_receptor_id) +#' ) +#' # Expected result: the snapshot contains the 955 chain rows and 871 +#' # receptors from the FR response group. +#' # n_chains n_receptors +#' # 955 871 #' -#' # Clean up -#' unlink(save_dir, recursive = TRUE) -#' } -read_immundata <- function(path, prudence = "stingy", verbose = TRUE) { - cli_alert_info("Reading ImmunData files from [{.path {path}}]") +#' list.files(file.path(project_dir, "snapshots", "fr-response")) +#' # Expected result: "v001" +#' +#' unlink(project_dir, recursive = TRUE) +read_immundata <- function(path, tag = NULL, version = NULL, prudence = "stingy", + verbose = getOption("immundata.verbose", TRUE)) { + checkmate::assert_character(path, len = 1, null.ok = FALSE) + checkmate::assert_character(tag, len = 1, null.ok = TRUE) + checkmate::assert_count(version, null.ok = TRUE) + checkmate::assert_flag(verbose) + + resolved_path <- resolve_snapshot_input(path, tag = tag, version = version) + if (verbose) { + cli_alert_info("Reading ImmunData files from [{.path {resolved_path}}]") + } + + assert_directory_exists(resolved_path) + assert_file_exists(file.path(resolved_path, imd_files()$annotations)) + assert_file_exists(file.path(resolved_path, imd_files()$metadata)) + + metadata_path <- file.path(resolved_path, imd_files()$metadata) + meta_raw <- jsonlite::read_json( + metadata_path, + simplifyVector = TRUE, + simplifyDataFrame = FALSE, + simplifyMatrix = FALSE + ) + metadata_json <- normalize_metadata_json(meta_raw) - assert_directory_exists(path) - assert_file_exists(file.path(path, imd_files()$annotations)) - assert_file_exists(file.path(path, imd_files()$metadata)) + if (verbose) { + annotation_data <- read_parquet_duckdb( + file.path(resolved_path, imd_files()$annotations), + prudence = prudence + ) + } else { + annotation_data <- suppressMessages(read_parquet_duckdb( + file.path(resolved_path, imd_files()$annotations), + prudence = prudence + )) + } + validate_snapshot_columns(metadata_json, annotation_data, resolved_path) - metadata_json <- jsonlite::read_json(file.path(path, imd_files()$metadata), simplifyVector = T) - annotation_data <- read_parquet_duckdb(file.path(path, imd_files()$annotations), prudence = prudence) + receptor_schema <- metadata_json[["schema_receptor"]] - receptor_schema <- metadata_json[[imd_meta_schema()$receptor_schema]] - # TODO: run checks/repairs: 1) no receptor schema, need to aggregate; 2) wrong columns; 3) receptor schema but no imd_receptor_id + strata_schema <- metadata_json[["schema_strata"]] + repertoire_data <- metadata_json[["repertoires"]] + if (!is.null(repertoire_data)) { + repertoire_data <- duckplyr::as_duckdb_tibble(repertoire_data) + } - repertoire_schema <- metadata_json[[imd_meta_schema()$repertoire_schema]] + strata_data <- NULL + if (!is.null(strata_schema)) { + strata_data <- repertoire_data |> + select(all_of(c( + imd_schema("strata"), + imd_schema("strata_name"), + strata_schema + ))) |> + distinct() + } idata <- ImmunData$new( schema = receptor_schema, - annotations = annotation_data + annotations = annotation_data, + repertoires = repertoire_data, + strata = strata_data ) + if (isTRUE(metadata_json$rebuild_repertoires)) { + idata <- agg_repertoires( + idata, + metadata_json$schema_repertoire, + verbose = verbose + ) + } + if (verbose) { cli_alert_success("Loaded ImmunData with the receptor schema: [{receptor_schema}]") } - if (length(repertoire_schema) > 0) { - idata <- agg_repertoires(idata, repertoire_schema) - + if (!is.null(idata$schema_repertoire) && length(idata$schema_repertoire) > 0) { if (verbose) { - cli_alert_success("Loaded ImmunData with the repertoire schema: [{repertoire_schema}]") + cli_alert_success("Loaded ImmunData with the repertoire schema: [{idata$schema_repertoire}]") } } + if (!is.null(idata$schema_strata) && length(idata$schema_strata) > 0 && verbose) { + cli_alert_success("Loaded ImmunData with the strata schema: [{idata$schema_strata}]") + } + + idata <- set_provenance( + idata, + metadata_json$provenance, + fallback_home_path = resolved_path, + current_path = resolved_path, + snapshot_id = metadata_json$snapshot_id, + lineage = metadata_json$lineage + ) + idata } diff --git a/R/io_immundata_read_utils.R b/R/io_immundata_read_utils.R new file mode 100644 index 0000000..b86c827 --- /dev/null +++ b/R/io_immundata_read_utils.R @@ -0,0 +1,277 @@ +normalize_json_character_field <- function(x) { + if (is.null(x)) { + return(NULL) + } + if (is.character(x)) { + return(unname(x)) + } + if (is.list(x)) { + vals <- unname(unlist(x, recursive = TRUE, use.names = FALSE)) + if (length(vals) == 0) { + return(NULL) + } + return(as.character(vals)) + } + cli::cli_abort("Expected a character field in metadata JSON, got type [{typeof(x)}].") +} + +is_legacy_metadata_v1 <- function(meta_raw) { + has_legacy_schema <- all(c("receptor_schema", "repertoire_schema") %in% names(meta_raw)) + has_v2_schema <- any(c("format_version", "schema_receptor", "snapshot_id", "lineage", "provenance", "extensions") %in% names(meta_raw)) + has_legacy_schema && !has_v2_schema +} + +upgrade_metadata_v1_to_v2 <- function(meta_raw) { + checkmate::assert_true(is_legacy_metadata_v1(meta_raw)) + + cli::cli_warn("Detected legacy v1 metadata.json. Upgrading to v2 in memory for loading.") + + package_version <- normalize_json_character_field(meta_raw$version) + if (is.null(package_version) || length(package_version) == 0) { + package_version <- as.character(packageVersion("immundata")) + } else { + package_version <- package_version[[1]] + } + + receptor_schema <- meta_raw$receptor_schema + checkmate::assert_list(receptor_schema) + checkmate::assert_names(names(receptor_schema), must.include = "features") + receptor_schema$features <- normalize_json_character_field(receptor_schema$features) + checkmate::assert_character(receptor_schema$features, min.len = 1) + receptor_schema$chains <- normalize_json_character_field(receptor_schema$chains) + + list( + format_version = 2L, + package_version = package_version, + schema_receptor = receptor_schema, + schema_repertoire = normalize_json_character_field(meta_raw$repertoire_schema), + schema_strata = NULL, + producer = list("function" = "metadata_upgrade_v1"), + snapshot_id = NULL, + lineage = list(), + provenance = list(), + extensions = list( + legacy = list( + source_format_version = 1L, + source_package_version = package_version + ) + ) + ) +} + +normalize_metadata_v2 <- function(meta_raw) { + required_fields <- c( + "format_version", "package_version", "schema_receptor", "schema_repertoire", + "producer", "snapshot_id", "lineage", "provenance", "extensions" + ) + missing <- setdiff(required_fields, names(meta_raw)) + if (length(missing) > 0) { + cli::cli_abort( + "metadata.json is missing required field(s): [{missing}]." + ) + } + + checkmate::assert_number(meta_raw$format_version, lower = 2, upper = 2) + checkmate::assert_character(meta_raw$package_version, len = 1) + checkmate::assert_list(meta_raw$schema_receptor) + checkmate::assert( + checkmate::test_list(meta_raw$schema_repertoire, null.ok = TRUE), + checkmate::test_character(meta_raw$schema_repertoire, null.ok = TRUE) + ) + checkmate::assert_list(meta_raw$producer) + is_legacy_upgrade <- !is.null(meta_raw$extensions$legacy) + if (is.null(meta_raw$snapshot_id) && !is_legacy_upgrade) { + cli::cli_abort("metadata.json field [snapshot_id] must be a character scalar.") + } + checkmate::assert_character(meta_raw$snapshot_id, len = 1, null.ok = is_legacy_upgrade) + checkmate::assert_list(meta_raw$lineage) + checkmate::assert_list(meta_raw$provenance) + checkmate::assert_list(meta_raw$extensions) + + for (event in meta_raw$lineage) { + checkmate::assert_list(event) + } + + schema_receptor <- meta_raw$schema_receptor + checkmate::assert_names(names(schema_receptor), must.include = "features") + schema_receptor$features <- normalize_json_character_field(schema_receptor$features) + checkmate::assert_character(schema_receptor$features, min.len = 1) + schema_receptor$chains <- normalize_json_character_field(schema_receptor$chains) + meta_raw$schema_receptor <- schema_receptor + + meta_raw$schema_repertoire <- normalize_json_character_field(meta_raw$schema_repertoire) + has_serialized_repertoires <- "repertoires" %in% names(meta_raw) + meta_raw$schema_strata <- normalize_json_character_field(meta_raw$schema_strata) + if (!"schema_strata" %in% names(meta_raw)) { + meta_raw$schema_strata <- NULL + } + if (!is.null(meta_raw$schema_strata)) { + checkmate::assert_character(meta_raw$schema_strata) + } + + if (has_serialized_repertoires) { + checkmate::assert_list(meta_raw$repertoires, null.ok = TRUE) + } else { + meta_raw$repertoires <- NULL + } + if (!is.null(meta_raw$repertoires)) { + meta_raw$repertoires <- as.data.frame( + meta_raw$repertoires, + stringsAsFactors = FALSE, + optional = TRUE, + check.names = FALSE + ) + } + + if (!is.null(meta_raw$schema_repertoire) && has_serialized_repertoires && is.null(meta_raw$repertoires)) { + cli::cli_abort( + "Snapshot declares a repertoire schema but does not contain serialized repertoire data." + ) + } + if (is.null(meta_raw$schema_repertoire) && !is.null(meta_raw$repertoires)) { + cli::cli_abort( + "Snapshot contains serialized repertoire data but does not declare a repertoire schema." + ) + } + if (!is.null(meta_raw$schema_strata) && is.null(meta_raw$schema_repertoire)) { + cli::cli_abort( + "Snapshot declares a strata schema but does not declare a repertoire schema." + ) + } + + # Old v2 snapshots can contain duplicated snapshot state in provenance. + # Preserve path information only; top-level metadata remains canonical. + meta_raw$provenance <- provenance_paths_for_metadata(meta_raw$provenance) + meta_raw$rebuild_repertoires <- !has_serialized_repertoires && + !is.null(meta_raw$schema_repertoire) + meta_raw +} + +normalize_metadata_json <- function(meta_raw) { + checkmate::assert_list(meta_raw) + + if (is_legacy_metadata_v1(meta_raw)) { + return(normalize_metadata_v2(upgrade_metadata_v1_to_v2(meta_raw))) + } + + if (!"format_version" %in% names(meta_raw)) { + cli::cli_abort("metadata.json is missing required field(s): [format_version].") + } + checkmate::assert_count(meta_raw$format_version) + if (!identical(as.integer(meta_raw$format_version), 2L)) { + cli::cli_abort( + "Unsupported ImmunData snapshot format version [{meta_raw$format_version}]. Supported versions: [1, 2]." + ) + } + + normalize_metadata_v2(meta_raw) +} + +validate_snapshot_columns <- function(metadata_json, annotation_data, snapshot_path) { + annotation_columns <- colnames(annotation_data) + issues <- character() + + receptor_schema <- metadata_json$schema_receptor + required_annotation_columns <- c( + imd_schema("receptor"), + imd_schema("barcode"), + imd_schema("chain"), + imd_schema("chain_count"), + imd_receptor_features(receptor_schema) + ) + if (length(imd_receptor_chains(receptor_schema)) == 2) { + required_annotation_columns <- c(required_annotation_columns, imd_schema("locus")) + } + + issues <- c( + issues, + format_missing_columns_issue( + setdiff(unique(required_annotation_columns), annotation_columns), + "annotations.parquet" + ) + ) + + repertoire_schema <- metadata_json$schema_repertoire + repertoire_data <- metadata_json$repertoires + if (!is.null(repertoire_schema)) { + issues <- c( + issues, + format_missing_columns_issue( + setdiff( + c( + repertoire_schema, + imd_schema("repertoire"), + imd_schema("count"), + imd_schema("proportion"), + imd_schema("n_repertoires") + ), + annotation_columns + ), + "annotations.parquet for the declared repertoire schema" + ) + ) + if (!isTRUE(metadata_json$rebuild_repertoires)) { + issues <- c( + issues, + format_missing_columns_issue( + setdiff( + c( + repertoire_schema, + imd_schema("repertoire"), + imd_schema("n_barcodes"), + imd_schema("n_receptors") + ), + colnames(repertoire_data) + ), + "metadata.json repertoires" + ) + ) + } + } + + strata_schema <- metadata_json$schema_strata + if (!is.null(strata_schema)) { + issues <- c( + issues, + format_missing_columns_issue( + setdiff(imd_schema("strata"), annotation_columns), + "annotations.parquet for the declared strata schema" + ), + format_missing_columns_issue( + setdiff( + c( + strata_schema, + imd_schema("strata"), + imd_schema("strata_name") + ), + colnames(repertoire_data) + ), + "metadata.json repertoires for the declared strata schema" + ) + ) + } + + if (length(issues) > 0) { + cli::cli_abort(c( + "Cannot load ImmunData snapshot because its schema is inconsistent.", + stats::setNames(issues, rep("x", length(issues))), + "i" = "Snapshot: {.path {snapshot_path}}", + "i" = "No data was loaded. Recreate the snapshot or correct its declared schema." + )) + } + + invisible(TRUE) +} + +format_missing_columns_issue <- function(missing, location) { + if (length(missing) == 0) { + return(character()) + } + + paste0( + location, + " is missing required column(s): ", + paste(missing, collapse = ", "), + "." + ) +} diff --git a/R/io_immundata_write.R b/R/io_immundata_write.R index a822b0a..15390b8 100644 --- a/R/io_immundata_write.R +++ b/R/io_immundata_write.R @@ -1,100 +1,123 @@ -#' @title Save ImmunData to disk +#' @title Save an ImmunData object to disk #' #' @description -#' Serializes the essential components of an `ImmunData` object to disk for -#' efficient storage and later retrieval. It saves the core annotation data -#' (`idata$annotations`) as a compressed Parquet file and accompanying metadata -#' (including receptor/repertoire schemas and package version) as a JSON file -#' within a specified directory. -#' -#' @param idata The `ImmunData` object to save. Must be an R6 object of class -#' `ImmunData` containing at least the `$annotations` table and schema information -#' (`$schema_receptor`, optionally `$schema_repertoire`). -#' @param output_folder Character(1). Path to the directory where the output files -#' will be written. If the directory does not exist, it will be created -#' recursively. +#' Save `ImmunData` to disk so you can close R and continue the work later (I cannot +#' believe it, but it works, I tried it). Use +#' `write_immundata()` after importing or transforming repertoire data, or when +#' you want a named snapshot before the next analysis step. +#' +#' The unit saved is the complete [ImmunData] object. This includes retained +#' chain rows, cell and receptor identifiers, repertoire and stratum definitions, +#' and provenance. Saving does not add, remove, or change any biological unit. +#' +#' @param idata An [ImmunData] object you want to save. +#' @param output_folder A character string or `NULL`. Directory in which to +#' write `annotations.parquet` and `metadata.json`. If `NULL`, the default, a +#' managed snapshot is created at `home_path/snapshots//vNNN`. The home +#' path comes from the object's provenance. +#' @param tag A character string or `NULL`. Snapshot tag. With +#' `output_folder = NULL`, it names the managed snapshot series; if `tag` is +#' also `NULL`, `"default"` is used. With an explicit `output_folder`, a +#' supplied tag is recorded in the lineage but does not change the output +#' path. +#' @param rehome A logical value. Whether an explicit `output_folder` becomes +#' the home for future managed snapshots. The default is `FALSE`, which +#' preserves an existing home. If the object has no home yet, its first +#' explicit output folder becomes the home with either value. `TRUE` requires +#' an explicit `output_folder`. +#' @param compression A character string or `NULL`. Parquet compression codec +#' passed to DuckDB. The default is `"zstd"`. Use `NULL` to let DuckDB choose. +#' @param compression_level A number or `NULL`. Compression level for codecs +#' that support it. The default is `9`. Use `NULL` to let DuckDB choose. +#' @param verbose A logical value. Whether to print progress and summary +#' messages. Defaults to `getOption("immundata.verbose", TRUE)`. #' #' @details -#' The function performs the following actions: -#' 1. Validates the input `idata` object and `output_folder` path. -#' 2. Creates the `output_folder` if it doesn't exist. -#' 3. Constructs a list containing metadata: `immundata` package version, -#' receptor schema (`idata$schema_receptor`), and repertoire schema -#' (`idata$schema_repertoire`). -#' 4. Writes the metadata list to `metadata.json` within `output_folder`. -#' 5. Writes the `idata$annotations` table (a `duckplyr_df` or similar) to -#' `annotations.parquet` within `output_folder`. Uses Zstandard compression -#' (`compression = "zstd"`, `compression_level = 9`) for a good balance -#' between file size and read/write speed. -#' 6. Uses internal helper `imd_files()` to determine the standard filenames -#' (`metadata.json`, `annotations.parquet`). -#' -#' The receptor data itself (if stored separately in future versions) is not -#' saved by this function; only the annotations linking to receptors are saved, -#' along with the schema needed to reconstruct/interpret them. -#' -#' @return -#' Invisibly returns the input `idata` object, saved to disk. -#' In other words, this allows you to create snapshots of the data in the -#' `output_folder`. Mind that by saving the object, you execute all the -#' stored computations, so this operations can take longer than expected. -#' Read more about snapshots on our website in the ["Concept" section](https://immunomind.github.io/docs/concepts/basics/immutability/). -#' -#' @seealso [read_immundata()] for loading the saved data, [read_repertoires()] -#' which uses this function internally, [ImmunData] class definition. +#' Save to an explicit folder for a direct saved state, or use the object's home +#' to create a versioned managed snapshot. +#' +#' @section Choose how to save: +#' +#' Supply `output_folder` to save a standalone state in a specific directory. +#' This is useful when sharing a dataset or choosing its first project home. +#' If the directory already contains an ImmunData dataset, its +#' `annotations.parquet` and `metadata.json` are replaced. +#' +#' Leave `output_folder = NULL` to create a managed snapshot. The function uses +#' the object's home path and writes the next version under +#' `snapshots//vNNN`, for example `snapshots/baseline/v001`. Later writes +#' with the same tag create `v002`, `v003`, and so on; earlier versions remain +#' available. Use [read_immundata()] with `tag` and `version` to reopen one. +#' +#' Every save receives a new snapshot identifier and appends a provenance event. +#' The returned object records the new saved directory as its current path. +#' +#' @section Backend and serialization: +#' +#' The retained chain-level annotation table is materialized as compressed +#' `annotations.parquet`. Materialization executes any pending lazy duckplyr +#' calculations. `metadata.json` serializes format and package versions, +#' receptor, repertoire, and stratum schemas, the small repertoire table, the +#' snapshot identifier, lineage events, and provenance paths. +#' +#' Receptor and stratum views are not written as separate files; they can be +#' reconstructed from the annotation table and metadata. This Parquet and JSON +#' pair is an ImmunData-specific serialization, not an RDS file. +#' +#' @return Invisibly returns a newly reopened, disk-backed [ImmunData] object +#' with provenance for the new save. The input `idata` remains unchanged. +#' +#' @seealso [read_immundata()] for continuing a saved analysis, +#' [read_repertoires()] for importing AIRR-seq files, [ImmunData] #' #' @concept ingestion #' @export #' #' @examples -#' \dontrun{ -#' # Assume 'my_idata' is an ImmunData object created previously -#' # my_idata <- read_repertoires(...) +#' library(immundata) +#' library(dplyr) +#' +#' options(immundata.verbose = FALSE) #' -#' # Define an output directory -#' save_dir <- tempfile("saved_immundata_") +#' # Save a small immune-repertoire analysis +#' idata <- get_test_idata() +#' save_dir <- tempfile("saved-immundata-") #' -#' # Save the ImmunData object -#' write_immundata(my_idata, save_dir) +#' saved_idata <- write_immundata(idata, save_dir) #' -#' # Check the created files -#' list.files(save_dir) # Should show "annotations.parquet" and "metadata.json" +#' list.files(save_dir) +#' # Expected result: the analysis is serialized as two files. +#' # [1] "annotations.parquet" "metadata.json" +#' +#' # Continue the analysis from the saved files +#' continued_idata <- read_immundata(save_dir) +#' +#' continued_idata |> +#' collect() |> +#' summarise( +#' n_chains = n(), +#' n_receptors = n_distinct(imd_receptor_id) +#' ) +#' # Expected result: all 1,902 chain rows and 1,668 receptors are restored. +#' # n_chains n_receptors +#' # 1902 1668 #' -#' # Clean up #' unlink(save_dir, recursive = TRUE) -#' } -write_immundata <- function(idata, output_folder) { - checkmate::assert_r6(idata, "ImmunData") - checkmate::assert_character(output_folder, - max.len = 1, - null.ok = FALSE - ) - - output_folder <- normalizePath(output_folder, mustWork = FALSE) - dir.create(output_folder, showWarnings = FALSE, recursive = TRUE) - - metadata_path <- file.path(output_folder, imd_files()$metadata) - annotations_path <- file.path(output_folder, imd_files()$annotations) - - metadata_json <- list( - version = as.character(packageVersion("immundata")), - receptor_schema = idata$schema_receptor, - repertoire_schema = idata$schema_repertoire - ) - - cli::cli_alert_info("Writing the receptor annotation data to [{annotations_path}]") - compute_parquet(idata$annotations, - annotations_path, - options = list( - compression = "zstd", - compression_level = 9 - ) +write_immundata <- function(idata, + output_folder = NULL, + tag = NULL, + rehome = FALSE, + compression = "zstd", + compression_level = 9, + verbose = getOption("immundata.verbose", TRUE)) { + write_immundata_internal( + idata = idata, + output_folder = output_folder, + snapshot_tag = tag, + rehome = rehome, + compression = compression, + compression_level = compression_level, + producer_function = "write_immundata", + verbose = verbose ) - - cli::cli_alert_info("Writing the metadata to [{metadata_path}]") - jsonlite::write_json(metadata_json, metadata_path) - - cli::cli_alert_success("ImmunData files saved to [{output_folder}]") - - invisible(read_immundata(output_folder)) } diff --git a/R/io_immundata_write_internal.R b/R/io_immundata_write_internal.R new file mode 100644 index 0000000..af76e9b --- /dev/null +++ b/R/io_immundata_write_internal.R @@ -0,0 +1,140 @@ +#' @title Internal writer for ImmunData snapshots +#' @description Internal helper used by `write_immundata()` and `read_repertoires()` +#' to write `metadata.json` (including repertoires) and `annotations.parquet`. +#' @keywords internal +#' @noRd +write_immundata_internal <- function(idata, + output_folder = NULL, + snapshot_tag = NULL, + rehome = FALSE, + compression = "zstd", + compression_level = 9, + producer_function = "write_immundata", + ingestion_payload = NULL, + metadata_extensions = NULL, + verbose = getOption("immundata.verbose", TRUE)) { + compression_was_provided <- !missing(compression) + compression_level_was_provided <- !missing(compression_level) + + checkmate::assert_r6(idata, "ImmunData") + checkmate::assert_character(output_folder, + max.len = 1, + null.ok = TRUE + ) + checkmate::assert_character(snapshot_tag, + max.len = 1, + null.ok = TRUE + ) + checkmate::assert_flag(rehome) + checkmate::assert_character(producer_function, + len = 1, + null.ok = FALSE + ) + checkmate::assert_character(compression, + max.len = 1, + null.ok = TRUE + ) + checkmate::assert_numeric(compression_level, + len = 1, + null.ok = TRUE + ) + checkmate::assert_list(ingestion_payload, null.ok = TRUE) + checkmate::assert_list(metadata_extensions, null.ok = TRUE) + checkmate::assert_flag(verbose) + + resolved_output <- resolve_snapshot_output_folder( + idata = idata, + output_folder = output_folder, + tag = snapshot_tag, + rehome = rehome + ) + output_folder <- resolved_output$output_folder + snapshot_tag <- resolved_output$tag + provenance_before <- resolved_output$provenance + dir.create(output_folder, showWarnings = FALSE, recursive = TRUE) + + metadata_path <- file.path(output_folder, imd_files()$metadata) + annotations_path <- file.path(output_folder, imd_files()$annotations) + + snapshot_metadata <- build_snapshot_metadata( + idata = idata, + producer_function = producer_function, + provenance_before = provenance_before, + output_folder = output_folder, + snapshot_tag = snapshot_tag, + rehome = rehome, + ingestion_payload = ingestion_payload, + metadata_extensions = metadata_extensions + ) + metadata_json <- snapshot_metadata$metadata + provenance_after <- snapshot_metadata$provenance + + if (verbose) { + cli::cli_alert_info("Writing the receptor annotation data to [{annotations_path}]") + } + duckplyr_is_1_2_0 <- isTRUE(utils::packageVersion("duckplyr") == "1.2.0") + parquet_options <- Filter( + f = \(x) !is.null(x), + x = list( + compression = compression, + compression_level = compression_level + ) + ) + + if (duckplyr_is_1_2_0) { + if (verbose && (compression_was_provided || compression_level_was_provided)) { + cli::cli_alert_warning( + "duckplyr 1.2.0 does not accept compression options in `compute_parquet()`; ignoring `compression` and `compression_level`." + ) + } + if (verbose) { + compute_parquet( + idata$annotations, + annotations_path + ) + } else { + suppressMessages(compute_parquet( + idata$annotations, + annotations_path + )) + } + } else if (length(parquet_options) == 0) { + if (verbose) { + compute_parquet( + idata$annotations, + annotations_path + ) + } else { + suppressMessages(compute_parquet( + idata$annotations, + annotations_path + )) + } + } else { + if (verbose) { + compute_parquet( + idata$annotations, + annotations_path, + options = parquet_options + ) + } else { + suppressMessages(compute_parquet( + idata$annotations, + annotations_path, + options = parquet_options + )) + } + } + + if (verbose) { + cli::cli_alert_info("Writing the metadata to [{metadata_path}]") + } + jsonlite::write_json(metadata_json, metadata_path, null = "null", auto_unbox = TRUE, pretty = TRUE) + + if (verbose) { + cli::cli_alert_success("ImmunData files saved to [{output_folder}]") + } + + written_idata <- read_immundata(output_folder, verbose = FALSE) + invisible(set_provenance(written_idata, provenance_after)) +} diff --git a/R/io_immundata_write_utils.R b/R/io_immundata_write_utils.R new file mode 100644 index 0000000..55e2e30 --- /dev/null +++ b/R/io_immundata_write_utils.R @@ -0,0 +1,70 @@ +generate_snapshot_id <- function() { + suffix <- paste0(sample(c(letters, 0:9), 8L, replace = TRUE), collapse = "") + paste0("imd_", format(Sys.time(), "%Y%m%dT%H%M%SZ", tz = "UTC"), "_", suffix) +} + +build_snapshot_metadata <- function(idata, + producer_function, + provenance_before, + output_folder, + snapshot_tag = NULL, + rehome = FALSE, + ingestion_payload = NULL, + metadata_extensions = NULL) { + serialized_repertoires <- idata$repertoires + if (!is.null(serialized_repertoires)) { + checkmate::assert_data_frame(serialized_repertoires) + serialized_repertoires <- as.list(serialized_repertoires) + factor_columns <- vapply(serialized_repertoires, is.factor, logical(1)) + serialized_repertoires[factor_columns] <- lapply( + serialized_repertoires[factor_columns], + as.character + ) + } + + snapshot_id <- generate_snapshot_id() + is_ingestion <- identical(producer_function, "read_repertoires") + event <- list( + event = if (is_ingestion) "ingestion" else "snapshot", + created_at = format(Sys.time(), "%Y-%m-%dT%H:%M:%SZ", tz = "UTC"), + snapshot_id = snapshot_id, + producer = list("function" = producer_function) + ) + if (is_ingestion) { + event <- c(event, ingestion_payload) + } else { + event$source_path <- provenance_before$current_path + event$snapshot_path <- output_folder + event$tag <- snapshot_tag + } + + lineage <- c(provenance_before$lineage, list(event)) + home_path <- provenance_before$home_path + if (is.null(home_path) || is_ingestion || isTRUE(rehome)) { + home_path <- output_folder + } + provenance_after <- normalize_provenance( + provenance_before, + home_path = home_path, + current_path = output_folder, + snapshot_id = snapshot_id, + lineage = lineage + ) + + list( + metadata = list( + format_version = 2L, + package_version = as.character(packageVersion("immundata")), + schema_receptor = idata$schema_receptor, + schema_repertoire = idata$schema_repertoire, + schema_strata = idata$schema_strata, + repertoires = serialized_repertoires, + producer = list("function" = producer_function), + snapshot_id = snapshot_id, + lineage = lineage, + provenance = provenance_paths_for_metadata(provenance_after), + extensions = if (is.null(metadata_extensions)) list() else metadata_extensions + ), + provenance = provenance_after + ) +} diff --git a/R/io_immundata_write_utils_provenance.R b/R/io_immundata_write_utils_provenance.R new file mode 100644 index 0000000..cbcbff2 --- /dev/null +++ b/R/io_immundata_write_utils_provenance.R @@ -0,0 +1,178 @@ +default_provenance <- function() { + list( + home_path = NULL, + current_path = NULL, + snapshot_root = NULL, + artifacts_root = NULL, + artifacts_path = NULL, + snapshot_id = NULL, + lineage = list() + ) +} + +normalize_nullable_path <- function(path) { + if (is.null(path)) { + return(NULL) + } + + checkmate::assert_character(path, len = 1, null.ok = FALSE) + normalizePath(path, mustWork = FALSE) +} + +resolve_artifacts_path <- function(home_path, + current_path, + snapshot_root, + snapshot_id = NULL) { + if (is.null(home_path) || is.null(current_path) || is.null(snapshot_root)) { + return(NULL) + } + + home_path <- normalizePath(home_path, mustWork = FALSE) + current_path <- normalizePath(current_path, mustWork = FALSE) + snapshot_root <- normalizePath(snapshot_root, mustWork = FALSE) + artifacts_root <- normalizePath(file.path(home_path, "artifacts"), mustWork = FALSE) + + if (identical(current_path, home_path)) { + return(normalizePath(file.path(artifacts_root, "root"), mustWork = FALSE)) + } + + is_managed_snapshot <- grepl("^v[0-9]+$", basename(current_path)) && + identical( + normalizePath(dirname(dirname(current_path)), mustWork = FALSE), + snapshot_root + ) + if (is_managed_snapshot) { + return(normalizePath( + file.path(artifacts_root, basename(dirname(current_path)), basename(current_path)), + mustWork = FALSE + )) + } + + if (!is.null(snapshot_id)) { + return(normalizePath( + file.path(artifacts_root, "by-id", snapshot_id), + mustWork = FALSE + )) + } + + NULL +} + +normalize_provenance <- function(provenance = NULL, + fallback_home_path = NULL, + home_path = NULL, + current_path = NULL, + snapshot_id = NULL, + lineage = NULL) { + if (is.null(provenance)) { + provenance <- list() + } + + # Validate every supplied value before resolving precedence or deriving paths. + checkmate::assert_list(provenance) + + path_fields <- c( + "home_path", "current_path", "snapshot_root", + "artifacts_root", "artifacts_path" + ) + for (field in path_fields) { + checkmate::assert_character(provenance[[field]], len = 1, null.ok = TRUE) + } + checkmate::assert_character(fallback_home_path, len = 1, null.ok = TRUE) + checkmate::assert_character(home_path, len = 1, null.ok = TRUE) + checkmate::assert_character(current_path, len = 1, null.ok = TRUE) + checkmate::assert_character(provenance$snapshot_id, len = 1, null.ok = TRUE) + checkmate::assert_character(snapshot_id, len = 1, null.ok = TRUE) + checkmate::assert_list(provenance$lineage, null.ok = TRUE) + checkmate::assert_list(lineage, null.ok = TRUE) + for (event in c(provenance$lineage, lineage)) { + checkmate::assert_list(event) + } + + # Explicit arguments override stored provenance. The fallback is used only + # when neither supplies a home path, as with legacy metadata. + resolved_home_path <- if (!is.null(home_path)) { + home_path + } else if (!is.null(provenance$home_path)) { + provenance$home_path + } else { + fallback_home_path + } + resolved_current_path <- if (!is.null(current_path)) { + current_path + } else { + provenance$current_path + } + resolved_snapshot_id <- if (!is.null(snapshot_id)) { + snapshot_id + } else { + provenance$snapshot_id + } + resolved_lineage <- if (!is.null(lineage)) { + lineage + } else if (!is.null(provenance$lineage)) { + provenance$lineage + } else { + list() + } + + resolved_home_path <- normalize_nullable_path(resolved_home_path) + resolved_current_path <- normalize_nullable_path(resolved_current_path) + + # These locations are derived from the canonical home/current state and are + # never accepted as independent sources of truth. + snapshot_root <- if (is.null(resolved_home_path)) { + NULL + } else { + normalizePath(file.path(resolved_home_path, "snapshots"), mustWork = FALSE) + } + artifacts_root <- if (is.null(resolved_home_path)) { + NULL + } else { + normalizePath(file.path(resolved_home_path, "artifacts"), mustWork = FALSE) + } + artifacts_path <- resolve_artifacts_path( + home_path = resolved_home_path, + current_path = resolved_current_path, + snapshot_root = snapshot_root, + snapshot_id = resolved_snapshot_id + ) + + list( + home_path = resolved_home_path, + current_path = resolved_current_path, + snapshot_root = snapshot_root, + artifacts_root = artifacts_root, + artifacts_path = artifacts_path, + snapshot_id = resolved_snapshot_id, + lineage = resolved_lineage + ) +} + +provenance_paths_for_metadata <- function(provenance) { + # Snapshot identity and lineage are canonical top-level metadata fields. + # Keep only normalized location fields in metadata$provenance. + path_fields <- c( + "home_path", "current_path", "snapshot_root", + "artifacts_root", "artifacts_path" + ) + normalize_provenance(provenance)[path_fields] +} + +get_provenance <- function(idata) { + checkmate::assert_r6(idata, "ImmunData") + private_env <- idata$.__enclos_env__$private + raw <- private_env$.provenance + if (is.null(raw)) { + return(default_provenance()) + } + + raw +} + +set_provenance <- function(idata, provenance, ...) { + checkmate::assert_r6(idata, "ImmunData") + normalized <- normalize_provenance(provenance, ...) + idata$.__enclos_env__$private$.provenance <- normalized + invisible(idata) +} diff --git a/R/io_immundata_write_utils_snapshots.R b/R/io_immundata_write_utils_snapshots.R new file mode 100644 index 0000000..3e21367 --- /dev/null +++ b/R/io_immundata_write_utils_snapshots.R @@ -0,0 +1,172 @@ +validate_snapshot_tag <- function(tag) { + checkmate::assert_character(tag, len = 1, null.ok = FALSE) + tag <- trimws(tag) + if (identical(tag, "")) { + cli::cli_abort("Snapshot {.arg tag} must be a non-empty string.") + } + + if (tag %in% c(".", "..") || grepl("[/\\\\]", tag)) { + cli::cli_abort("Snapshot {.arg tag} must not include path separators or reserved values '.'/'..'.") + } + + if (identical(tolower(tag), "root")) { + cli::cli_abort( + "Snapshot {.arg tag} [root] is reserved for the original ingestion state." + ) + } + + if (!grepl("^[A-Za-z0-9._-]+$", tag)) { + cli::cli_abort( + "Snapshot {.arg tag} may only contain letters, numbers, dot, underscore, and dash." + ) + } + + tag +} + +format_snapshot_version <- function(version) { + checkmate::assert_count(version) + sprintf("v%03d", as.integer(version)) +} + +list_snapshot_versions <- function(tag_dir) { + if (!dir.exists(tag_dir)) { + return(integer()) + } + + children <- list.files(tag_dir, full.names = FALSE, recursive = FALSE, all.files = FALSE) + version_dirnames <- children[grepl("^v[0-9]+$", children)] + versions <- as.integer(sub("^v", "", version_dirnames)) + sort(unique(versions)) +} + +list_snapshot_tags <- function(home_path) { + snapshot_root <- file.path(home_path, "snapshots") + if (!dir.exists(snapshot_root)) { + return(character()) + } + + tags <- list.files(snapshot_root, full.names = FALSE, recursive = FALSE, all.files = FALSE) + tags[file.info(file.path(snapshot_root, tags))$isdir %in% TRUE] |> sort() +} + +resolve_snapshot_version <- function(home_path, tag, version = NULL, allocate = FALSE) { + checkmate::assert_character(home_path, len = 1, null.ok = FALSE) + checkmate::assert_character(tag, len = 1, null.ok = FALSE) + checkmate::assert_count(version, null.ok = TRUE) + checkmate::assert_flag(allocate) + + home_path <- normalizePath(home_path, mustWork = FALSE) + tag <- validate_snapshot_tag(tag) + tag_dir <- file.path(home_path, "snapshots", tag) + + if (allocate) { + dir.create(tag_dir, recursive = TRUE, showWarnings = FALSE) + versions <- list_snapshot_versions(tag_dir) + next_version <- if (length(versions) == 0) 1L else max(versions) + 1L + return(file.path(tag_dir, format_snapshot_version(next_version))) + } + + if (!dir.exists(tag_dir)) { + available_tags <- list_snapshot_tags(home_path) + if (length(available_tags) == 0) { + cli::cli_abort( + "Snapshot tag [{tag}] was not found under [{home_path}/snapshots]. No snapshot tags are available." + ) + } + cli::cli_abort( + "Snapshot tag [{tag}] was not found under [{home_path}/snapshots]. Available tags: [{available_tags}]." + ) + } + + available_versions <- list_snapshot_versions(tag_dir) + if (length(available_versions) == 0) { + cli::cli_abort( + "Snapshot tag [{tag}] exists under [{tag_dir}] but has no version directories (expected vNNN)." + ) + } + + if (is.null(version)) { + version <- max(available_versions) + } + if (!version %in% available_versions) { + formatted <- format_snapshot_version(available_versions) + cli::cli_abort( + "Snapshot version [{format_snapshot_version(version)}] was not found for tag [{tag}]. Available versions: [{formatted}]." + ) + } + + file.path(tag_dir, format_snapshot_version(version)) +} + +resolve_snapshot_input <- function(path, tag = NULL, version = NULL) { + checkmate::assert_character(path, len = 1, null.ok = FALSE) + checkmate::assert_character(tag, len = 1, null.ok = TRUE) + checkmate::assert_count(version, null.ok = TRUE) + + path <- normalizePath(path, mustWork = FALSE) + if (!is.null(version) && is.null(tag)) { + cli::cli_abort("`version` can only be used together with {.arg tag}.") + } + if (is.null(tag)) { + return(path) + } + path_is_snapshot_version <- grepl("^v[0-9]+$", basename(path)) && + identical(basename(dirname(dirname(path))), "snapshots") + if (path_is_snapshot_version) { + cli::cli_abort( + "Path [{path}] already points to a concrete snapshot version folder; do not combine it with {.arg tag}/{.arg version}." + ) + } + + resolve_snapshot_version(path, tag, version, allocate = FALSE) +} + +resolve_snapshot_output_folder <- function(idata, + output_folder = NULL, + tag = NULL, + rehome = FALSE) { + checkmate::assert_r6(idata, "ImmunData") + checkmate::assert_character(output_folder, len = 1, null.ok = TRUE) + checkmate::assert_character(tag, len = 1, null.ok = TRUE) + checkmate::assert_flag(rehome) + + provenance <- get_provenance(idata) + + if (!is.null(output_folder)) { + return(list( + output_folder = normalizePath(output_folder, mustWork = FALSE), + tag = if (is.null(tag)) NULL else validate_snapshot_tag(tag), + provenance = provenance, + output_was_auto = FALSE + )) + } + + if (rehome) { + cli::cli_abort("`rehome = TRUE` requires an explicit {.arg output_folder}.") + } + + if (is.null(provenance$home_path)) { + cli::cli_abort( + "Cannot infer snapshot home path from `idata`. Please provide {.arg output_folder} or load data using `read_immundata()` / `read_repertoires()` first." + ) + } + + if (is.null(tag)) { + tag <- "default" + } + tag <- validate_snapshot_tag(tag) + + snapshot_folder <- resolve_snapshot_version( + home_path = provenance$home_path, + tag = tag, + allocate = TRUE + ) + + list( + output_folder = normalizePath(snapshot_folder, mustWork = FALSE), + tag = tag, + provenance = provenance, + output_was_auto = TRUE + ) +} diff --git a/R/io_manifest_read.R b/R/io_manifest_read.R new file mode 100644 index 0000000..6393958 --- /dev/null +++ b/R/io_manifest_read.R @@ -0,0 +1,122 @@ +#' @title Load and Validate a Manifest for Immune Repertoire Files +#' +#' @description +#' This function loads a manifest from either a file path or a data frame, +#' validates the presence of a column with repertoire file paths, and converts +#' all file paths to absolute paths. It is used to support flexible pipelines +#' for loading bulk or single-cell immune repertoire data across samples. +#' +#' If the input is a file path, the function reads it with `readr::read_delim`. +#' If the input is a data frame, it checks whether file paths are absolute; +#' relative paths are only allowed when the manifest is loaded from a file. +#' +#' It warns the user if many of the files listed in the manifest are missing, +#' and stops execution if none of the files exist. +#' +#' The column with file paths is normalized into the internal filename schema. +#' +#' @param manifest A manifest table. Can be either: +#' - a data frame with per-file annotations, +#' - or a path to a CSV/TSV/TXT manifest file. +#' +#' @param file_col A string specifying the name of the column in the manifest +#' that contains paths to repertoire files. Defaults to `"file"`. +#' +#' @param delim Delimiter used to read the manifest file. If `NULL`, it is +#' inferred from the extension: comma for `.csv`, tab for `.tsv` and `.txt`. +#' +#' @param verbose Logical(1). Whether to print informative messages. Defaults to +#' `getOption("immundata.verbose", TRUE)`. +#' +#' @param ... Additional arguments passed to `readr::read_delim()` when reading +#' a manifest from a file. +#' +#' @return A validated and updated manifest data frame with absolute file paths +#' and an additional internal column named `imd_filename`. +#' +#' @concept ingestion +#' @export +read_manifest <- function(manifest, file_col = "file", delim = NULL, ..., + verbose = getOption("immundata.verbose", TRUE)) { + checkmate::assert_flag(verbose) + + if (!checkmate::test_data_frame(manifest) && !checkmate::test_file_exists(manifest)) { + cli_abort("Error in manifest: the input manifest should be either a data frame or an existing file.") + } + + manifest_source <- NA + if (checkmate::test_file_exists(manifest)) { + manifest_basename <- tolower(basename(manifest)) + if (manifest_basename %in% c("metadata.csv", "metadata.tsv", "metadata.txt")) { + cli_abort("Input repertoire metadata tables are now manifests. Rename [{basename(manifest)}] to [manifest.csv] and use {.fn read_manifest}. Snapshot metadata.json is not affected.") + } + + if (is.null(delim)) { + extension <- tolower(tools::file_ext(manifest)) + delim <- switch(extension, + csv = ",", + tsv = "\t", + txt = "\t", + cli_abort("Unknown manifest file type: [{extension}]. Supported manifest types: CSV, TSV, TXT.") + ) + } + + if (verbose) { + manifest_table <- read_delim(manifest, delim = delim, ...) + } else { + manifest_table <- suppressMessages(read_delim(manifest, delim = delim, ...)) + } + manifest_source <- "file" # TODO: enum + } else { + manifest_table <- manifest + manifest_source <- "df" + } + + # Check for the column + if (!(file_col %in% colnames(manifest_table))) { + cli_abort("Error: no column [{file_col}] with full file paths and names in the input manifest.") + } + + # Preprocess the files: + # - either they are in the same folder as manifest + # - or those are full paths to the files + manifest_table[[file_col]] <- sapply(manifest_table[[file_col]], function (path) { + if (!grepl("^(?:[A-Za-z]:/|/)", path)) { + # The path is not a full path, so we should add the manifest directory. + if (manifest_source == "file") { + path <- file.path(dirname(manifest), path) + } else { + cli_abort("Error: the input manifest is a data frame, but paths are relative. + Provide full paths in the [{file_col}] column, e.g., + [/Users/username/projects/data/sample1.tsv] instead of [sample1.tsv] or [sample1]") + } + } + + normalizePath(path) + }) + names(manifest_table[[file_col]]) <- NULL + + # Check how many files from the file column exist: + file_list <- manifest_table[[file_col]] + file_existed <- sapply(file_list, test_file_exists) + n_existed <- sum(file_existed) + n_threshold <- round(length(file_list) * 0.1) + 1 + + if (verbose) { + cli_alert_info("Found {n_existed}/{length(file_list)} repertoire files from the manifest on disk") + } + if (n_existed == 0) { + cli_abort("Error: found zero (!) repertoire files passed in the manifest. Are the file paths in the manifest correct?") + } else if (n_existed <= n_threshold && verbose) { + cli_alert_warning("Warning: found only {n_existed} files out of {length(file_list)} in the manifest. Please check if you planned to work with more repertoire files. Continuing the execution.") + } + + if (verbose) { + cli_alert_success("Manifest parsed successfully") + } + + immundata_filename_col <- IMD_GLOBALS$schema$manifest_filename + manifest_table[[immundata_filename_col]] <- manifest_table[[file_col]] + + manifest_table +} diff --git a/R/io_metadata_read.R b/R/io_metadata_read.R deleted file mode 100644 index 04c192a..0000000 --- a/R/io_metadata_read.R +++ /dev/null @@ -1,94 +0,0 @@ -#' @title Load and Validate Metadata Table for Immune Repertoire Files -#' -#' @description -#' This function loads a metadata table from either a file path or a data frame, -#' validates the presence of a column with repertoire file paths, and converts all -#' file paths to absolute paths. It is used to support flexible pipelines for -#' loading bulk or single-cell immune repertoire data across samples. -#' -#' If the input is a file path, the function attempts to read it with `readr::read_delim`. -#' If the input is a data frame, it checks whether file paths are absolute; -#' relative paths are only allowed when metadata is loaded from a file. -#' -#' It warns the user if many of the files listed in the metadata table are missing, -#' and stops execution if none of the files exist. -#' -#' The column with file paths is normalized and renamed to match the internal filename schema. -#' -#' @param metadata A metadata table. Can be either: -#' - a data frame with metadata, -#' - or a path to a text/TSV/CSV file that can be read with `readr::read_delim`. -#' -#' @param filename_col A string specifying the name of the column in the metadata table -#' that contains paths to repertoire files. Defaults to `"File"`. -#' -#' @param delim Delimiter used to read the metadata file (if a path is provided). Defaults to `"\t"`. -#' -#' @param ... Additional arguments passed to `readr::read_delim()` when reading metadata from a file. -#' -#' @return A validated and updated metadata data frame with absolute file paths, -#' and an additional column renamed according to `IMD_GLOBALS$schema$filename`. -#' -#' @concept ingestion -#' @export -read_metadata <- function(metadata, filename_col = "File", delim = "\t", ...) { - - if (!checkmate::test_data_frame(metadata) && !checkmate::test_file_exists(metadata)) { - cli_abort("Error in metadata: the input metadata should be either a data frame or an existing file.") - } - - # Get the metadata table - metadata_source <- NA - if (checkmate::test_file_exists(metadata)) { - metadata_table <- read_delim(metadata, delim = delim, ...) - metadata_source <- "file" # TODO: enum - } else { - metadata_table <- metadata - metadata_source <- "df" - } - - # Check for the column - if (!(filename_col %in% colnames(metadata_table))) { - cli_abort("Error: no column [{filename_col}] with full file paths and names in the input metadata table.") - } - - # Preprocess the files: - # - either they are in the same folder as metadata - # - or those are full paths to the files - metadata_table[[filename_col]] <- sapply(metadata_table[[filename_col]], function (path) { - if (!grepl("^(?:[A-Za-z]:/|/)", path)) { - # The path is not a full path, so we should add the directory of metadata to it - if it's a file - if (metadata_source == "file") { - path <- file.path(dirname(metadata), path) - } else { - cli_abort("Error: the input metadata is a data frame, but paths are relative. - Provide full paths in the [{filename_col}] column, e.g., - [/Users/username/projects/data/sample1.tsv] instead of [sample1.tsv] or [sample1]") - } - } - - normalizePath(path) - }) - names(metadata_table[[filename_col]]) <- NULL - - # Check how many files from the file column exist: - file_list <- metadata_table[[filename_col]] - file_existed <- sapply(file_list, test_file_exists) - n_existed <- sum(file_existed) - n_threshold <- round(length(file_list) * 0.1) + 1 - - cli_alert_info("Found {n_existed}/{length(file_list)} repertoire files from the metadata on the disk") - if (n_existed == 0) { - cli_abort("Error: found zero (!) repertoire files, passed in the metadata. Are the file paths in the metadata correct?") - } else if (n_existed <= n_threshold) { - cli_alert_warning("Warning: found only {n_existed} files out of {length(file_list)} in the metadata. Please check if you planned to work with more repertoire files. Continuing the execution.") - } - - cli_alert_success("Metadata parsed successfully") - - # Renaming columns could lead to downstream schema issues - immundata_filename_col <- IMD_GLOBALS$schema$filename - metadata_table[[immundata_filename_col]] <- metadata_table[[filename_col]] - - metadata_table -} diff --git a/R/io_repertoires_processing.R b/R/io_repertoires_processing.R index f06747c..4a41f47 100644 --- a/R/io_repertoires_processing.R +++ b/R/io_repertoires_processing.R @@ -1,88 +1,132 @@ -#' @title Preprocessing and postprocessing of input immune repertoire files -#' -#' @details -#' This collection of "maker" functions generates common preprocessing and -#' postprocessing function steps tailored for immune repertoire data. -#' Each `make_*` function returns a new function that can then be applied -#' to a dataset. -#' -#' These functions are designed to be flexible components in constructing -#' custom data processing workflows. -#' -#' @details -#' The functions generated by these factories typically expect a `dataset` -#' (e.g., a `duckplyr` with annotations) as their first argument -#' and may accept additional arguments via `...` (though often unused in the -#' predefined steps). -#' -#' - `make_default_preprocessing()` and `make_default_postprocessing()` assemble -#' a list of such processing functions. -#' - The individual `make_exclude_columns()`, `make_productive_filter()`, and -#' `make_barcode_prefix()` functions create specific transformation steps. -#' -#' These steps are often used when reading data to standardize formats, filter -#' unwanted records, or enrich information like cell barcodes. They are designed -#' to gracefully handle cases where an operation is not applicable (e.g., a specified -#' column is not found) by issuing a warning and returning the dataset unmodified. -#' -#' @section Functions: -#' * `make_default_preprocessing()`: Creates a default list of preprocessing -#' functions suitable for "airr" or "10x" formatted data. This typically -#' includes steps to exclude unnecessary columns and filter for productive sequences. -#' * `make_default_postprocessing()`: Creates a default list of postprocessing -#' functions, such as adding a prefix to cell barcodes. -#' * `make_exclude_columns()`: Creates a function that, when applied to a -#' dataset, removes a specified set of columns. -#' * `make_productive_filter()`: Creates a function that filters a dataset -#' to retain only rows where sequences are marked as productive, based on -#' a specified column and set of "truthy" values. -#' * `make_barcode_prefix()`: Creates a function that prepends a prefix -#' (sourced from a specified column in the dataset) to the cell barcodes. -#' -#' @param format For `make_default_preprocessing()`, a character string specifying -#' the input data format. Currently supports `"airr"` (default) or `"10x"`. -#' This determines the default set of columns to exclude and the values -#' considered "productive". -#' @param cols For `make_exclude_columns()`, a character vector of column names -#' to be removed from the dataset. Defaults to `imd_drop_cols("airr")`. -#' If empty, the returned function will not remove any columns. -#' @param col_name For `make_productive_filter()`, a character vector of potential -#' column names that indicate sequence productivity (e.g., `"productive"`). -#' The first matching column found in the dataset will be used. -#' @param truthy For `make_productive_filter()`, a value or vector of values -#' that signify a productive sequence in the `col_name` column. -#' Can be a logical `TRUE` (default for "airr" format) or a character vector -#' of strings (e.g., `c("true", "TRUE", "True", "t", "T", "1")` for "10x" format). -#' @param prefix_col For `make_barcode_prefix()`, the name of the column in the -#' dataset that contains the prefix string to be added to each cell barcode. -#' Defaults to `"Prefix"`. The barcode column itself is identified internally -#' via `imd_schema("barcode")`. -#' -#' @return -#' Each `make_*` function returns a *new function*. This returned function takes -#' a `dataset` as its first argument and `...` for any additional arguments, -#' and performs the specific processing step. -#' `make_default_preprocessing()` and `make_default_postprocessing()` return a -#' *named list* of such functions. -#' -#' @seealso -#' [read_repertoires()] +#' @title Process chain rows while reading repertoire files +#' +#' @description +#' Use these functions to preprocess or postprocess rows of the input data before +#' returning the final `ImmunData` object to the session. A couple of example +#' use cases: keep productive receptor chains, remove technical +#' columns, or make cell barcodes unique while importing repertoire files with +#' [read_repertoires()]. +#' +#' The defaults provide steps for common AIRR or 10x inputs. Use an individual step +#' when your files need only one operation or when you are building a custom +#' `preprocess` or `postprocess` list. +#' +#' Preprocessing changes chain rows before receptors are defined. Barcode +#' prefixing changes the cell identifier after receptor and manifest information +#' are combined. The input files and input table are not changed: every step +#' returns a new duckplyr table. +#' +#' @section Choose processing steps: +#' +#' * `make_default_preprocessing()` returns two steps. The first removes common +#' technical columns. The second keeps rows whose `productive` value indicates +#' a productive chain. If the `productive` column is absent, the filtering +#' step gives a warning and keeps all rows. +#' * `make_default_postprocessing()` returns one step that adds a sample-specific +#' prefix to cell barcodes. If the prefix column is absent, the step gives a +#' warning and leaves barcodes unchanged. +#' * `make_exclude_columns()` creates one step that removes the columns in +#' `cols`. Column names that are not present are ignored. +#' * `make_productive_filter()` creates one step that keeps rows whose value in +#' `col_name` matches any value in `truthy`. +#' * `make_barcode_prefix()` creates one step that joins a prefix, such as +#' `"Tumor_"`, to the start of each `imd_barcode` value. +#' +#' `read_repertoires()` applies functions in list order. You can therefore add, +#' remove, or reorder steps in a custom list. +#' +#' @section Input formats: +#' +#' For `make_default_preprocessing()`, `format = "default"` removes the union of +#' the standard AIRR and 10x technical columns. Use `format = "airr"` or +#' `format = "10x"` to remove only the columns expected for that format. All +#' three defaults recognize common text representations of a productive value, +#' including `"TRUE"`, `"true"`, `"yes"`, and `"1"`. +#' +#' +#' @param format A character string. One input format: `"default"`, `"airr"`, +#' or `"10x"`. The default is `"default"`. This choice controls which +#' technical columns are removed. It does not rename columns. +#' @param cols A character vector. Columns to remove. The default is +#' `imd_drop_cols("airr")`. Use `character()` to create a step that removes +#' no columns. +#' @param col_name A character string. Column containing the productive-chain +#' indicator. The default is `"productive"`. +#' @param truthy A vector. Values that mean the chain is productive. Values are +#' compared as text. The default is `TRUE`; use a character vector when the +#' source uses several representations, for example +#' `c("TRUE", "true", "1")`. +#' @param prefix_col A character vector. One or more candidate columns +#' containing the text to place before each cell barcode. The first candidate +#' present in the data is used. The default is `"Prefix"`. +#' +#' @return `make_default_preprocessing()` and +#' `make_default_postprocessing()` return named lists of processing functions. +#' The other functions return one processing function. Each processing +#' function accepts a duckplyr table as its first argument, accepts unused +#' arguments through `...`, and returns a new duckplyr table. +#' +#' @seealso [read_repertoires()], [imd_drop_cols()], [imd_rename_cols()] +#' +#' @examples +#' library(immundata) +#' library(dplyr) +#' +#' # Three 10x chain rows from two samples. One chain is non-productive. +#' chains <- duckplyr::duckdb_tibble( +#' imd_barcode = c("AAAC-1", "AAAG-1", "AATT-1"), +#' cdr3_aa = c("CASSA", "CASSB", "CASSC"), +#' productive = c("TRUE", "FALSE", "TRUE"), +#' full_length = c(TRUE, TRUE, TRUE), +#' Prefix = c("Tumor_", "Tumor_", "Blood_") +#' ) +#' +#' # read_repertoires() performs these calls for you. They are shown here to +#' # make the effect of each list clear. +#' prepared <- Reduce( +#' function(data, step) step(data), +#' make_default_preprocessing("10x"), +#' init = chains +#' ) +#' prepared <- Reduce( +#' function(data, step) step(data), +#' make_default_postprocessing(), +#' init = prepared +#' ) +#' +#' prepared |> +#' collect() |> +#' select(imd_barcode, cdr3_aa, productive) +#' # Expected result: +#' # imd_barcode cdr3_aa productive +#' # Tumor_AAAC-1 CASSA TRUE +#' # Blood_AATT-1 CASSC TRUE +#' +#' # The non-productive chain was removed, `full_length` was dropped, and the +#' # sample prefixes made the retained cell barcodes unique. #' #' @concept processing #' @rdname preprocess_postprocess #' @export -make_default_preprocessing <- function(format = c("airr", "10x")) { +make_default_preprocessing <- function(format = c("default", "airr", "10x")) { format <- match.arg(format) - if (format == "airr") { + truthy <- c("TRUE", "True", "true", "T", "t", "YES", "Yes", "yes", "Y", "y", "1") + + if (format == "default") { + list( + exclude_columns = make_exclude_columns(imd_drop_cols("universal")), + filter_nonproductive = make_productive_filter(truthy = truthy) + ) + } else if (format == "airr") { list( exclude_columns = make_exclude_columns(imd_drop_cols("airr")), - filter_nonproductive = make_productive_filter(truthy = TRUE) + filter_nonproductive = make_productive_filter(truthy = truthy) ) } else if (format == "10x") { list( exclude_columns = make_exclude_columns(imd_drop_cols("10x")), - filter_nonproductive = make_productive_filter(truthy = c("true", "TRUE", "True", "t", "T", "1")) + filter_nonproductive = make_productive_filter(truthy = truthy) ) } } @@ -117,7 +161,7 @@ make_exclude_columns <- function(cols = imd_drop_cols("airr")) { #' @export make_productive_filter <- function(col_name = c("productive"), truthy = TRUE) { - checkmate::assert_character(col_name) + checkmate::assert_string(col_name) fun <- function(dataset, ...) { col_name <- intersect( @@ -126,16 +170,21 @@ make_productive_filter <- function(col_name = c("productive"), ) if (length(col_name) == 0) { - cli::cli_alert_warning("No columns with productive specification found; skipping the filtering") + cli::cli_alert_warning("No columns with the productive specification found; skipping the filtering") dataset } else { - col <- col_name[[1]] + prod_col <- paste0("imd_", col_name) + truthy <- truthy |> as.character() - if (checkmate::test_logical(truthy)) { - dataset |> filter(!!rlang::sym(col_name) == truthy) + dataset <- dataset |> mutate(!!rlang::sym(prod_col) := dd$concat(!!rlang::sym(col_name), "")) + + if (length(truthy) == 1) { + dataset <- dataset |> filter(!!rlang::sym(prod_col) == truthy) } else { - dataset |> filter(!!rlang::sym(col_name) %in% truthy) + dataset <- dataset |> filter(!!rlang::sym(prod_col) %in% truthy) } + + dataset |> select(-!!rlang::sym(prod_col)) } } diff --git a/R/io_repertoires_read.R b/R/io_repertoires_read.R index 3521dec..8e2264b 100644 --- a/R/io_repertoires_read.R +++ b/R/io_repertoires_read.R @@ -1,180 +1,276 @@ -#' @title Read and process immune repertoire files to immundata +#' @title Read immune repertoire files into ImmunData #' #' @description -#' This is the main function for reading immune repertoire data into the -#' `immundata` framework. It reads one or more repertoire files (AIRR TSV, -#' 10X CSV, Parquet), performs optional preprocessing and column renaming, -#' aggregates sequences into receptors based on a provided schema, optionally -#' joins external metadata, performs optional postprocessing, and returns -#' an `ImmunData` object. -#' -#' The function handles different data types (bulk, single-cell) based on -#' the presence of `barcode_col` and `count_col`. For efficiency with large -#' datasets, it processes the data and saves intermediate results (annotations) -#' as a Parquet file before loading them back into the final `ImmunData` object. -#' -#' @param path Character vector. Path(s) to input repertoire files (e.g., -#' `"/path/to/data/*.tsv.gz"`). Supports glob patterns via [Sys.glob()]. -#' Files can be Parquet, CSV, TSV, or gzipped versions thereof. All files -#' must be of the same type. -#' Alternatively, pass the special string `""` to read file paths -#' from the `metadata` table (see `metadata` and `metadata_file_col` params). -#' @param schema Defines how unique receptors are identified. Can be: -#' - A character vector of column names (e.g., `c("v_call", "j_call", "junction_aa")`). -#' - A schema object created by [make_receptor_schema()], allowing specification -#' of chains for pairing (e.g., `make_receptor_schema(features = c("v_call", "junction_aa"), chains = c("TRA", "TRB"))`). -#' @param metadata Optional. A data frame containing -#' metadata to be joined with the repertoire data, read by -#' [read_metadata()] function. If `path = ""`, this table *must* -#' be provided and contain the file paths column specified by `metadata_file_col`. -#' Default: `NULL`. -#' @param barcode_col Character(1). Name of the column containing cell barcodes -#' or other unique cell/clone identifiers for single-cell data. Triggers -#' single-cell processing logic in [agg_receptors()]. Default: `NULL`. -#' @param count_col Character(1). Name of the column containing UMI counts or -#' frequency counts for bulk sequencing data. Triggers bulk processing logic -#' in [agg_receptors()]. Default: `NULL`. Cannot be specified if `barcode_col` is also -#' specified. -#' @param locus_col Character(1). Name of the column specifying the receptor chain -#' locus (e.g., "TRA", "TRB", "IGH", "IGK", "IGL"). Required if `schema` -#' specifies chains for pairing. Default: `NULL`. -#' @param umi_col Character(1). Name of the column containing UMI counts for -#' single-cell data. Used during paired-chain processing to select the most -#' abundant chain per barcode per locus. Default: `NULL`. -#' @param preprocess List. A named list of functions to apply sequentially to the -#' raw data *before* receptor aggregation. Each function should accept a -#' data frame (or duckplyr_df) as its first argument. See -#' [make_default_preprocessing()] for examples. -#' Default: `make_default_preprocessing()`. Set to `NULL` or `list()` to disable. -#' @param postprocess List. A named list of functions to apply sequentially to the -#' annotation data *after* receptor aggregation and metadata joining. Each -#' function should accept a data frame (or duckplyr_df) as its first argument. -#' See [make_default_postprocessing()] for examples. -#' Default: `make_default_postprocessing()`. Set to `NULL` or `list()` to disable. -#' @param rename_columns Named character vector. Optional mapping to rename columns -#' in the input files using `dplyr::rename()` syntax (e.g., -#' `c(new_name = "old_name", barcode = "cell_id")`). Renaming happens *before* -#' preprocessing and schema application. See [imd_rename_cols()] for presets. -#' Default: `imd_rename_cols("10x")`. -#' @param enforce_schema Logical(1). If `TRUE` (default), reading multiple files -#' requires them to have the exact same columns and types. If `FALSE`, columns -#' are unioned across files (potentially slower, requires more memory). -#' Default: `TRUE`. -#' @param metadata_file_col Character(1). The name of the column in the `metadata` -#' table that contains the full paths to the repertoire files. Only used when -#' `path = ""`. Default: `"File"`. -#' @param output_folder Character(1). Path to a directory where intermediate -#' processed annotation data will be saved as `annotations.parquet` and -#' `metadata.json`. If `NULL` (default), a folder named -#' `immundata-` is created in the same directory as the -#' first input file specified in `path`. The final `ImmunData` object reads -#' from these saved files. Default: `NULL`. -#' @param repertoire_schema Character vector or Function. Defines columns used to -#' group annotations into distinct repertoires (e.g., by sample or donor). -#' If provided, [agg_repertoires()] is called after loading to add repertoire-level -#' summaries and metrics. Default: `NULL`. +#' `read_repertoires()` is the main function for importing AIRR-seq data. It +#' reads one or more repertoire files, defines biological receptors, adds +#' sample information from an optional manifest, and returns an [ImmunData] +#' object. +#' +#' The function saves the processed data in `output_folder`. This lets you work +#' with large datasets without loading everything into memory and reopen the +#' result later with [read_immundata()]. +#' +#' @param path One or more repertoire file paths, or a glob pattern such as +#' `"/path/to/data/*.tsv.gz"`. Supported formats are Parquet, CSV, TSV, and +#' gzipped CSV or TSV. All input files must have the same file type. +#' +#' Use `""` to take file paths from `manifest` instead. In that +#' case, `manifest` is required. +#' @param schema Definition of receptor identity. Supply either: +#' +#' * A character vector naming the features that must match, such as +#' `c("v_call", "j_call", "junction_aa")`. +#' * An object created by [make_receptor_schema()] to select one locus or pair +#' two loci from the same cell. +#' +#' Use column names as they appear *after* `rename_columns` is applied. For +#' example, if the input columns are `CDR3.aa` and `V.name`, use +#' `rename_columns = c(cdr3_aa = "CDR3.aa", v_call = "V.name")` together with +#' `schema = c("cdr3_aa", "v_call")`. +#' @param manifest An optional data frame with one row per repertoire file and +#' columns containing sample, donor, tissue, treatment, or other information. +#' Use [read_manifest()] to read and validate a manifest file. Manifest paths +#' must be unique. When `path = ""`, the column named by +#' `manifest_file_col` supplies the repertoire file paths. The default is +#' `NULL`. +#' @param barcode_col Name of the column containing cell barcodes. Supplying it +#' selects single-cell processing, requires `umi_col`, and prevents use of +#' `count_col`. Use the column name after renaming. The default is `NULL`. +#' @param count_col Name of the column containing non-negative abundance values +#' for bulk repertoire data. It cannot be used with `barcode_col`. Use the +#' column name after renaming. The default is `NULL`. +#' @param locus_col Name of the column containing receptor loci such as `"TRA"`, +#' `"TRB"`, `"IGH"`, `"IGK"`, or `"IGL"`. It is required when `schema` +#' selects or pairs chains. Use the column name after renaming. The default is +#' `NULL`. +#' @param umi_col Name of the column containing per-chain UMI or read counts. +#' It is required whenever `barcode_col` is supplied and is used to choose one +#' chain when a cell contains several chains from the same locus. Use the +#' column name after renaming. The default is `NULL`. +#' @param preprocess A named list of functions applied in order before receptors +#' are defined. Each function must accept a duckplyr table as its first +#' argument and return a duckplyr table. By default, +#' [make_default_preprocessing()] removes selected technical columns and keeps +#' productive sequences when a `productive` column is available. Use `NULL` +#' or `list()` to disable preprocessing. +#' @param postprocess A named list of functions applied in order after receptors +#' are defined and manifest information is added. Each function must accept +#' and return a duckplyr table. By default, [make_default_postprocessing()] +#' prefixes cell barcodes when the manifest contains a `Prefix` column. Use +#' `NULL` or `list()` to disable postprocessing. +#' @param rename_columns An optional named character vector in the form +#' `c(new_name = "old_name")`. Renaming occurs before preprocessing and +#' receptor definition. The default, `imd_rename_cols("10x")`, standardizes +#' common 10x names such as `v_gene` to `v_call` and `chain` to `locus` when +#' those source columns are present. Use `NULL` to preserve all input names. +#' @param enforce_schema Whether multiple input files must have the same columns +#' and column types. The default is `TRUE`. If `FALSE`, columns are combined +#' by name and missing values are added where necessary. This is slower and +#' can require more memory. +#' @param manifest_file_col Name of the manifest column containing repertoire +#' file paths when `path = ""`. The default is `"file"`. Use the +#' same name passed as `file_col` to [read_manifest()] when it is not `"file"`. +#' @param output_folder Directory in which to write `annotations.parquet` and +#' `metadata.json`. These files are the persistent backing storage for the +#' returned object. If `NULL`, a folder beginning with `immundata-` is created +#' beside the first input file. Supplying an existing folder replaces its +#' `annotations.parquet` and `metadata.json`. The default is `NULL`. +#' @param repertoire_schema Definition of repertoires. Supply one of: +#' +#' * A character vector naming columns that define one repertoire, such as +#' `c("donor", "timepoint")`. +#' * `""`, the default. This creates one repertoire per input file, or +#' one per manifest row when `path = ""`. +#' * `""`, which uses all manifest columns when a manifest is +#' available, or the input filename otherwise. +#' * `NULL` to leave repertoires undefined. +#' @param verbose Whether to print progress and summary messages. Defaults to +#' `getOption("immundata.verbose", TRUE)`. +#' @param prematerialize Whether CSV, TSV, and compressed text inputs should be +#' combined into a temporary Parquet file before receptor processing. This +#' avoids repeatedly scanning text input during downstream lazy queries. +#' Existing Parquet input is used directly. The default is `TRUE`. +#' @param prematerialize_folder Directory in which to create the temporary +#' combined Parquet file. If `NULL`, the default, [tempdir()] is used. The +#' directory is created when necessary. The temporary file is deleted when +#' `read_repertoires()` exits, including after an error. #' #' @details -#' The function executes the following steps: -#' 1. Validates inputs. -#' 2. Determines the list of input files based on `path` and `metadata`. Checks file extensions. -#' 3. Reads data using `duckplyr` (`read_parquet_duckdb` or `read_csv_duckdb`). Handles `.gz`. -#' 4. Applies column renaming if `rename_columns` is provided. -#' 5. Applies preprocessing steps sequentially if `preprocess` is provided. -#' 6. Aggregates sequences into receptors using [agg_receptors()], based on `schema`, `barcode_col`, `count_col`, `locus_col`, and `umi_col`. This creates the core annotation table. -#' 7. Joins the `metadata` table if provided. -#' 8. Applies postprocessing steps sequentially if `postprocess` is provided. -#' 9. Creates a temporary `ImmunData` object in memory. -#' 10. Determines the `output_folder` path. -#' 11. Saves the processed annotation table and metadata using [write_immundata()] to the `output_folder`. -#' 12. Loads the data back from the saved Parquet files using [read_immundata()] to create the final `ImmunData` object. This ensures the returned object is backed by efficient storage. -#' 13. If `repertoire_schema` is provided, calls [agg_repertoires()] on the loaded object to define and summarize repertoires. -#' 14. Returns the final `ImmunData` object. -#' -#' @return An `ImmunData` object containing the processed receptor annotations. -#' If `repertoire_schema` was provided, the object will also contain repertoire -#' definitions and summaries calculated by [agg_repertoires()]. -#' -#' @seealso [ImmunData], [read_immundata()], [write_immundata()], [read_metadata()], -#' [agg_receptors()], [agg_repertoires()], [make_receptor_schema()], -#' [make_default_preprocessing()], [make_default_postprocessing()] +#' The required arguments depend on how receptor observations are represented in +#' the input files. +#' +#' @section Choose arguments for your data: +#' +#' * **Uncounted repertoire table:** Supply `schema`. Leave `barcode_col` and +#' `count_col` as `NULL`. Each retained row represents one observed chain. +#' * **Bulk repertoire with abundance:** Supply `schema` and `count_col`. The +#' abundance values are preserved for later repertoire statistics. +#' * **Single-cell, one selected chain:** Use [make_receptor_schema()] with one +#' chain and supply `barcode_col`, `locus_col`, and `umi_col`. +#' * **Single-cell, paired chains:** Use [make_receptor_schema()] with two chains +#' and supply `barcode_col`, `locus_col`, and `umi_col`. Only cells containing +#' both requested chains are retained. +#' * **Single-cell, relaxed paired chains:** Use a schema such as +#' `chains = c("IGH", "IGL|IGK")` with `barcode_col`, `locus_col`, and +#' `umi_col`. This accepts either an IGH-IGL or IGH-IGK receptor. +#' +#' In single-cell data, the chain with the highest `umi_col` value is retained +#' when a cell contains several chains from the same locus. +#' +#' @section What happens by default: +#' +#' Unless you override the relevant arguments, `read_repertoires()`: +#' +#' * temporarily combines text input into Parquet before processing; +#' * standardizes common 10x column names; +#' * removes selected technical columns; +#' * keeps productive sequences when productivity information is present; +#' * prefixes barcodes when a manifest `Prefix` column is present; +#' * creates repertoires automatically; and +#' * writes the completed dataset to disk. +#' +#' Set `rename_columns`, `preprocess`, `postprocess`, or `repertoire_schema` to +#' `NULL` to disable the corresponding behavior. +#' +#' @section Processing order: +#' +#' The function: +#' +#' 1. finds and reads the input files as one duckplyr table; +#' 2. temporarily combines non-Parquet input into one Parquet file when +#' `prematerialize = TRUE`; +#' 3. renames columns; +#' 4. applies preprocessing; +#' 5. defines receptors using `schema`; +#' 6. adds manifest information; +#' 7. applies postprocessing; +#' 8. defines repertoires when requested; and +#' 9. writes and reopens the completed [ImmunData] dataset. +#' +#' @section Manifests and repertoires: +#' +#' A manifest *annotates* each input file with biological information. The +#' `repertoire_schema` argument chooses which annotation columns *define a +#' repertoire* and therefore determine receptor counts and proportions. +#' +#' With `path = ""` and the default `repertoire_schema = ""`, +#' all manifest columns are used and each manifest row becomes one repertoire. +#' With an explicit file path or vector of paths, `""` creates one +#' repertoire per input file. +#' +#' @section Output storage: +#' +#' The output folder is not a temporary cache. The returned object reads its +#' receptor annotations from `annotations.parquet`, while `metadata.json` stores +#' its schemas, repertoire summaries, and provenance. Keep this folder for as +#' long as you need the object, or reopen it later with [read_immundata()]. +#' +#' **Important:** Reusing the same `output_folder` replaces the existing +#' `annotations.parquet` and `metadata.json` without creating a new version. +#' +#' @return A disk-backed [ImmunData] object containing the retained chain rows, +#' receptor definitions, manifest annotations, and ingestion provenance. If +#' `repertoire_schema` is not `NULL`, it also contains repertoire definitions +#' and summary statistics calculated by [agg_repertoires()]. +#' +#' @seealso [read_manifest()], [make_receptor_schema()], [agg_receptors()], +#' [agg_repertoires()], [make_default_preprocessing()], +#' [make_default_postprocessing()], [read_immundata()], [write_immundata()], +#' [ImmunData] #' #' @concept ingestion #' @export #' #' @examples -#' \dontrun{ -#' # -#' # Example 1: single-chain, one file -#' # -#' # Read a single AIRR TSV file, defining receptors by V/J/CDR3_aa -#' # Assume "my_sample.tsv" exists and follows AIRR format -#' -#' # Create a dummy file for illustration -#' airr_data <- data.frame( -#' sequence_id = paste0("seq", 1:5), -#' v_call = c("TRBV1", "TRBV1", "TRBV2", "TRBV1", "TRBV3"), -#' j_call = c("TRBJ1", "TRBJ1", "TRBJ2", "TRBJ1", "TRBJ1"), -#' junction_aa = c("CASSL...", "CASSL...", "CASSD...", "CASSL...", "CASSF..."), -#' productive = c(TRUE, TRUE, TRUE, FALSE, TRUE), -#' locus = c("TRB", "TRB", "TRB", "TRB", "TRB") +#' library(immundata) +#' library(dplyr) +#' +#' options(immundata.verbose = FALSE) +#' +#' # Read one bulk AIRR file and preserve its abundance column +#' bulk_file <- system.file( +#' "extdata/tsv", +#' "sample_0_1k.tsv", +#' package = "immundata" #' ) -#' readr::write_tsv(airr_data, "my_sample.tsv") #' -#' # Define receptor schema -#' receptor_def <- c("v_call", "j_call", "junction_aa") +#' bulk_idata <- read_repertoires( +#' path = bulk_file, +#' schema = c("cdr3_aa", "v_call"), +#' count_col = "counts", +#' output_folder = tempfile("immundata-bulk-") +#' ) #' -#' # Specify output folder -#' out_dir <- tempfile("immundata_output_") +#' tibble( +#' n_records = bulk_idata |> count() |> pull(n), +#' n_receptors = bulk_idata$receptors |> count() |> collect() |> pull(n), +#' n_repertoires = nrow(bulk_idata$repertoires) +#' ) +#' # Expected result: +#' # n_records n_receptors n_repertoires +#' # 955 871 1 +#' +#' # Read multiple files and their sample information from a manifest +#' manifest_path <- system.file( +#' "extdata/tsv", +#' "manifest.csv", +#' package = "immundata" +#' ) +#' manifest <- read_manifest(manifest_path) #' -#' # Read the data (disabling default preprocessing for this simple example) -#' idata <- read_repertoires( -#' path = "my_sample.tsv", -#' schema = receptor_def, -#' output_folder = out_dir, -#' preprocess = NULL, # Disable default productive filter for demo -#' postprocess = NULL # Disable default barcode prefixing +#' manifest_idata <- read_repertoires( +#' path = "", +#' manifest = manifest, +#' schema = c("cdr3_aa", "v_call"), +#' count_col = "counts", +#' output_folder = tempfile("immundata-manifest-") #' ) #' -#' print(idata) -#' print(idata$annotations) -#' -#' # -#' # Example 2: single-chain, multiple files -#' # -#' # Read multiple files using metadata -#' # Create dummy files and metadata -#' readr::write_tsv(airr_data[1:2, ], "sample1.tsv") -#' readr::write_tsv(airr_data[3:5, ], "sample2.tsv") -#' meta <- data.frame( -#' SampleID = c("S1", "S2"), -#' Tissue = c("PBMC", "Tumor"), -#' FilePath = c(normalizePath("sample1.tsv"), normalizePath("sample2.tsv")) +#' manifest_idata$repertoires |> +#' select(Therapy, Response, n_barcodes, n_receptors) |> +#' arrange(Response) +#' # Expected result: +#' # Therapy Response n_barcodes n_receptors +#' # ICI FR 4725 871 +#' # CAR-T PR 4758 867 +#' +#' # Read paired TRA-TRB receptors from a small single-cell table +#' paired_input <- tibble( +#' cell_id = c("cell1", "cell1", "cell2", "cell2", "cell3"), +#' locus = c("TRA", "TRB", "TRA", "TRB", "TRA"), +#' v_call = c("TRAV1", "TRBV1", "TRAV1", "TRBV1", "TRAV2"), +#' j_call = c("TRAJ1", "TRBJ1", "TRAJ1", "TRBJ1", "TRAJ2"), +#' junction_aa = c("CAVA", "CASSB", "CAVA", "CASSB", "CAVC"), +#' umi_count = c(10L, 8L, 12L, 9L, 7L) #' ) -#' readr::write_tsv(meta, "metadata.tsv") -#' -#' idata_multi <- read_repertoires( -#' path = "", -#' metadata = meta, -#' metadata_file_col = "FilePath", -#' schema = receptor_def, -#' repertoire_schema = "SampleID", # Aggregate by SampleID -#' output_folder = tempfile("immundata_multi_"), -#' preprocess = make_default_preprocessing("airr"), # Use default AIRR filters -#' postprocess = NULL +#' paired_file <- tempfile(fileext = ".tsv") +#' readr::write_tsv(paired_input, paired_file) +#' +#' paired_idata <- read_repertoires( +#' path = paired_file, +#' schema = make_receptor_schema( +#' features = c("v_call", "j_call", "junction_aa"), +#' chains = c("TRA", "TRB") +#' ), +#' barcode_col = "cell_id", +#' locus_col = "locus", +#' umi_col = "umi_count", +#' repertoire_schema = NULL, +#' output_folder = tempfile("immundata-paired-") #' ) #' -#' print(idata_multi) -#' print(idata_multi$repertoires) # Check repertoire summary +#' tibble( +#' n_chains = paired_idata |> count() |> pull(n), +#' n_cells = paired_idata |> collect() |> distinct(imd_barcode) |> nrow(), +#' n_receptors = paired_idata$receptors |> count() |> collect() |> pull(n) +#' ) +#' # Expected result: +#' # n_chains n_cells n_receptors +#' # 4 2 1 #' -#' # Clean up dummy files -#' file.remove("my_sample.tsv", "sample1.tsv", "sample2.tsv", "metadata.tsv") -#' unlink(out_dir, recursive = TRUE) -#' unlink(attr(idata_multi, "output_folder"), recursive = TRUE) # Get path used by function -#' } read_repertoires <- function(path, schema, - metadata = NULL, + manifest = NULL, barcode_col = NULL, count_col = NULL, locus_col = NULL, @@ -183,10 +279,20 @@ read_repertoires <- function(path, postprocess = make_default_postprocessing(), rename_columns = imd_rename_cols("10x"), enforce_schema = TRUE, - metadata_file_col = "File", + manifest_file_col = "file", output_folder = NULL, - repertoire_schema = NULL) { + repertoire_schema = "", + verbose = getOption("immundata.verbose", TRUE), + prematerialize = TRUE, + prematerialize_folder = NULL) { start_time <- Sys.time() + prematerialized_path <- NULL + prematerialization_applied <- FALSE + on.exit({ + if (!is.null(prematerialized_path) && file.exists(prematerialized_path)) { + unlink(prematerialized_path) + } + }, add = TRUE) checkmate::assert_character(path) @@ -196,8 +302,8 @@ read_repertoires <- function(path, assert_receptor_schema(schema) - checkmate::assert_data_frame(metadata, null.ok = T) - checkmate::assert_character(metadata_file_col, null.ok = T) + checkmate::assert_data_frame(manifest, null.ok = TRUE) + checkmate::assert_character(manifest_file_col, null.ok = TRUE) checkmate::assert_character( barcode_col, min.len = 1, @@ -230,36 +336,81 @@ read_repertoires <- function(path, ) checkmate::assert_character(rename_columns, null.ok = TRUE) checkmate::assert_logical(enforce_schema) + checkmate::assert_flag(verbose) + checkmate::assert_flag(prematerialize) + checkmate::assert_string( + prematerialize_folder, + null.ok = TRUE + ) checkmate::assert_list(preprocess, null.ok = TRUE) if (!is.null(preprocess)) { sapply(preprocess, checkmate::assert_function) } + requested_rename_columns <- rename_columns + applied_rename_columns <- requested_rename_columns[0] + missing_rename_columns <- requested_rename_columns[0] + dropped_columns <- character() + repertoire_schema_was_special <- is_special_repertoire_schema(repertoire_schema) + # - # Preprocessing the metadata + # Preprocessing the manifest # - # TODO: define "" in globals.R - immundata_filename_col <- IMD_GLOBALS$schema$filename + # TODO: define "" in globals.R + immundata_filename_col <- IMD_GLOBALS$schema$manifest_filename + path_from_manifest <- identical(path[1], "") if (path[1] == "") { - if (!is.null(metadata)) { - path <- normalizePath(metadata[[metadata_file_col]]) - metadata[[immundata_filename_col]] <- path + cli::cli_abort("Input repertoire metadata tables are now manifests. Use {.code path = ''}, {.arg manifest}, and {.arg manifest_file_col}. Snapshot metadata.json is not affected.") + } + + if (path_from_manifest) { + if (!is.null(manifest)) { + if (!manifest_file_col %in% colnames(manifest)) { + cli::cli_abort("Passed {.code path = ''}, but the manifest has no column {.field {manifest_file_col}}. Available manifest columns: [{colnames(manifest)}].") + } + + if (any(is.na(manifest[[manifest_file_col]]) | manifest[[manifest_file_col]] == "")) { + cli::cli_abort("Column {.field {manifest_file_col}} in manifest contains empty/NA paths. Please provide valid file paths for all rows.") + } + + path <- normalizePath(manifest[[manifest_file_col]]) + manifest[[immundata_filename_col]] <- path } else { - cli::cli_abort("Passed ``, but no `metadata` table provided. Please provide either a list of file paths or a metadata table.") + cli::cli_abort("Passed ``, but no `manifest` table provided. Please provide either a list of file paths or a manifest.") } } else { path <- normalizePath(Sys.glob(path), mustWork = FALSE) + + if (!is.null(manifest) && immundata_filename_col %in% colnames(manifest)) { + manifest[[immundata_filename_col]] <- normalizePath( + manifest[[immundata_filename_col]], + mustWork = FALSE + ) + } + } + + if (!is.null(manifest) && immundata_filename_col %in% colnames(manifest)) { + assert_unique_manifest_paths(manifest[[immundata_filename_col]]) } checkmate::assert_file_exists(path) + resolved_repertoire_schema <- resolve_repertoire_schema( + repertoire_schema = repertoire_schema, + manifest = manifest, + path_from_manifest = path_from_manifest, + filename_col = immundata_filename_col + ) + # Read the dataset - cli::cli_h3("Reading repertoire data") - file_check_results <- check_file_extensions(path) + if (verbose) { + cli::cli_h3("Reading repertoire data") + } + file_check_results <- check_file_extensions(path, verbose = verbose) input_file_type <- file_check_results$filetype delim <- file_check_results$delim - raw_dataset <- switch(input_file_type, + raw_dataset <- suppressMessages(switch(input_file_type, parquet = read_parquet_duckdb(path, prudence = "stingy", options = list( @@ -282,17 +433,68 @@ read_repertoires <- function(path, union_by_name = !enforce_schema ) ) - ) + )) + + raw_dataset <- raw_dataset |> + rename(!!immundata_filename_col := any_of("filename")) + + if (isTRUE(prematerialize) && input_file_type != "parquet") { + if (is.null(prematerialize_folder)) { + prematerialize_folder <- tempdir() + } + + dir.create( + prematerialize_folder, + showWarnings = FALSE, + recursive = TRUE + ) + if (!dir.exists(prematerialize_folder)) { + cli::cli_abort( + "Cannot create {.arg prematerialize_folder} at [{prematerialize_folder}]." + ) + } + + prematerialized_path <- tempfile( + pattern = "immundata-prematerialized-", + tmpdir = prematerialize_folder, + fileext = ".parquet" + ) + + if (verbose) { + cli::cli_h3("Prematerializing repertoire data") + cli::cli_alert_info( + "Combining text input into temporary Parquet at [{prematerialized_path}]" + ) + compute_parquet(raw_dataset, prematerialized_path) + } else { + suppressMessages(compute_parquet(raw_dataset, prematerialized_path)) + } + + raw_dataset <- suppressMessages(read_parquet_duckdb( + prematerialized_path, + prudence = "stingy" + )) + prematerialization_applied <- TRUE + + if (verbose) { + cli::cli_alert_success("Prematerialization is finished") + } + } # Rename columns if (!is.null(rename_columns)) { - cli::cli_h3("Renaming the columns and schemas") + if (verbose) { + cli::cli_h3("Renaming the columns and schemas") + } old_colnames <- colnames(raw_dataset) + applied_rename_columns <- rename_columns[unname(rename_columns) %in% old_colnames] + missing_rename_columns <- rename_columns[!unname(rename_columns) %in% old_colnames] + raw_dataset <- raw_dataset |> rename(any_of(rename_columns)) new_colnames <- colnames(raw_dataset) renamed_cols <- setdiff(new_colnames, old_colnames) - if (length(renamed_cols)) { + if (length(renamed_cols) && verbose) { cli_alert_success("Introduced new renamed columns: {renamed_cols}") } @@ -302,86 +504,136 @@ read_repertoires <- function(path, } } - if (!is.null(repertoire_schema)) { - for (i in seq_along(repertoire_schema)) { - if (repertoire_schema[i] %in% rename_columns) { - repertoire_schema[i] <- names(rename_columns)[repertoire_schema[i] == rename_columns] + if (!is.null(resolved_repertoire_schema) && !repertoire_schema_was_special && is.character(resolved_repertoire_schema)) { + for (i in seq_along(resolved_repertoire_schema)) { + if (resolved_repertoire_schema[i] %in% rename_columns) { + resolved_repertoire_schema[i] <- names(rename_columns)[resolved_repertoire_schema[i] == rename_columns] } } } - cli::cli_alert_success("Renaming is finished") + if (verbose) { + cli::cli_alert_success("Renaming is finished") + } } # # Preprocess the data # if (length(preprocess)) { - cli::cli_h3("Preprocessing the data") + if (verbose) { + cli::cli_h3("Preprocessing the data") + } + preprocess_input_cols <- colnames(raw_dataset) - ol <- cli::cli_ol() - cli::cli_ol() + if (verbose) { + ol <- cli::cli_ol() + cli::cli_ol() + } for (strategy_i in seq_along(preprocess)) { - cli::cli_li(names(preprocess)[strategy_i]) - raw_dataset <- preprocess[[strategy_i]](raw_dataset, metadata = metadata) + if (verbose) { + cli::cli_li(names(preprocess)[strategy_i]) + raw_dataset <- preprocess[[strategy_i]](raw_dataset, manifest = manifest) + } else { + raw_dataset <- suppressMessages(preprocess[[strategy_i]](raw_dataset, manifest = manifest)) + } } - cli::cli_end() - cli::cli_end(ol) + if (verbose) { + cli::cli_end() + cli::cli_end(ol) + } + + dropped_columns <- setdiff(preprocess_input_cols, colnames(raw_dataset)) - cli::cli_alert_success("Preprocessing plan is ready") + if (verbose) { + cli::cli_alert_success("Preprocessing plan is ready") + } } # # Aggregate the data # - cli::cli_h3("Aggregating the data to receptors") + if (verbose) { + cli::cli_h3("Aggregating the data to receptors") + } - annotation_data <- agg_receptors( - dataset = raw_dataset, - schema = schema, - barcode_col = barcode_col, - count_col = count_col, - locus_col = locus_col, - umi_col = umi_col - ) + if (verbose) { + annotation_data <- agg_receptors( + dataset = raw_dataset, + schema = schema, + barcode_col = barcode_col, + count_col = count_col, + locus_col = locus_col, + umi_col = umi_col, + verbose = verbose + ) + } else { + annotation_data <- suppressMessages(agg_receptors( + dataset = raw_dataset, + schema = schema, + barcode_col = barcode_col, + count_col = count_col, + locus_col = locus_col, + umi_col = umi_col, + verbose = verbose + )) + } - cli::cli_alert_success("Execution plan for receptor data aggregation and annotation is ready") + if (verbose) { + cli::cli_alert_success("Execution plan for receptor data aggregation and annotation is ready") + } # - # Joining with the metadata table + # Joining with the manifest table # - if (!is.null(metadata)) { - if (!immundata_filename_col %in% colnames(metadata)) { - cli::cli_abort("No '{immundata_filename_col}' in the metadata table. It is imperative to have this column - `immundata` uses it to annotate the AIRR files") + if (!is.null(manifest)) { + if (!immundata_filename_col %in% colnames(manifest)) { + cli::cli_abort("No '{immundata_filename_col}' in the manifest. It is imperative to have this column - `immundata` uses it to annotate the AIRR files") } - cli::cli_h3("Joining the metadata table with the dataset using '{immundata_filename_col}' column") + if (verbose) { + cli::cli_h3("Joining the manifest with the dataset using '{immundata_filename_col}' column") + } - metadata_duckdb <- duckdb_tibble(metadata) + manifest_duckdb <- duckdb_tibble(manifest) annotation_data <- annotation_data |> - left_join(metadata_duckdb, by = immundata_filename_col) + left_join(manifest_duckdb, by = immundata_filename_col) - cli::cli_alert_success("Joining plan is ready") + if (verbose) { + cli::cli_alert_success("Joining plan is ready") + } } # # Postprocess the data # if (length(postprocess)) { - cli::cli_h3("Postprocessing the data") + if (verbose) { + cli::cli_h3("Postprocessing the data") + } - ol <- cli::cli_ol() - cli::cli_ol() + if (verbose) { + ol <- cli::cli_ol() + cli::cli_ol() + } for (strategy_i in seq_along(postprocess)) { - cli::cli_li(names(postprocess)[strategy_i]) - annotation_data <- postprocess[[strategy_i]](annotation_data) + if (verbose) { + cli::cli_li(names(postprocess)[strategy_i]) + annotation_data <- postprocess[[strategy_i]](annotation_data) + } else { + annotation_data <- suppressMessages(postprocess[[strategy_i]](annotation_data)) + } + } + if (verbose) { + cli::cli_end() + cli::cli_end(ol) } - cli::cli_end() - cli::cli_end(ol) - cli::cli_alert_success("Postprocessing plan is ready") + if (verbose) { + cli::cli_alert_success("Postprocessing plan is ready") + } } idata <- ImmunData$new( @@ -396,48 +648,218 @@ read_repertoires <- function(path, } dir.create(output_folder, showWarnings = FALSE, recursive = TRUE) + # + # Create repertoires + # + if (!is.null(resolved_repertoire_schema)) { + if (verbose) { + cli::cli_h3("Aggregating repertoires...") + } + if (verbose) { + idata <- agg_repertoires(idata, resolved_repertoire_schema, verbose = verbose) + } else { + idata <- suppressMessages( + agg_repertoires(idata, resolved_repertoire_schema, verbose = verbose) + ) + } + if (verbose) { + cli_alert_success("Aggregation is finished") + } + } + # # Save the created ImmunData on disk # - cli::cli_h3("Saving the newly created ImmunData to disk") + if (verbose) { + cli::cli_h3("Saving the newly created ImmunData to disk") + } - write_immundata(idata, output_folder) + write_immundata_internal( + idata = idata, + output_folder = output_folder, + producer_function = "read_repertoires", + ingestion_payload = list( + inputs = list( + files = path, + manifest_joined = !is.null(manifest), + enforce_schema = enforce_schema + ), + args = list( + barcode_col = barcode_col, + count_col = count_col, + locus_col = locus_col, + umi_col = umi_col, + manifest_file_col = manifest_file_col + ), + column_lineage = list( + renamed = list( + requested = requested_rename_columns, + applied = applied_rename_columns, + not_found = missing_rename_columns + ), + dropped = list( + applied = dropped_columns + ) + ), + pipeline = list( + prematerialize = list( + requested = prematerialize, + applied = prematerialization_applied + ), + preprocess = names(preprocess), + postprocess = names(postprocess) + ) + ), + verbose = verbose + ) # # ... and load it again so the source will be fast Parquet files # idata <- read_immundata(output_folder, verbose = FALSE) - # - # Create repertoires - # - if (!is.null(repertoire_schema)) { - cli::cli_h3("Aggregating repertoires...") - idata <- agg_repertoires(idata, repertoire_schema) - cli_alert_success("Aggregation is finished") + if (verbose) { + cli::cli_h3("Summary") } - # TODO: we need to create repertoires -> - # without repertoire aggregating (!) write it on disk with (!!) the repertoire schema - - cli::cli_h3("Summary") final_time <- format(round(Sys.time() - start_time, 2)) - cli_alert_info("Time elapsed: {.emph {final_time}}") + if (verbose) { + cli_alert_info("Time elapsed: {.emph {final_time}}") + } idata_size <- idata |> count() |> pull("n") - cli_alert_success("Loaded ImmunData with the receptor schema: [{schema}]") + idata_receptors <- idata$annotations |> + distinct(!!imd_schema_sym("receptor")) |> + count() |> + pull("n") - if (!is.null(repertoire_schema)) { - cli_alert_success("Loaded ImmunData with the repertoire schema: [{repertoire_schema}]") + if (verbose) { + cli_alert_success("Loaded ImmunData with the receptor schema: [{schema}]") } - if (idata_size == 0) { - cli_alert_warning("Loaded ImmunData with zero (!) chains. Possible problems: wrong {.code 'chain'} specification to the receptor schema (e.g., {.code 'TCRB'} instead of {.code 'TRB'}), or preproces/postprocess filters") - } else { - cli_alert_success("Loaded ImmunData with [{idata_size}] chains") + if (!is.null(resolved_repertoire_schema) && verbose) { + cli_alert_success("Loaded ImmunData with the repertoire schema: [{resolved_repertoire_schema}]") + } + + if (verbose) { + if (idata_size == 0) { + cli_alert_warning("Loaded ImmunData with zero (!) chains. Possible problems: wrong {.code 'chain'} specification to the receptor schema (e.g., {.code 'TCRB'} instead of {.code 'TRB'}), or preproces/postprocess filters") + } else { + cli_alert_success("Loaded ImmunData with [{idata_size}] chains and [{idata_receptors}] receptors") + } } idata } + +is_special_repertoire_schema <- function(repertoire_schema, value = NULL) { + is_special <- is.character(repertoire_schema) && + length(repertoire_schema) == 1 && + repertoire_schema %in% c("", "") + + if (is.null(value)) { + return(is_special) + } + + is_special && identical(repertoire_schema, value) +} + +assert_unique_manifest_paths <- function(paths) { + duplicated_paths <- unique(paths[duplicated(paths)]) + + if (length(duplicated_paths) == 0) { + return(invisible(TRUE)) + } + + duplicated_path_details <- vapply( + duplicated_paths, + function(duplicated_path) { + duplicated_rows <- which(paths == duplicated_path) + paste0( + duplicated_path, + " (rows ", + paste(duplicated_rows, collapse = ", "), + ")" + ) + }, + character(1) + ) + + cli::cli_abort(c( + "Manifest contains duplicated repertoire file paths after normalization.", + "!" = "Each repertoire file must appear only once.", + "x" = "Duplicated paths: {paste(duplicated_path_details, collapse = '; ')}" + )) +} + +resolve_repertoire_schema <- function(repertoire_schema, + manifest, + path_from_manifest, + filename_col) { + if (is.null(repertoire_schema) || is.function(repertoire_schema)) { + return(repertoire_schema) + } + + if (is_special_repertoire_schema(repertoire_schema, "")) { + if (isTRUE(path_from_manifest)) { + repertoire_schema <- "" + } else { + return(filename_col) + } + } + + if (is_special_repertoire_schema(repertoire_schema, "")) { + if (!is.null(manifest)) { + return(colnames(manifest)) + } + + return(filename_col) + } + + repertoire_schema +} + +check_file_extensions <- function(path, verbose = TRUE) { + if (verbose) { + ol <- cli_ol() + cli_ol(path) + cli_end(ol) + + cli_alert_info("Checking if all files are of the same type") + } + + input_file_type <- NA + delim <- NA + + unique_extensions <- file_ext(path) |> + unique() |> + tolower() + + if (length(unique_extensions) == 1) { + if (unique_extensions %in% c("gz", "gzip")) { + unique_extensions <- strsplit(path[1], ".", fixed = TRUE)[[1]] + unique_extensions <- paste(tail(unique_extensions, 2), collapse = ".") + } + + # TODO: I have no idea how to make it more elegant. + # TODO: make enum-like list for file types + if (unique_extensions %in% c("parquet", "csv", "tsv", "csv.gz", "tsv.gz", "csv.gzip", "tsv.gzip")) { + input_file_type <- strsplit(unique_extensions, ".", fixed = TRUE)[[1]][1] + + if (input_file_type == "tsv") { + delim <- "\t" + } + } else { + cli_abort("Unknown file type: [{unique_extensions}]. Supported file types: Parquet, CSV, TSV, gzipped CSV and TSV") + } + if (verbose) { + cli_alert_success("All files have the same extension") + } + } else { + cli_abort("Not all files of the same type. Please convert them all to the same type, and try again") + } + + list(filetype = input_file_type, delim = delim) +} diff --git a/R/io_repertoires_utils.R b/R/io_repertoires_utils.R deleted file mode 100644 index b346824..0000000 --- a/R/io_repertoires_utils.R +++ /dev/null @@ -1,40 +0,0 @@ -check_file_extensions <- function(path, verbose = TRUE) { - - ol <- cli_ol() - cli_ol(path) - cli_end(ol) - - cli_alert_info("Checking if all files are of the same type") - - input_file_type <- NA - delim <- NA - - unique_extensions <- file_ext(path) |> - unique() |> - tolower() - - if (length(unique_extensions) == 1) { - - if (unique_extensions %in% c("gz", "gzip")) { - unique_extensions <- strsplit(path[1], ".", fixed = TRUE)[[1]] - unique_extensions <- paste(tail(unique_extensions, 2), collapse = ".") - } - - # TODO: I have no idea how to make it more elegant. - # TODO: make enum-like list for file types - if (unique_extensions %in% c("parquet", "csv", "tsv", "csv.gz", "tsv.gz", "csv.gzip", "tsv.gzip")) { - input_file_type <- strsplit(unique_extensions, ".", fixed = TRUE)[[1]][1] - - if (input_file_type == "tsv") { - delim <- "\t" - } - } else { - cli_abort("Unknown file type: [{unique_extensions}]. Supported file types: Parquet, CSV, TSV, gzipped CSV and TSV") - } - cli_alert_success("All files have the same extension") - } else { - cli_abort("Not all files of the same type. Please convert them all to the same type, and try again") - } - - list(filetype = input_file_type, delim = delim) -} diff --git a/R/operations_agg.R b/R/operations_agg.R deleted file mode 100644 index a83642f..0000000 --- a/R/operations_agg.R +++ /dev/null @@ -1,517 +0,0 @@ -#' @title Aggregate AIRR data into repertoires -#' -#' @description -#' Groups the annotation table of an `ImmunData` object by user-specified -#' columns to define distinct *repertoires* (e.g., based on sample, donor, -#' time point). It then calculates summary statistics both per-repertoire and -#' per-receptor within each repertoire. -#' -#' Calculated **per repertoire**: -#' * `n_barcodes`: Total number of unique cells/barcodes within the repertoire -#' (sum of `imd_chain_count`, effectively summing unique cells if input was SC, -#' or total counts if input was bulk). -#' * `n_receptors`: Number of unique receptors (`imd_receptor_id`) found within -#' the repertoire. -#' -#' Calculated **per annotation row** (receptor within repertoire context): -#' * `imd_count`: Total count of a specific receptor (`imd_receptor_id`) within -#' the specific repertoire it belongs to in that row (sum of relevant -#' `imd_chain_count`). -#' * `imd_proportion`: The proportion of the repertoire's total `n_barcodes` -#' accounted for by that specific receptor (`imd_count / n_barcodes`). -#' * `n_repertoires`: The total number of distinct repertoires (across the entire -#' dataset) in which this specific receptor (`imd_receptor_id`) appears. -#' -#' These statistics are added to the annotation table, and a summary table is -#' stored in the `$repertoires` slot of the returned object. -#' -#' @param idata An `ImmunData` object, typically the output of [read_repertoires()] -#' or [read_immundata()]. Must contain the `$annotations` table with columns -#' specified in `schema` and internal columns like `imd_receptor_id` and -#' `imd_chain_count`. -#' @param schema Character vector. Column name(s) in `idata$annotations` that -#' define a unique repertoire. For example, `c("SampleID")` or -#' `c("DonorID", "TimePoint")`. Columns must exist in `idata$annotations`. -#' Default: `"repertoire_id"` (assumes such a column exists). -#' -#' @details -#' The function operates on the `idata$annotations` table: -#' 1. **Validation:** Checks `idata` and existence of `schema` columns. Removes -#' any pre-existing repertoire summary columns to prevent duplication. -#' 2. **Repertoire Definition:** Groups annotations by the `schema` columns. -#' Calculates total counts (`n_barcodes`) per group. Assigns a unique integer -#' `imd_repertoire_id` to each distinct repertoire group. This forms the -#' initial `repertoires_table`. -#' 3. **Receptor Counts & Proportion:** Calculates the sum of `imd_chain_count` -#' for each receptor within each repertoire (`imd_count`). Calculates the -#' proportion (`imd_proportion`) of each receptor within its repertoire. -#' 4. **Repertoire & Receptor Stats:** Counts unique receptors per repertoire -#' (`n_receptors`, added to `repertoires_table`). Counts the number of -#' distinct repertoires each unique receptor appears in (`n_repertoires`). -#' 5. **Join Results:** Joins the calculated `imd_count`, `imd_proportion`, and -#' `n_repertoires` back to the annotation table based on repertoire columns -#' and `imd_receptor_id`. -#' 6. **Return New Object:** Creates and returns a *new* `ImmunData` object -#' containing the updated `$annotations` table (with the added statistics) -#' and the `$repertoires` slot populated with the `repertoires_table` -#' (containing `schema` columns, `imd_repertoire_id`, `n_barcodes`, `n_receptors`). -#' -#' The original `idata` object remains unmodified. Internal column names are -#' typically managed by `immundata:::imd_schema()`. -#' -#' @return A **new** `ImmunData` object. Its `$annotations` table includes the -#' added columns (`imd_repertoire_id`, `imd_count`, `imd_proportion`, `n_repertoires`). -#' Its `$repertoires` slot contains the summary table linking `schema` columns -#' to `imd_repertoire_id`, `n_barcodes`, and `n_receptors`. -#' -#' @seealso [read_repertoires()] (which can call this function), [ImmunData] class. -#' -#' @concept aggregation -#' @export -#' -#' @examples -#' \dontrun{ -#' # Assume 'idata_raw' is an ImmunData object loaded via read_repertoires -#' # but *without* providing 'repertoire_schema' initially. -#' # It has $annotations but $repertoires is likely NULL or empty. -#' # Assume idata_raw$annotations has columns "SampleID" and "TimePoint". -#' -#' # Define repertoires based on SampleID and TimePoint -#' idata_aggregated <- agg_repertoires(idata_raw, schema = c("SampleID", "TimePoint")) -#' -#' # Explore the results -#' print(idata_aggregated) -#' print(idata_aggregated$repertoires) -#' print(head(idata_aggregated$annotations)) # Note the new columns -#' } -agg_repertoires <- function(idata, schema = "repertoire_id") { - checkmate::assert_r6(idata, "ImmunData") - checkmate::assert_character(schema, min.len = 1) - - missing_cols <- setdiff(schema, colnames(idata$annotations)) - if (length(missing_cols) > 0) { - stop( - "Missing columns in `annotations`: ", - paste(missing_cols, collapse = ", ") - ) - } - - receptor_id <- imd_schema()$receptor - repertoire_id <- imd_schema()$repertoire - repertoire_schema_sym <- to_sym(schema) - prop_col <- imd_schema()$proportion - imd_count_col <- imd_schema("count") - chain_count_col <- imd_schema("chain_count") - n_receptors_col <- imd_schema("n_receptors") - n_barcodes_col <- imd_schema("n_barcodes") - n_repertoires_col <- imd_schema("n_repertoires") - - cols_to_drop <- c(repertoire_id, imd_count_col, prop_col, n_receptors_col, n_barcodes_col, n_repertoires_col) - - new_annotations <- idata$annotations |> select(-any_of(cols_to_drop)) - - repertoires_table <- new_annotations |> - summarise( - .by = schema, - n_barcodes = sum(!!to_sym(chain_count_col)) - ) |> - mutate( - {{ repertoire_id }} := row_number() - ) |> - relocate({{ repertoire_id }}) - - # - # proportions - # - receptor_cells <- new_annotations |> summarise( - .by = c(schema, receptor_id), - {{ imd_count_col }} := sum(!!rlang::sym(chain_count_col)) - ) - - receptor_props <- receptor_cells |> - left_join(repertoires_table, by = schema) |> - mutate({{ prop_col }} := !!rlang::sym(imd_count_col) / n_barcodes) |> - select(-n_barcodes) - - new_annotations <- new_annotations |> - left_join(receptor_props, by = c(schema, receptor_id)) - - # - # n_repertoires & n_receptors - # - unique_receptors <- new_annotations |> - distinct(!!rlang::sym(receptor_id), !!rlang::sym(repertoire_id)) - - n_receptor_df <- unique_receptors |> - summarise(.by = !!rlang::sym(repertoire_id), n_receptors = n()) - - repertoires_table <- repertoires_table |> left_join(n_receptor_df, by = repertoire_id) - - repertoire_counts <- unique_receptors |> - summarise(.by = all_of(receptor_id), n_repertoires = n()) - - new_annotations <- new_annotations |> left_join(repertoire_counts, by = receptor_id) - - ImmunData$new( - schema = idata$schema_receptor, - annotations = new_annotations, - repertoires = repertoires_table - ) -} - - -#' @title Aggregates AIRR data into receptors -#' -#' @description -#' Processes a table of immune receptor sequences (chains or clonotypes) to -#' identify unique receptors based on a specified schema. It assigns a unique -#' identifier (`imd_receptor_id`) to each distinct receptor signature and -#' returns an annotated table linking the original sequence data to these -#' receptor IDs. -#' -#' This function is a core component used within [read_repertoires()] and handles -#' different input data structures: -#' * Simple tables (no counts, no cell IDs). -#' * Bulk sequencing data (using a count column). -#' * Single-cell data (using a barcode/cell ID column). For single-cell data, -#' it can perform chain pairing if the schema specifies multiple chains -#' (e.g., TRA and TRB). -#' -#' @param dataset A data frame or `duckplyr_df` containing sequence/clonotype data. -#' Must include columns specified in `schema` and potentially `barcode_col`, -#' `count_col`, `locus_col`, `umi_col`. Could be `idata$annotations`. -#' @param schema Defines how a unique receptor is identified. Can be: -#' * A character vector of column names representing receptor features -#' (e.g., `c("v_call", "j_call", "junction_aa")`). -#' * A list created by `make_receptor_schema()`, specifying both `features` -#' (character vector) and optionally `chains` (character vector of locus -#' names like `"TRA"`, `"TRB"`, `"IGH"`, `"IGK"`, `"IGL"`, max length 2). -#' Specifying `chains` triggers filtering by locus and enables pairing logic -#' if two chains are given. -#' @param barcode_col Character(1). The name of the column containing cell -#' identifiers (barcodes). Required for single-cell processing and chain pairing. -#' Default: `NULL`. -#' @param count_col Character(1). The name of the column containing counts -#' (e.g., UMI counts for bulk, clonotype frequency). Used for bulk data -#' processing. Default: `NULL`. Cannot be specified if `barcode_col` is set. -#' @param locus_col Character(1). The name of the column specifying the chain locus -#' (e.g., "TRA", "TRB"). Required if `schema` includes `chains` for filtering -#' or pairing. Default: `NULL`. -#' @param umi_col Character(1). The name of the column containing UMI counts. -#' Required for *paired-chain single-cell* data (`length(schema$chains) == 2`). -#' Used to select the most abundant chain per locus within a cell when multiple -#' chains of the same locus are present. Default: `NULL`. -#' -#' @details -#' The function performs the following main steps: -#' 1. **Validation:** Checks inputs, schema validity, and existence of required columns. -#' 2. **Schema Parsing:** Determines receptor features and target chains from `schema`. -#' 3. **Locus Filtering:** If `schema$chains` is provided, filters the dataset -#' to include only rows matching the specified locus/loci. -#' 4. **Processing Logic (based on `barcode_col` and `count_col`):** -#' * **Simple Table/Bulk (No Barcodes):** Assigns unique internal barcode/chain IDs. -#' Identifies unique receptors based on `schema$features`. Calculates -#' `imd_chain_count` (1 for simple table, from `count_col` for bulk). -#' * **Single-Cell (Barcodes Provided):** Uses `barcode_col` for `imd_barcode_id`. -#' * **Single Chain:** (`length(schema$chains) <= 1`). Identifies unique -#' receptors based on `schema$features`. `imd_chain_count` is 1. -#' * **Paired Chain:** (`length(schema$chains) == 2`). Requires `locus_col` -#' and `umi_col`. Filters chains within each cell/locus group based -#' on max `umi_col`. Creates paired receptors by joining the two -#' specified loci for each cell based on `schema$features` from both. -#' Assigns a unique `imd_receptor_id` to each *pair*. -#' `imd_chain_count` is 1 (representing the chain record). -#' 5. **Output:** Returns an annotated data frame containing original columns plus -#' internal identifiers (`imd_receptor_id`, `imd_barcode_id`, `imd_chain_id`) -#' and counts (`imd_chain_count`). -#' -#' Internal column names are typically managed by `immundata:::imd_schema()`. -#' -#' @return A `duckplyr_df` (or data frame) representing the annotated sequences. -#' This table links each original sequence record (chain) to a defined receptor -#' and includes standardized columns: -#' * `imd_receptor_id`: Integer ID unique to each distinct receptor signature. -#' * `imd_barcode_id`: Integer ID unique to each cell/barcode (or row if no barcode). -#' * `imd_chain_id`: Integer ID unique to each input row (chain). -#' * `imd_chain_count`: Integer count associated with the chain (1 for SC/simple, -#' from `count_col` for bulk). -#' This output is typically assigned to the `$annotations` field of an `ImmunData` object. -#' -#' @seealso [read_repertoires()], [make_receptor_schema()], [ImmunData] -#' -#' @concept aggregation -#' @export -agg_receptors <- function(dataset, schema, barcode_col = NULL, count_col = NULL, locus_col = NULL, umi_col = NULL) { - checkmate::assert_data_frame(dataset) - checkmate::check_character(barcode_col, max.len = 1, null.ok = TRUE) - checkmate::check_character(count_col, max.len = 1, null.ok = TRUE) - checkmate::check_character(locus_col, max.len = 1, null.ok = TRUE) - checkmate::check_character(locus_col, max.len = 1, null.ok = TRUE) - - if (checkmate::test_character(schema, min.len = 1)) { - schema <- make_receptor_schema(schema) - } else if (assert_receptor_schema(schema)) { - if (!is.null(schema$locus)) { - if (is.null(locus_col)) { - cli::cli_abort("Found issues with the schema. The passed schema has a `chain` to aggregate receptors by, but `'locus_col'` is NULL. Please provide `'locus_col'` or aggregate receptors without using several chains.") - } else if (is.null(barcode_col) && length(schema$locus) == 2) { - cli::cli_abort("Found issues with the schema. The passed schema has a `chain` to aggregate receptors by, but `'barcode_col'` is NULL. Please provide `'barcode_col'` or aggregate receptors without using several chains.") - } - } - } else { - cli::cli_abort("Found issues with the schema. Please either pass one or several column names or use function {.run immundata::check_receptor_schema()} to create a schema.") - } - - receptor_features <- imd_receptor_features(schema) - receptor_chains <- imd_receptor_chains(schema) - - receptor_cols_existence <- setdiff(c(receptor_features, locus_col), colnames(dataset)) - if (length(receptor_cols_existence) != 0) { - cli::cli_abort("Not all columns in the receptor schema present in the data: [{receptor_cols_existence}]. Please double check and run again.") - } - - # TODO: - # if (checkmate::test_r6(idata, "ImmunData")) { - # dataset <- idata$annotations - # } else { - # dataset <- idata - # } - - immundata_barcode_col <- imd_schema("barcode") - immundata_receptor_id_col <- imd_schema("receptor") - immundata_chain_id_col <- imd_schema("chain") - immundata_count_col <- imd_schema("count") - immundata_chain_count <- imd_schema("chain_count") - - # TODO: refactor - if (!is.null(locus_col)) { - if (locus_col != imd_schema("locus")) { - cli::cli_alert_info("Renaming {locus_col} to {imd_schema('locus')}") - - locus_col <- imd_schema("locus") - names(locus_col) <- locus_col - - dataset <- rename(locus_col) - } - } - - # TODO: - # Prefilter locus - if (is.null(receptor_chains)) { - cli::cli_alert_info("No locus information found") - } else if (length(receptor_chains) == 1) { - # '==' should be faster than 'in' hence a separate use case. - dataset <- dataset |> filter(!!rlang::sym(locus_col) == receptor_chains) - cli::cli_alert_info("Found target locus: {receptor_chains}. The dataset will be pre-filtered to leave chains for this locus only") - } else { - dataset <- dataset |> filter(!!rlang::sym(locus_col) %in% receptor_chains) - cli::cli_alert_info("Found locus pair: {receptor_chains}. The dataset will be pre-filtered to leave chains for these loci only") - } - - - # - # 1) Case #1: simple receptor table - no barcodes, no count column - # - if (is.null(barcode_col) && is.null(count_col)) { - cli::cli_alert_info("Processing data as immune repertoire tables - no counts, no barcodes, no chain pairing possible") - - dataset <- dataset |> - mutate( - {{ immundata_barcode_col }} := row_number(), - {{ immundata_chain_id_col }} := !!rlang::sym(immundata_barcode_col) - ) - - receptor_data <- dataset |> - summarise(.by = all_of(receptor_features)) |> - mutate( - {{ immundata_receptor_id_col }} := row_number() - ) - - annotation_data <- dataset |> - left_join(receptor_data, by = receptor_features) |> - mutate( - {{ immundata_chain_count }} := 1, - {{ immundata_count_col }} := 0 - ) - } - - # - # 2) Case 2: bulk data - no barcodes, but with the count column - # - else if (is.null(barcode_col) && !is.null(count_col)) { - cli::cli_alert_info("Processing data as bulk sequencing immune repertoires - with counts, no barcodes, no chain pairing possible") - - dataset <- dataset |> - mutate( - {{ immundata_barcode_col }} := row_number(), - {{ immundata_chain_id_col }} := !!rlang::sym(immundata_barcode_col) - ) - - receptor_data <- dataset |> - summarise(.by = all_of(receptor_features)) |> - mutate( - {{ immundata_receptor_id_col }} := row_number() - ) - - annotation_data <- dataset |> - left_join(receptor_data, by = receptor_features) |> - mutate( - {{ immundata_chain_count }} := !!rlang::sym(count_col), - {{ immundata_count_col }} := 0 - ) - } - - # - # 3) Case 3: single-cell data - barcodes, no counts - # - else if (!is.null(barcode_col) && is.null(count_col)) { - cli::cli_alert_info("Processing data as single-cell sequencing immune repertoires - no counts, with barcodes, chain pairing is possible") - - dataset <- dataset |> - mutate( - {{ immundata_barcode_col }} := !!rlang::sym(barcode_col), - {{ immundata_chain_id_col }} := row_number() - ) - - # - # 3.1) Case 3.1: single chain - # - if (length(receptor_chains) <= 1) { - receptor_data <- dataset |> - summarise(.by = all_of(receptor_features)) |> - mutate( - {{ immundata_receptor_id_col }} := row_number() - ) - - annotation_data <- dataset |> - left_join(receptor_data, by = receptor_features) |> - mutate({{ immundata_chain_count }} := 1, {{ immundata_count_col }} := 0) - } - - # - # 3.2) Case 3.2: paired chain - # - else if (length(receptor_chains) == 2) { - locus_1 <- receptor_chains[1] - locus_2 <- receptor_chains[2] - - # Step 1: filter out bad chains, i.e., find the most abundance pairs of chains per barcode per locus - receptor_data <- dataset |> - select(all_of(c(immundata_barcode_col, umi_col, locus_col))) |> - mutate( - .by = c(immundata_barcode_col, locus_col), - temp__reads = max(!!rlang::sym(umi_col), na.rm = TRUE) - ) |> - filter(!!rlang::sym(umi_col) == temp__reads) - # distinct(!!rlang::sym(immundata_barcode_col), !!rlang::sym(locus_col), .keep_all = TRUE) - - # TODO: what if temp_reads == max with several receptors? - - # Step 2: create receptors by self-join - # TODO: optimize plz - - r1 <- receptor_data |> - filter(!!rlang::sym(locus_col) == locus_1) - r2 <- receptor_data |> - filter(!!rlang::sym(locus_col) == locus_2) - - receptor_data <- r1 |> - semi_join( - r2, - by = immundata_barcode_col - ) |> - mutate( - {{ immundata_receptor_id_col }} := row_number() - ) |> - select(all_of(c( - immundata_receptor_id_col, - immundata_barcode_col - ))) - - annotation_data <- receptor_data |> - left_join( - dataset, - by = immundata_barcode_col - ) |> - mutate( - {{ immundata_chain_count }} := 1, - {{ immundata_count_col }} := 0 - ) - } - - # - # Case 3.3: unsupported multiple chain - # - else { - cli::cli_abort("Unsupported case: more than two chains in [{receptor_chains}]") - } - } else { - # - # 4) Something weird is happening... - # - cli_abort("Undefined case: passed column names for both cell identifiers and receptor counts.") - } - - annotation_data -} - - -#' @title Create or validate a receptor schema object -#' -#' @description -#' Helper functions for defining and validating the `schema` used by -#' [agg_receptors()] to identify unique receptors. -#' -#' `make_receptor_schema()` creates a schema list object. -#' `assert_receptor_schema()` checks if an object is a valid schema list and throws -#' an error if not. -#' `test_receptor_schema()` checks if an object is a valid schema list or a -#' character vector (which `agg_receptors` can also accept) and returns `TRUE` -#' or `FALSE`. -#' -#' @param features Character vector. Column names defining the features of a -#' single receptor chain (e.g., V gene, J gene, CDR3 sequence). -#' @param chains Optional character vector (max length 2). Locus names (e.g., -#' `"TRA"`, `"TRB"`) to filter by or pair. If `NULL` or length 1, only -#' filtering occurs. If length 2, pairing logic is enabled in [agg_receptors()]. -#' Default: `NULL`. -#' @param schema An object to test or assert as a valid schema. Can be a list -#' created by `make_receptor_schema` or a character vector (for `test_receptor_schema`). -#' -#' @return -#' `make_receptor_schema` returns a list with elements `features` and `chains`. -#' `assert_receptor_schema` returns `TRUE` invisibly if valid, or stops execution. -#' `test_receptor_schema` returns `TRUE` or `FALSE`. -#' -#' @rdname make_receptor_schema -#' @concept utils -#' @export -make_receptor_schema <- function(features, chains = NULL) { - checkmate::check_character(features, min.len = 1) - checkmate::check_character(chains, max.len = 2, null.ok = TRUE) - - list(features = features, chains = chains) -} - - -#' @rdname make_receptor_schema -#' @export -assert_receptor_schema <- function(schema) { - # TODO: globals.R with schema list - - checkmate::assert( - checkmate::test_character(schema, min.len = 1), - checkmate::test_list(schema, len = 2, null.ok = FALSE) && - checkmate::test_names(names(schema), must.include = c("features", "chains")) - ) -} - - -#' @rdname make_receptor_schema -#' @export -test_receptor_schema <- function(schema) { - checkmate::test_character(schema, min.len = 1) || ( - checkmate::test_list(schema, len = 2, null.ok = FALSE) && - checkmate::test_subset(names(schema), c("features", "chains"), empty.ok = FALSE) - ) -} diff --git a/R/operations_agg_receptors.R b/R/operations_agg_receptors.R new file mode 100644 index 0000000..bae4146 --- /dev/null +++ b/R/operations_agg_receptors.R @@ -0,0 +1,483 @@ +#' @title Group AIRR sequence rows into receptors +#' +#' @description +#' `agg_receptors()` is a low-level function used during AIRR data ingestion. It +#' decides which sequence rows represent the same biological receptor and adds +#' package-standard identifiers and counts to the input table. +#' +#' A receptor can be one chain or a pair of chains from the same cell. The +#' `schema` argument defines which sequence features and loci make two receptors +#' identical. +#' +#' This function works with a prepared duckplyr table and returns a duckplyr +#' table. It does not accept or return an [ImmunData] object. Most analysis +#' workflows should provide the same arguments to [read_repertoires()], which +#' calls `agg_receptors()` during import. +#' +#' @param dataset A duckplyr table containing AIRR sequence data, with one row +#' per chain or bulk clonotype. It must contain the columns named in `schema` +#' and in any of `barcode_col`, `count_col`, `locus_col`, and `umi_col` that +#' are supplied. +#' @param schema Definition of receptor identity. Supply either: +#' +#' * A character vector naming the features that must match, such as +#' `c("v_call", "j_call", "junction_aa")`. +#' * An object created by [make_receptor_schema()]. Its `features` define chain +#' identity, while its optional `chains` select one locus or define a pair +#' of loci. +#' +#' A schema can contain at most two chain entries. Use syntax such as +#' `c("IGH", "IGL|IGK")` to accept either IGH-IGL or IGH-IGK pairs. +#' @param barcode_col Name of the column containing cell barcodes. Supply this +#' for single-cell data. `umi_col` is then also required, and `count_col` +#' cannot be supplied. If `imd_filename` is present, identical barcode values +#' from different source files are treated as different cells. The default is +#' `NULL`. +#' @param count_col Name of the column containing non-negative abundance values +#' in bulk repertoire data. These values are copied to `imd_n_chains`. +#' `count_col` cannot be used together with `barcode_col`. The default is +#' `NULL`. +#' @param locus_col Name of the column containing loci such as `"TRA"`, `"TRB"`, +#' or `"IGH"`. It is required when `schema` specifies one or more chains. The +#' column is renamed to the standard name `locus` when necessary. The default +#' is `NULL`. +#' @param umi_col Name of the column containing per-chain UMI or read counts. +#' It is required when `barcode_col` is supplied and is used to choose one +#' chain when a cell contains several chains from the same locus. The default +#' is `NULL`. +#' @param verbose Whether to print information about the selected processing +#' mode and loci. Defaults to `getOption("immundata.verbose", TRUE)`. +#' +#' @details +#' The receptor features are the columns that define the identity of one chain. +#' Two chains with the same values in all feature columns receive the same +#' receptor identity in a single-chain analysis. Common features include V gene, +#' J gene, and CDR3 amino acid sequence. +#' +#' The function supports three input modes: +#' +#' * **Uncounted sequence table:** If neither `barcode_col` nor `count_col` is +#' supplied, every input row is treated as one observed chain. A synthetic +#' barcode is created for each row, and `imd_n_chains` is set to `1`. +#' * **Bulk repertoire:** If `count_col` is supplied, every input row receives a +#' synthetic barcode and its abundance is copied to `imd_n_chains`. +#' * **Single-cell repertoire:** If `barcode_col` is supplied, rows are grouped +#' by cell. `umi_col` is required and `imd_n_chains` is set to `1` for every +#' retained cell-chain observation. +#' +#' When one chain is specified in `schema`, only that locus is retained. If a +#' cell contains several chains from that locus, the row with the highest value +#' in `umi_col` is retained. If the highest values are tied, the first row is +#' retained. +#' +#' When two chains are specified, only cells containing both requested loci are +#' retained. The selected chains are paired by barcode, and both rows receive +#' the same `imd_receptor_id`. Cells with incomplete pairs are excluded. +#' +#' A relaxed pair such as `c("IGH", "IGL|IGK")` requires IGH and exactly one of +#' the two alternative light-chain loci. Cells containing both IGL and IGK are +#' excluded. +#' +#' Numeric `imd_receptor_id` values identify receptors within the returned +#' table. The particular number assigned to a receptor is not a biological +#' identifier and may change when the data are aggregated again. +#' +#' @return A duckplyr table containing the retained input rows and these +#' package-standard columns: +#' +#' * `imd_receptor_id`: links rows that belong to the same receptor. +#' * `imd_barcode`: contains the input cell barcode, or a synthetic row-level +#' barcode for uncounted and bulk data. +#' * `imd_chain_id`: identifies an individual retained chain row. +#' * `imd_n_chains`: contains `1` for uncounted and single-cell data, or the +#' value from `count_col` for bulk data. +#' * `imd_count`: initialized to `0`; receptor counts are calculated later by +#' [agg_repertoires()]. +#' +#' @seealso [read_repertoires()], [make_receptor_schema()], [agg_repertoires()], +#' [ImmunData] +#' +#' @concept aggregation +#' @export +agg_receptors <- function(dataset, schema, barcode_col = NULL, count_col = NULL, locus_col = NULL, umi_col = NULL, + verbose = getOption("immundata.verbose", TRUE)) { + checkmate::assert_data_frame(dataset) + checkmate::assert_string(barcode_col, min.chars = 1, null.ok = TRUE) + checkmate::assert_string(count_col, min.chars = 1, null.ok = TRUE) + checkmate::assert_string(locus_col, min.chars = 1, null.ok = TRUE) + checkmate::assert_string(umi_col, min.chars = 1, null.ok = TRUE) + checkmate::assert_flag(verbose) + + if (!is.null(barcode_col) && !is.null(count_col)) { + cli::cli_abort("Please pass either {.arg barcode_col} (single-cell mode) or {.arg count_col} (bulk mode), not both.") + } + + if (checkmate::test_character(schema, min.len = 1)) { + schema <- make_receptor_schema(schema) + } else if (assert_receptor_schema(schema)) { + if (!is.null(schema$chains)) { + if (is.null(locus_col)) { + cli::cli_abort("Found issues with the schema. The passed schema has a `chain` to aggregate receptors by, but `'locus_col'` is NULL. Please provide `'locus_col'` or aggregate receptors without using several chains.") + } else if (is.null(barcode_col) && length(schema$chains) == 2) { + cli::cli_abort("Found issues with the schema. The passed schema has a `chain` to aggregate receptors by, but `'barcode_col'` is NULL. Please provide `'barcode_col'` or aggregate receptors without using several chains.") + } + } + } else { + cli::cli_abort("Found issues with the schema. Please either pass one or several column names or use function {.run immundata::check_receptor_schema()} to create a schema.") + } + + receptor_features <- imd_receptor_features(schema) + receptor_chains <- imd_receptor_chains(schema) + + if (!is.null(barcode_col) && is.null(umi_col)) { + cli::cli_abort("Single-cell mode requires {.arg umi_col}. Please provide the column with per-chain UMI/reads to resolve chain multiplicity within each barcode.") + } + + is_relaxed_pairing <- FALSE + relaxed_chain_alternatives <- NULL + parsed_chains <- receptor_chains + + if (!is.null(receptor_chains)) { + if (length(receptor_chains) == 2) { + if (grepl("\\|", receptor_chains[2])) { + is_relaxed_pairing <- TRUE + relaxed_chain_alternatives <- trimws(unlist(strsplit(receptor_chains[2], "\\|"))) + parsed_chains <- c(receptor_chains[1], relaxed_chain_alternatives) + } + } + } + + receptor_cols_existence <- setdiff(receptor_features, colnames(dataset)) + if (length(receptor_cols_existence) != 0) { + cli::cli_abort("Missing receptor feature column(s) required by {.arg schema}: [{receptor_cols_existence}].") + } + + required_arg_cols <- c( + if (!is.null(locus_col)) stats::setNames(locus_col, "locus_col"), + if (!is.null(barcode_col)) stats::setNames(barcode_col, "barcode_col"), + if (!is.null(count_col)) stats::setNames(count_col, "count_col"), + if (!is.null(umi_col)) stats::setNames(umi_col, "umi_col") + ) + + missing_arg_cols <- setdiff(unname(required_arg_cols), colnames(dataset)) + if (length(missing_arg_cols) != 0) { + missing_args <- names(required_arg_cols)[match(missing_arg_cols, required_arg_cols)] + cli::cli_abort( + "Missing column(s) referenced by arguments: [{missing_arg_cols}] (from [{missing_args}])." + ) + } + + if (!is.null(count_col)) { + negative_count_summary <- dataset |> + filter(!!rlang::sym(count_col) < 0) |> + summarise(n_negative = n()) |> + collect() + + if (negative_count_summary$n_negative[[1]] > 0) { + cli::cli_abort( + "Bulk counts in {.field {count_col}} must be non-negative." + ) + } + } + + # TODO: + # if (checkmate::test_r6(idata, "ImmunData")) { + # dataset <- idata$annotations + # } else { + # dataset <- idata + # } + + immundata_barcode_col <- imd_schema("barcode") + immundata_filename_col <- imd_schema("manifest_filename") + immundata_receptor_id_col <- imd_schema("receptor") + immundata_chain_id_col <- imd_schema("chain") + immundata_count_col <- imd_schema("count") + immundata_chain_count_col <- imd_schema("chain_count") + + # TODO: refactor + if (!is.null(locus_col)) { + canonical_locus_col <- imd_schema("locus") + + if (locus_col != canonical_locus_col) { + original_locus_col <- locus_col + + if (canonical_locus_col %in% colnames(dataset)) { + cli::cli_abort( + "Cannot standardize {.arg locus_col}: the dataset contains both the custom locus column {.field {original_locus_col}} and the canonical locus column {.field {canonical_locus_col}}." + ) + } + + if (verbose) { + cli::cli_alert_info("Renaming {original_locus_col} to {canonical_locus_col}") + } + + dataset <- dataset |> + rename(!!canonical_locus_col := all_of(original_locus_col)) + locus_col <- canonical_locus_col + } + } + + # Prefilter locus + if (is.null(receptor_chains)) { + if (verbose) { + cli::cli_alert_info("No locus information found") + } + } else if (length(parsed_chains) == 1) { + dataset <- dataset |> filter(!!rlang::sym(locus_col) == parsed_chains) + if (verbose) { + cli::cli_alert_info("Found target locus: {parsed_chains}. The dataset will be pre-filtered to leave chains for this locus only") + } + } else { + dataset <- dataset |> filter(!!rlang::sym(locus_col) %in% parsed_chains) + if (verbose) { + if (is_relaxed_pairing) { + cli::cli_alert_info("Found relaxed locus pair: {receptor_chains[1]} + ({receptor_chains[2]}). The dataset will be pre-filtered to leave chains for these loci only") + } else { + cli::cli_alert_info("Found locus pair: {receptor_chains}. The dataset will be pre-filtered to leave chains for these loci only") + } + } + } + + + # + # 1) Case #1: simple receptor table - no barcodes, no count column + # + if (is.null(barcode_col) && is.null(count_col)) { + if (verbose) { + cli::cli_alert_info("Processing data as immune repertoire tables - no counts, no barcodes, no chain pairing possible") + } + + dataset <- dataset |> + mutate( + {{ immundata_barcode_col }} := row_number(), + {{ immundata_chain_id_col }} := !!rlang::sym(immundata_barcode_col) + ) + + receptor_data <- dataset |> + summarise(.by = all_of(receptor_features)) |> + mutate( + {{ immundata_receptor_id_col }} := row_number() + ) + + annotation_data <- dataset |> + left_join(receptor_data, by = receptor_features) |> + mutate( + {{ immundata_chain_count_col }} := 1, + {{ immundata_count_col }} := 0 + ) + } + + # + # 2) Case #2: bulk data - no barcodes, but with the count column + # + else if (is.null(barcode_col) && !is.null(count_col)) { + if (verbose) { + cli::cli_alert_info("Processing data as bulk sequencing immune repertoires - with counts, no barcodes, no chain pairing possible") + } + + dataset <- dataset |> + mutate( + {{ immundata_barcode_col }} := row_number(), + {{ immundata_chain_id_col }} := !!rlang::sym(immundata_barcode_col) + ) + + receptor_data <- dataset |> + summarise(.by = all_of(receptor_features)) |> + mutate( + {{ immundata_receptor_id_col }} := row_number() + ) + + annotation_data <- dataset |> + left_join(receptor_data, by = receptor_features) |> + mutate( + {{ immundata_chain_count_col }} := !!rlang::sym(count_col), + {{ immundata_count_col }} := 0 + ) + } + + # + # 3) Case #3: single-cell data - barcodes, no counts + # + else if (!is.null(barcode_col) && is.null(count_col)) { + if (verbose) { + cli::cli_alert_info("Processing data as single-cell sequencing immune repertoires - no counts, with barcodes, chain pairing is possible") + } + + dataset <- dataset |> + mutate( + {{ immundata_barcode_col }} := !!rlang::sym(barcode_col), + {{ immundata_chain_id_col }} := row_number() + ) + + # Raw barcodes are only unique within their source library. Scope all + # single-cell selection and pairing operations by the source filename when + # it is available, while retaining the raw barcode in `imd_barcode`. + cell_group_cols <- c( + if (immundata_filename_col %in% colnames(dataset)) immundata_filename_col, + immundata_barcode_col + ) + + # + # 3.1) Case #3.1: single chain + # + if (length(receptor_chains) <= 1) { + # We still need to filter out receptors from barcodes + # with more than one receptor + + filtered_chains <- dataset |> + select(all_of(c( + immundata_chain_id_col, + cell_group_cols, + umi_col + ))) |> + mutate( + .by = all_of(cell_group_cols), + temp__reads = max(!!rlang::sym(umi_col), na.rm = TRUE) + ) |> + filter(!!rlang::sym(umi_col) == temp__reads) |> + distinct(!!!rlang::syms(cell_group_cols), .keep_all = TRUE) |> + select(all_of(c(cell_group_cols, immundata_chain_id_col))) + + dataset <- dataset |> + semi_join(filtered_chains, by = immundata_chain_id_col) + + receptor_data <- dataset |> + summarise(.by = all_of(receptor_features)) |> + mutate( + {{ immundata_receptor_id_col }} := row_number() + ) + + annotation_data <- dataset |> + left_join(receptor_data, by = receptor_features) |> + mutate({{ immundata_chain_count_col }} := 1, {{ immundata_count_col }} := 0) + } + + # + # 3.2) Case #3.2: paired chain + # + else if (length(receptor_chains) == 2) { + paired_receptor_features <- do.call(paste0, expand.grid(c(receptor_features, locus_col), c(".x", ".y"))) + + locus_1 <- parsed_chains[1] + locus_2 <- parsed_chains[2] + + if (is_relaxed_pairing) { + locus_3 <- parsed_chains[3] + } + + # Step 1: find the target chains: + # - find the most abundant pairs of chains per barcode per locus + filtered_chains <- dataset |> + select(all_of(c( + immundata_chain_id_col, + cell_group_cols, + umi_col, + locus_col + ))) |> + mutate( + .by = all_of(c(cell_group_cols, locus_col)), + temp__reads = max(!!rlang::sym(umi_col), na.rm = TRUE) + ) |> + filter(!!rlang::sym(umi_col) == temp__reads) |> + # If there are ties, keep the first one + distinct(!!!rlang::syms(c(cell_group_cols, locus_col)), .keep_all = TRUE) |> + select(all_of(c(cell_group_cols, locus_col, immundata_chain_id_col))) + + if (!is_relaxed_pairing) { + # - find barcodes with both loci + valid_barcodes <- filtered_chains |> + summarise( + .by = all_of(cell_group_cols), + n = n() + ) |> + filter(n == 2) + } else { + # - find barcodes with one main locus and only one of the alternative loci + valid_barcodes <- filtered_chains |> + summarise( + .by = all_of(cell_group_cols), + has_l1 = any(!!rlang::sym(locus_col) == locus_1), + has_l2 = any(!!rlang::sym(locus_col) == locus_2), + has_l3 = any(!!rlang::sym(locus_col) == locus_3), + ) |> + filter(.data$has_l1, (.data$has_l2 & !.data$has_l3) | (!.data$has_l2 & .data$has_l3)) + } + + # - get back to chains to select only those which are paired + filtered_chains <- filtered_chains |> + semi_join(valid_barcodes, + by = cell_group_cols + ) + + # Looks like back-and-forth, but I'm not sure how to make it better, tbh + # Alternative: filter out bad barcodes first, but it would require n_distinct + # as a first step, so pretty much the same as currently. + # TODO: benchmark this + + # Step 2: create receptors and their identifiers by self-join + + annotated_filtered_chains <- dataset |> + select(all_of(c(receptor_features, locus_col, immundata_chain_id_col, cell_group_cols))) |> + semi_join(filtered_chains, by = immundata_chain_id_col) + + r1 <- annotated_filtered_chains |> + filter(!!rlang::sym(locus_col) == locus_1) + + if (!is_relaxed_pairing) { + r2 <- annotated_filtered_chains |> + filter(!!rlang::sym(locus_col) == locus_2) + } else { + r2 <- annotated_filtered_chains |> + filter(!!rlang::sym(locus_col) %in% c(locus_2, locus_3)) + } + + receptor_barcode_mapping <- r1 |> + left_join( + r2, + by = cell_group_cols + ) + + receptor_chain_mapping <- receptor_barcode_mapping |> + summarise( + .by = all_of(paired_receptor_features) + ) |> + mutate( + {{ immundata_receptor_id_col }} := row_number() + ) |> + right_join(receptor_barcode_mapping, + by = paired_receptor_features + ) |> + select(all_of(c(immundata_receptor_id_col, paste0(immundata_chain_id_col, c(".x", ".y"))))) + + receptor_chain_mapping <- union_all( + receptor_chain_mapping |> select(all_of(immundata_receptor_id_col), {{ immundata_chain_id_col }} := 2), + receptor_chain_mapping |> select(all_of(immundata_receptor_id_col), {{ immundata_chain_id_col }} := 3), + ) + + # Step 3: merge back + + annotation_data <- receptor_chain_mapping |> + left_join(dataset, + by = immundata_chain_id_col + ) |> + mutate( + {{ immundata_chain_count_col }} := 1, + {{ immundata_count_col }} := 0 + ) + } + + # + # Case 3.3: unsupported multiple chain + # + else { + cli::cli_abort("Unsupported case: more than two chains in [{receptor_chains}]") + } + } else { + # + # 4) Something weird is happening... + # + cli_abort("Undefined case: passed column names for both cell identifiers and receptor counts.") + } + + annotation_data +} diff --git a/R/operations_agg_repertoires.R b/R/operations_agg_repertoires.R new file mode 100644 index 0000000..98db3b8 --- /dev/null +++ b/R/operations_agg_repertoires.R @@ -0,0 +1,226 @@ +#' @title Define biological repertoires and calculate receptor abundance +#' +#' @description +#' Use `agg_repertoires()` to define which receptor observations belong to the +#' same biological repertoire and calculate receptor abundance within each +#' repertoire. +#' +#' Use this function after importing data without repertoire definitions, or +#' when you want to redefine repertoires using sample information. One +#' repertoire usually represents one biological sample. It can also represent +#' one sample and time-point combination. The columns in `schema` define these +#' groups. +#' +#' The unit being defined is the repertoire. The function does not remove chain +#' rows or redefine cells or receptors. It returns a new [ImmunData] object. The +#' original object is not changed. +#' +#' @details +#' The function calculates summaries at repertoire and receptor levels while +#' keeping the original chain rows. +#' +#' @section What the function calculates: +#' +#' The returned repertoire summary contains one row for each repertoire: +#' +#' * `imd_repertoire_id`: a new integer identifier for the repertoire. +#' * `n_barcodes`: the number of observed cells for single-cell data, or the +#' total abundance for bulk data. +#' * `n_receptors`: the number of distinct receptors in the repertoire. +#' +#' The function also adds these values to each chain row: +#' +#' * `imd_repertoire_id`: the repertoire containing the row. +#' * `imd_count`: the number of cells carrying that receptor in single-cell +#' data, or its summed abundance in bulk data, within the repertoire. +#' * `imd_proportion`: the receptor's fraction of the repertoire, calculated as +#' `imd_count / n_barcodes`. +#' * `n_repertoires`: the number of repertoires in which the receptor occurs. +#' +#' Values calculated for a receptor are repeated on all chain rows belonging to +#' that receptor in the same repertoire. +#' +#' Calling `agg_repertoires()` again replaces previous repertoire definitions, +#' receptor counts, proportions, and related strata summaries. +#' +#' @section Backend and storage: +#' +#' Large-table calculations run on the duckplyr annotation table. The annotation +#' data remain lazy when the input is lazy. The small repertoire summary is +#' collected and stored in the returned object. +#' +#' Aggregation can be expensive for a large dataset. After checking the result, +#' consider saving it so later analyses do not repeat the calculation. Use +#' `write_immundata(idata, tag = "by-sample")` to create a managed snapshot in +#' the object's project home. Managed snapshots are versioned, so another write +#' with the same tag creates a new version and keeps the earlier version. +#' +#' Use `write_immundata(idata, output_folder = "path/to/result")` when you need a +#' standalone saved state in a specific folder, for example to share it or to +#' choose a new storage location. Unlike a managed snapshot, writing to an +#' existing explicit folder replaces the ImmunData files in that folder. Both +#' forms materialize pending duckplyr calculations and return a disk-backed +#' object that can be reopened with [read_immundata()]. +#' +#' @param idata An [ImmunData] object containing receptor observations and the +#' columns named in `schema`. This is usually created by [read_repertoires()] +#' or [read_immundata()]. +#' @param schema A non-empty character vector. One or more column names that +#' together define a repertoire. For example, `"Sample"` creates one +#' repertoire per sample, and +#' `c("Sample", "TimePoint")` creates one repertoire per sample and time-point +#' combination. The default is `"repertoire_id"`; this column must exist if +#' the default is used. +#' @param verbose A logical value. Accepted for consistency with other +#' aggregation functions. It currently does not change the output. Defaults to +#' `getOption("immundata.verbose", TRUE)`. +#' +#' @return A new [ImmunData] object with repertoire definitions and abundance +#' statistics. Its repertoire summary contains the `schema` columns, +#' `imd_repertoire_id`, `n_barcodes`, and `n_receptors`. Its chain rows also +#' contain `imd_repertoire_id`, `imd_count`, `imd_proportion`, and +#' `n_repertoires`. +#' +#' @seealso [read_repertoires()], [agg_strata()], [write_immundata()], [ImmunData] +#' +#' @concept aggregation +#' @export +#' +#' @examples +#' library(immundata) +#' library(dplyr) +#' +#' options(immundata.verbose = FALSE) +#' +#' # Create a small bulk T-cell receptor dataset from two biological samples +#' bulk_data <- tibble( +#' Sample = c("Tumor", "Tumor", "Blood", "Blood"), +#' cdr3_aa = c("CASSA", "CASSB", "CASSA", "CASSC"), +#' v_call = c("TRBV1", "TRBV2", "TRBV1", "TRBV3"), +#' abundance = c(20L, 5L, 4L, 6L) +#' ) +#' +#' bulk_file <- tempfile(fileext = ".tsv") +#' readr::write_tsv(bulk_data, bulk_file) +#' +#' # Import receptors without defining repertoires +#' idata <- read_repertoires( +#' path = bulk_file, +#' schema = c("cdr3_aa", "v_call"), +#' count_col = "abundance", +#' repertoire_schema = NULL, +#' output_folder = tempfile("immundata-example-") +#' ) +#' +#' # Define one repertoire for each biological sample +#' sample_repertoires <- idata |> +#' agg_repertoires(schema = "Sample") +#' +#' sample_repertoires$repertoires |> +#' select(Sample, n_barcodes, n_receptors) |> +#' arrange(Sample) +#' # Expected result: +#' # Sample n_barcodes n_receptors +#' # Blood 10 2 +#' # Tumor 25 2 +#' +#' # For example, CASSA forms 80% of the Tumor repertoire and 40% of the +#' # Blood repertoire. It occurs in two repertoires. +#' +#' # For a large dataset, save the result as a managed snapshot so this +#' # aggregation does not need to run again. +#' saved_repertoires <- write_immundata( +#' sample_repertoires, +#' tag = "by-sample" +#' ) +agg_repertoires <- function(idata, schema = "repertoire_id", + verbose = getOption("immundata.verbose", TRUE)) { + checkmate::assert_r6(idata, "ImmunData") + checkmate::assert_character(schema, min.len = 1) + checkmate::assert_flag(verbose) + + missing_cols <- setdiff(schema, colnames(idata$annotations)) + if (length(missing_cols) > 0) { + stop( + "Missing columns in `annotations`: ", + paste(missing_cols, collapse = ", ") + ) + } + + receptor_id <- imd_schema("receptor") + repertoire_id <- imd_schema("repertoire") + prop_col <- imd_schema("proportion") + imd_count_col <- imd_schema("count") + barcode_col <- imd_schema("barcode") + chain_count_col <- imd_schema("chain_count") + n_receptors_col <- imd_schema("n_receptors") + n_barcodes_col <- imd_schema("n_barcodes") + n_repertoires_col <- imd_schema("n_repertoires") + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + + # Remove columns from the previous repertoire aggregation if any + cols_to_drop <- c(repertoire_id, strata_col, strata_name_col, imd_count_col, prop_col, n_receptors_col, n_barcodes_col, n_repertoires_col) + + new_annotations <- idata$annotations |> + select(-any_of(cols_to_drop)) + + single_chain_annotations <- new_annotations |> + # Deduplicate receptor/barcode rows without distinct(.keep_all = TRUE), + # which is unstable on duckdb 1.5.x due to the optimizer. + # https://github.com/duckdb/duckdb/issues/21348 + summarise( + .by = all_of(c(schema, receptor_id, barcode_col)), + {{ chain_count_col }} := dplyr::first(!!rlang::sym(chain_count_col)) + ) + + # + # proportions + # + receptor_cells <- single_chain_annotations |> + summarise( + .by = all_of(c(schema, receptor_id)), + {{ imd_count_col }} := sum(!!rlang::sym(chain_count_col)) + ) + + repertoires_table <- receptor_cells |> + summarise( + .by = all_of(schema), + !!n_barcodes_col := sum(!!rlang::sym(imd_count_col)), + !!n_receptors_col := n() + ) |> + arrange(!!!rlang::syms(schema)) |> + mutate( + {{ repertoire_id }} := row_number() + ) |> + relocate({{ repertoire_id }}) |> + collect() + + receptor_props <- receptor_cells |> + left_join(repertoires_table, by = schema, na_matches = "na") |> + mutate(!!prop_col := !!rlang::sym(imd_count_col) / !!rlang::sym(n_barcodes_col)) |> + select(-all_of(c(n_barcodes_col, n_receptors_col))) + + new_annotations <- new_annotations |> + left_join( + receptor_props, + by = c(schema, receptor_id), + na_matches = "na" + ) + + # + # n_repertoires + # + repertoire_counts <- receptor_cells |> + summarise(.by = all_of(receptor_id), n_repertoires = n()) + + new_annotations <- new_annotations |> + left_join(repertoire_counts, by = receptor_id, na_matches = "na") + + ImmunData$new( + schema = idata$schema_receptor, + annotations = new_annotations, + repertoires = repertoires_table, + provenance = get_provenance(idata) + ) +} diff --git a/R/operations_agg_strata.R b/R/operations_agg_strata.R new file mode 100644 index 0000000..acb4187 --- /dev/null +++ b/R/operations_agg_strata.R @@ -0,0 +1,345 @@ +#' @title Group repertoires into biological strata +#' +#' @description +#' Use `agg_strata()` to place sample repertoires into biological comparison +#' groups, such as treatment arms, tissues, or disease groups. +#' +#' Use this function after [agg_repertoires()] when several repertoires should be +#' analysed as one group. A *stratum* contains every repertoire with the same +#' value, or the same combination of values, in `schema`. +#' +#' The unit being grouped is a whole repertoire. The function returns a new +#' [ImmunData] object. The original object is not changed. +#' +#' @param idata An [ImmunData] object with repertoires already defined. Use +#' [agg_repertoires()] first if the object does not contain repertoires. +#' @param schema A non-empty character vector. One or more repertoire-level +#' columns that define a stratum. For example, use `"Therapy"` for treatment +#' arms or `c("Tissue", "Disease")` for each tissue and disease combination. +#' The columns must be present in `idata$repertoires`. +#' @param prefix A non-empty character string. Prefix for the +#' automatic stratum labels. The default, `"Strata"`, produces labels such as +#' `"Strata1"` and `"Strata2"`. You can use [rename_strata()] to assign meaningful +#' labels later instead. +#' +#' @return A new [ImmunData] object in which every repertoire belongs to one +#' stratum. The `$strata` table lists the strata, their defining biological +#' values, and their automatic labels. Repertoire definitions and summary +#' statistics are preserved. +#' +#' @details +#' If `schema` contains several columns, a separate stratum is created for each +#' observed combination. For example, `c("Tissue", "Therapy")` can define +#' separate blood and tumour strata within each treatment arm. +#' +#' Calling `agg_strata()` again replaces the existing strata with groups defined +#' by the new `schema`. +#' +#' @section Identifiers and storage: +#' +#' `imd_strata_id` is an internal identifier and can change when strata are +#' rebuilt. It is added to the repertoire table and to the underlying chain +#' annotations. `strata_name` is stored only in the smaller repertoire and +#' strata tables. +#' +#' Calling [agg_repertoires()] again rebuilds the repertoires, so it removes the +#' existing strata. Call `agg_strata()` again after redefining repertoires. +#' +#' @seealso [agg_repertoires()], [rename_strata()], [ImmunData] +#' +#' @concept aggregation +#' @export +#' +#' @examples +#' library(immundata) +#' library(dplyr) +#' +#' options(immundata.verbose = FALSE) +#' +#' # Define sample repertoires using the biological metadata in the test data +#' idata <- get_test_idata() |> +#' agg_repertoires(c("Response", "Therapy")) +#' +#' # Group the sample repertoires into treatment arms +#' treatment_groups <- idata |> +#' agg_strata(schema = "Therapy") +#' +#' treatment_groups$repertoires |> +#' select(Therapy, Response, imd_strata_id, strata_name) |> +#' arrange(imd_strata_id) +#' # Expected result: +#' # Therapy Response imd_strata_id strata_name +#' # CAR-T PR 1 Strata1 +#' # ICI FR 2 Strata2 +#' +#' # Each repertoire is now assigned to its treatment stratum. Any additional +#' # repertoire with the same Therapy value would receive the same stratum ID. +agg_strata <- function(idata, schema, prefix = "Strata") { + checkmate::assert_r6(idata, "ImmunData") + checkmate::assert_character(schema, min.len = 1, unique = TRUE, any.missing = FALSE) + checkmate::assert_string(prefix, min.chars = 1) + + if (is.null(idata$repertoires) || is.null(idata$schema_repertoire)) { + cli::cli_abort( + "Repertoire aggregation is required for {.fn agg_strata}. Run {.fn agg_repertoires} first." + ) + } + + repertoire_col <- imd_schema("repertoire") + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + + if (!(repertoire_col %in% colnames(idata$repertoires))) { + cli::cli_abort( + "Required column {.field {repertoire_col}} is missing in {.field idata$repertoires}." + ) + } + if (!(repertoire_col %in% colnames(idata$annotations))) { + cli::cli_abort( + "Required column {.field {repertoire_col}} is missing in {.field idata$annotations}." + ) + } + + rep_tbl_clean <- idata$repertoires |> + select(-any_of(c(strata_col, strata_name_col))) + + missing_schema_repertoires <- setdiff(schema, colnames(rep_tbl_clean)) + if (length(missing_schema_repertoires) > 0) { + cli::cli_abort( + "Column(s) [{missing_schema_repertoires}] specified in {.arg schema} are not found in {.field idata$repertoires}." + ) + } + + strata_defs <- rep_tbl_clean |> + select(all_of(schema)) |> + distinct() |> + arrange(!!!rlang::syms(schema)) |> + mutate( + {{ strata_col }} := row_number(), + {{ strata_name_col }} := paste0(prefix, .data[[strata_col]]) + ) |> + select(all_of(c(strata_col, strata_name_col, schema))) + + row_id_col <- ".__row_id" + + rep_tbl_stratified <- rep_tbl_clean |> + mutate(!!row_id_col := row_number()) |> + left_join( + strata_defs |> select(all_of(c(schema, strata_col, strata_name_col))), + by = schema + ) |> + arrange(!!rlang::sym(row_id_col)) |> + select(-all_of(row_id_col)) + + rep_to_strata <- rep_tbl_stratified |> + select(all_of(c(repertoire_col, strata_col))) |> + distinct() + + annotations_stratified <- idata$annotations |> + select(-any_of(strata_col)) |> + left_join( + duckplyr::as_duckdb_tibble(rep_to_strata), + by = repertoire_col + ) + + ImmunData$new( + schema = idata$schema_receptor, + annotations = annotations_stratified, + repertoires = rep_tbl_stratified, + strata = duckplyr::as_duckdb_tibble(strata_defs), + provenance = get_provenance(idata) + ) +} + + +#' @title Give biological strata readable labels +#' +#' @description +#' Use `rename_strata()` to replace automatic stratum labels with names that are +#' clear in figures and result tables, such as `"Control"`, `"Treated"`, or +#' `"Tumour tissue"`. +#' +#' Use this function after [agg_strata()] when labels such as `"Strata1"` do not +#' describe the biological groups. The unit being changed is the stratum label. +#' Stratum membership and the repertoires, receptors, cells, and chains remain +#' unchanged. +#' +#' The function returns a new [ImmunData] object. The original object is not +#' changed. +#' +#' @param idata An [ImmunData] object with strata already created by +#' [agg_strata()]. +#' @param names A named character vector or a data frame. New labels matched to +#' `imd_strata_id`. Supply either: +#' +#' * a named character vector, such as +#' `c("1" = "Control", "2" = "Treated")`; or +#' * a data frame with columns `imd_strata_id` and `strata_name`. +#' +#' Every new label must be non-empty and unique. +#' @param unnamed A character string. What to do when `names` does not include +#' every stratum. The default, `"error"`, asks for a complete mapping. Use +#' `"auto"` to generate labels for missing strata or `"keep"` to preserve +#' their current labels. +#' @param auto_prefix A non-empty character string. Prefix used to generate +#' labels when `unnamed = "auto"`. The default is `"Strata"`. +#' +#' @return A new [ImmunData] object with the requested labels in its `$strata` +#' and `$repertoires` tables. All biological group assignments and repertoire +#' summaries are preserved. +#' +#' @details +#' The names of a named character vector are the stratum IDs, not the current +#' labels. Inspect `idata$strata` to find the ID for each biological group. +#' +#' The mapping cannot contain unknown or repeated IDs, and the resulting labels +#' must be unique across strata. +#' +#' @section Storage details: +#' +#' The readable `strata_name` is stored in the repertoire and strata tables. The +#' underlying chain annotations keep only `imd_strata_id`, so renaming a stratum +#' does not rewrite or regroup chain-level data. +#' +#' @seealso [agg_strata()], [agg_repertoires()] +#' +#' @concept aggregation +#' @export +#' +#' @examples +#' library(immundata) +#' library(dplyr) +#' +#' options(immundata.verbose = FALSE) +#' +#' # Create treatment strata for the sample repertoires in the test data +#' treatment_groups <- get_test_idata() |> +#' agg_repertoires(c("Response", "Therapy")) |> +#' agg_strata(schema = "Therapy") +#' +#' # Replace automatic labels with names suitable for a figure +#' labeled_groups <- treatment_groups |> +#' rename_strata( +#' names = c("1" = "CAR-T arm", "2" = "ICI arm") +#' ) +#' +#' labeled_groups$strata |> +#' select(Therapy, imd_strata_id, strata_name) |> +#' arrange(imd_strata_id) +#' # Expected result: +#' # Therapy imd_strata_id strata_name +#' # CAR-T 1 CAR-T arm +#' # ICI 2 ICI arm +#' +#' # Only the labels changed. Each sample repertoire remains in the same +#' # treatment stratum. +rename_strata <- function(idata, names, unnamed = c("error", "auto", "keep"), auto_prefix = "Strata") { + checkmate::assert_r6(idata, "ImmunData") + checkmate::assert_string(auto_prefix, min.chars = 1) + unnamed <- match.arg(unnamed) + + if (is.null(idata$repertoires) || is.null(idata$schema_repertoire)) { + cli::cli_abort( + "Repertoire aggregation is required for {.fn rename_strata}. Run {.fn agg_repertoires} and {.fn agg_strata} first." + ) + } + + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + + rep_tbl <- idata$repertoires + + if (!(strata_col %in% colnames(rep_tbl))) { + cli::cli_abort( + "Required column {.field {strata_col}} is missing in {.field idata$repertoires}. Run {.fn agg_strata} first." + ) + } + + if (!(strata_name_col %in% colnames(rep_tbl))) { + rep_tbl[[strata_name_col]] <- paste0(auto_prefix, rep_tbl[[strata_col]]) + } + + if (checkmate::test_character(names, min.len = 1, names = "named", any.missing = FALSE)) { + map_tbl <- data.frame( + id = base::names(names), + label = unname(names), + stringsAsFactors = FALSE + ) + names(map_tbl) <- c(strata_col, strata_name_col) + } else if (checkmate::test_data_frame(names)) { + map_tbl <- names + required_cols <- c(strata_col, strata_name_col) + missing_cols <- setdiff(required_cols, colnames(map_tbl)) + if (length(missing_cols) > 0) { + cli::cli_abort( + "Missing required column(s) in {.arg names}: [{missing_cols}]." + ) + } + map_tbl <- map_tbl[, required_cols, drop = FALSE] + } else { + cli::cli_abort( + "{.arg names} must be either a named character vector or a data frame with columns [{strata_col}, {strata_name_col}]." + ) + } + + map_tbl[[strata_col]] <- suppressWarnings(as.integer(as.character(map_tbl[[strata_col]]))) + if (any(is.na(map_tbl[[strata_col]]))) { + cli::cli_abort("All strata IDs in {.arg names} must be coercible to integer.") + } + + map_tbl[[strata_name_col]] <- trimws(as.character(map_tbl[[strata_name_col]])) + if (any(is.na(map_tbl[[strata_name_col]]) | map_tbl[[strata_name_col]] == "")) { + cli::cli_abort("All strata labels in {.arg names} must be non-empty strings.") + } + + if (anyDuplicated(map_tbl[[strata_col]]) > 0) { + cli::cli_abort("Found duplicated strata IDs in {.arg names}.") + } + if (anyDuplicated(map_tbl[[strata_name_col]]) > 0) { + cli::cli_abort("Found duplicated strata labels in {.arg names}.") + } + + strata_ids <- unique(rep_tbl[[strata_col]]) + unknown_ids <- setdiff(unique(map_tbl[[strata_col]]), strata_ids) + if (length(unknown_ids) > 0) { + cli::cli_abort( + "Unknown strata ID(s) in {.arg names}: [{unknown_ids}]." + ) + } + + idx <- match(rep_tbl[[strata_col]], map_tbl[[strata_col]]) + has_map <- !is.na(idx) + rep_tbl[[strata_name_col]][has_map] <- map_tbl[[strata_name_col]][idx[has_map]] + + unmapped_ids <- unique(rep_tbl[[strata_col]][!has_map]) + if (length(unmapped_ids) > 0) { + if (identical(unnamed, "error")) { + cli::cli_abort( + "Missing names for strata ID(s): [{unmapped_ids}]. Provide a complete mapping or use {.code unnamed = 'auto'} / {.code unnamed = 'keep'}." + ) + } else if (identical(unnamed, "auto")) { + rep_tbl[[strata_name_col]][!has_map] <- paste0(auto_prefix, rep_tbl[[strata_col]][!has_map]) + } + } + + uniq_labels <- unique(rep_tbl[c(strata_col, strata_name_col)]) + if (any(is.na(uniq_labels[[strata_name_col]]) | trimws(uniq_labels[[strata_name_col]]) == "")) { + cli::cli_abort("Resulting strata labels contain missing or empty values.") + } + if (anyDuplicated(uniq_labels[[strata_name_col]]) > 0) { + cli::cli_abort("Resulting strata labels are not unique across strata.") + } + + strata_table <- rep_tbl |> + select(all_of(c(strata_col, strata_name_col, idata$schema_strata))) |> + distinct() |> + duckplyr::as_duckdb_tibble() + + ImmunData$new( + schema = idata$schema_receptor, + annotations = idata$annotations, + repertoires = rep_tbl, + strata = strata_table, + provenance = get_provenance(idata) + ) +} diff --git a/R/operations_annotate.R b/R/operations_annotate.R index 23444ba..e30dd5e 100644 --- a/R/operations_annotate.R +++ b/R/operations_annotate.R @@ -1,91 +1,225 @@ -#' @title Annotate ImmunData object -#' -#' @description Joins additional annotation data to the annotations slot of an `ImmunData` object. -#' -#' This function allows you to add extra information to your repertoire data by joining a -#' dataframe of annotations based on specified columns. It supports joining by -#' one or more columns. -#' -#' @param idata An `ImmunData` R6 object containing repertoire and annotation data. -#' @param annotations A data frame containing the annotations to be joined. -#' @param by A named character vector specifying the columns to join by. The names of the -#' vector should be the column names in `idata$annotations` and the values should be -#' the corresponding column names in the `annotations` data frame. -#' @param annot_col A character vector specifying the column with receptor, barcode or chain identifiers -#' to annotate a corresponding receptors, barode or chains in `idata`. -#' @param keep_repertoires Logical. If `TRUE` (default) and the `ImmunData` object -#' contains repertoire data (`idata$schema_repertoire` is not NULL), the repertoires -#' will be re-aggregated after joining the annotations. Set to `FALSE` if you do not -#' want to re-aggregate repertoires immediately. -#' @param remove_limit Logical. If `FALSE` (default), a warning will be issued if the -#' `annotations` data frame has 100 or more columns, suggesting potential performance -#' issues. Set to `TRUE` to disable this warning and allow joining of annotations -#' with an arbitrary number of columns. Use with caution, as joining wide dataframes -#' can be memory-intensive and slow. -#' -#' @return A new `ImmunData` object with the annotations joined to the `annotations` slot. -#' -#' @details The function performs a left join operation, keeping all rows from -#' `idata$annotations` and adding matching columns from the `annotations` data frame. -#' If there are multiple matches in `annotations` for a row in `idata$annotations`, -#' all combinations will be returned, potentially increasing the number of rows -#' in the resulting annotations table. -#' -#' The function uses `checkmate` to validate the input types and structure. -#' -#' A check is performed to ensure that the columns specified in `by` exist in both -#' `idata$annotations` and the `annotations` data frame. -#' -#' The `annotations` data frame is converted to a duckdb tibble internally for -#' efficient joining, especially with large datasets. -#' -#' @section Warning: -#' By default (`remove_limit = FALSE`), joining an `annotations` data frame with 100 or -#' more columns will trigger a warning. This is a safeguard to prevent accidental -#' joining of very wide data (e.g., gene expression data) that could lead to -#' performance degradation or crashes. If you understand the risks and intend to join -#' a wide data frame, set `remove_limit = TRUE`. +#' @title Add external information to ImmunData +#' +#' @description +#' Use the `annotate_*()` functions to add information stored in another data +#' frame to an [ImmunData] object. For example, you can add cell types from a +#' single-cell analysis, antigen labels for receptors, or clinical information +#' for samples. +#' +#' When matching identifiers are unique, the functions keep every row in +#' `idata`. When a row has no match in `annotations`, the new columns contain +#' `NA`. Each function returns a new [ImmunData] object. The original object is +#' not changed. +#' +#' @details +#' The functions differ in how they select the columns used for matching. The +#' rules for duplicate identifiers, column conflicts, and preserved summaries +#' are the same for all functions. +#' +#' @section Choose a function: +#' +#' Use the function that matches the type of information you want to add: +#' +#' * [annotate_barcodes()] matches cell or barcode identifiers. +#' * [annotate_receptors()] matches receptor identifiers. All rows belonging to +#' a matched receptor receive the new information. +#' * [annotate_chains()] matches chain identifiers. +#' * `annotate()` matches any one or more columns that you specify in `by`. +#' +#' The first three functions select the correct `ImmunData` identifier for you. +#' `annotate_immundata()` is an alternative name for `annotate()`. +#' +#' @section How matching works: +#' +#' For `annotate_barcodes()`, `annotate_receptors()`, and `annotate_chains()`, +#' `annot_col` names the identifier column in `annotations`. For example, +#' `annot_col = "barcode"` matches the `barcode` column in `annotations` with +#' the barcode identifier in `idata`. +#' +#' For general matching, supply `by` in the form +#' `c(immundata_column = "annotation_column")`. For example, +#' `by = c(Response = "response_code")` matches the `Response` column in +#' `idata` with the `response_code` column in `annotations`. To match columns +#' with the same name, use a value such as `by = c(Response = "Response")`. +#' You can include more than one pair of columns in `by`. +#' +#' Columns from `annotations` that are not used for matching are added to the +#' result. Rows in `idata` without a match receive `NA`. Rows in `annotations` +#' without a match are ignored. +#' +#' @section Annotation identifiers must be unique: +#' +#' `annotations` must contain at most one row for each identifier, or each +#' combination of identifiers when matching several columns. For example, a +#' barcode annotation table must contain at most one row per barcode. +#' +#' The function does not check this rule because the annotation table may be +#' very large. If an identifier occurs several times, the corresponding rows in +#' `idata` are repeated. This can make receptor counts, proportions, and other +#' summaries incorrect. +#' +#' @section Existing annotation columns: +#' +#' By default, the function stops if a new annotation column has the same name +#' as a column already present in `idata`. This prevents accidental replacement. +#' +#' Use `conflicts = "replace"` to replace existing annotation columns. Columns +#' that define receptors, repertoires, strata, or other `ImmunData` state are +#' protected and cannot be replaced. The old column is removed before matching, +#' so rows without a new match receive `NA`. +#' +#' @section Repertoire and strata summaries: +#' +#' With the default `keep_repertoires = TRUE`, existing repertoire and strata +#' summaries are copied to the new object without recalculation. Use this option +#' when you are only adding information and the matching identifiers in +#' `annotations` are unique. +#' +#' Set `keep_repertoires = FALSE` when you plan to filter rows or define new +#' repertoires using the added information. This removes existing repertoire and +#' strata summaries and their derived columns. After annotation and filtering, +#' use [agg_repertoires()] to define the new repertoires. +#' +#' @section Very wide annotation tables: +#' +#' By default, the function stops when `annotations` contains 100 or more +#' columns. Adding a very wide table, such as a complete gene-expression matrix, +#' can be slow and require a large amount of memory. If you understand this cost, +#' set `remove_limit = TRUE` to allow the operation. +#' +#' @param idata An [ImmunData] object. +#' @param annotations A data frame containing the information to add. It must +#' contain the columns used for matching and at most one row for each matching +#' identifier or combination of identifiers. +#' @param by A named character vector describing how columns are matched. Names +#' are columns in `idata`; values are the corresponding columns in +#' `annotations`. For example, `c(Response = "response_code")`. +#' @param annot_col Name of the identifier column in `annotations`. For +#' `annotate_receptors()` and `annotate_chains()`, the default is the standard +#' `ImmunData` receptor or chain identifier. For `annotate_barcodes()`, the +#' default `""` uses the row names of `annotations`. Supplying an +#' explicit barcode column is usually clearer. +#' @param keep_repertoires Whether to preserve existing repertoire and strata +#' summaries without recalculation. The default is `TRUE`. If `FALSE`, these +#' summaries and their derived annotation columns are removed. +#' @param remove_limit Whether to allow an annotation table with 100 or more +#' columns. The default is `FALSE`, which stops the operation for such tables. +#' Set to `TRUE` only when the wide join is intentional. +#' @param conflicts How to handle new annotation columns whose names already +#' exist in `idata`. `"error"`, the default, stops the operation. `"replace"` +#' replaces existing columns that are not protected by `ImmunData`. +#' +#' @return A new [ImmunData] object containing the added annotation columns. +#' Existing repertoire and strata summaries are preserved when +#' `keep_repertoires = TRUE`. +#' +#' @seealso [dplyr::left_join()], [agg_repertoires()], [filter_immundata()], +#' [mutate_immundata()], [ImmunData] #' -#' @concept Annotation #' @examples -#' \dontrun{ -#' # Assuming 'my_immun_data' is an ImmunData object and 'sample_info' is a data frame -#' # with a column 'sample_id' matching 'sample' in my_immun_data$annotations -#' # and additional columns like 'treatment' and 'disease_status'. -#' -#' sample_info <- data.frame( -#' sample_id = c("sample1", "sample2", "sample3", "sample4"), -#' treatment = c("Treatment A", "Treatment B", "Treatment A", "Treatment C"), -#' disease_status = c("Healthy", "Disease", "Healthy", "Disease"), -#' stringsAsFactors = FALSE # Important to keep characters as characters +#' library(immundata) +#' library(dplyr) +#' +#' options(immundata.verbose = FALSE) +#' +#' # Load data included with immundata +#' idata <- get_test_idata() +#' +#' # Add cell types by matching barcode identifiers +#' cell_labels <- tibble( +#' barcode = c("S1_1", "S1_2"), +#' cell_type = c("CD8 T cell", "CD4 T cell") #' ) #' -#' # Join sample information using the 'sample' column -#' my_immun_data_annotated <- annotate( -#' idata = my_immun_data, -#' annotations = sample_info, -#' by = c("sample" = "sample_id") +#' idata_with_cells <- idata |> +#' annotate_barcodes( +#' annotations = cell_labels, +#' annot_col = "barcode" +#' ) +#' +#' idata_with_cells |> +#' collect() |> +#' filter(imd_barcode %in% c("S1_1", "S1_2", "S1_3")) |> +#' select(imd_barcode, cell_type) |> +#' arrange(imd_barcode) +#' # Expected result: +#' # imd_barcode cell_type +#' # S1_1 CD8 T cell +#' # S1_2 CD4 T cell +#' # S1_3 NA +#' +#' # Add antigen labels to selected receptors +#' receptor_labels <- tibble( +#' receptor_id = c(738L, 1567L), +#' antigen = c("CMV", "CMV") #' ) #' -#' # New sample_info +#' idata_with_antigens <- idata |> +#' annotate_receptors( +#' annotations = receptor_labels, +#' annot_col = "receptor_id" +#' ) +#' +#' idata_with_antigens |> +#' collect() |> +#' filter(!is.na(antigen)) |> +#' distinct(imd_receptor_id, cdr3_aa, antigen) |> +#' arrange(imd_receptor_id) +#' # Expected result: +#' # imd_receptor_id cdr3_aa antigen +#' # 738 ASRAGAGTGELF CMV +#' # 1567 ASFPVLSPYNEQF CMV #' -#' # Join data by multiple columns, e.g., 'sample' and 'barcode' -#' # Assuming 'cell_annotations' is a data frame with 'sample_barcode' and 'cell_type' -#' my_immun_data_cell_annotated <- annotate( -#' idata = my_immun_data, -#' annotations = cell_annotations, -#' by = c("sample" = "sample", "barcode" = "sample_barcode") +#' # Match columns with different names +#' response_info <- tibble( +#' response_code = c("FR", "PR"), +#' response_label = c("Full response", "Partial response") #' ) #' -#' # Join a wide dataframe, suppressing the column limit warning -#' # Assuming 'gene_expression' is a data frame with 'barcode' and many gene columns -#' my_immun_data_gene_expression <- annotate( -#' idata = my_immun_data, -#' annotations = gene_expression, -#' by = c("barcode" = "barcode"), -#' remove_limit = TRUE +#' idata_with_response <- idata |> +#' annotate( +#' annotations = response_info, +#' by = c(Response = "response_code") +#' ) +#' +#' idata_with_response |> +#' collect() |> +#' distinct(Response, response_label) |> +#' arrange(Response) +#' # Expected result: +#' # Response response_label +#' # FR Full response +#' # PR Partial response +#' +#' # Replace an annotation column intentionally +#' revised_cell_labels <- tibble( +#' barcode = c("S1_1", "S1_2"), +#' cell_type = c("Cytotoxic T cell", "Helper T cell") #' ) -#' } +#' +#' idata_with_revised_cells <- idata_with_cells |> +#' annotate_barcodes( +#' annotations = revised_cell_labels, +#' annot_col = "barcode", +#' conflicts = "replace" +#' ) +#' +#' # Remove old repertoire summaries before defining repertoires by cell type +#' cell_repertoires <- idata |> +#' annotate_barcodes( +#' annotations = cell_labels, +#' annot_col = "barcode", +#' keep_repertoires = FALSE +#' ) |> +#' filter(!is.na(cell_type)) |> +#' agg_repertoires(schema = "cell_type") +#' +#' cell_repertoires$repertoires |> +#' arrange(cell_type) +#' # Expected result: +#' # imd_repertoire_id cell_type n_barcodes n_receptors +#' # 1 CD4 T cell 1 1 +#' # 2 CD8 T cell 1 1 #' #' @concept annotation #' @rdname annotate_immundata @@ -94,11 +228,13 @@ annotate_immundata <- function(idata, annotations, by, keep_repertoires = TRUE, - remove_limit = FALSE) { + remove_limit = FALSE, + conflicts = c("error", "replace")) { checkmate::assert_r6(idata, "ImmunData") checkmate::assert_data_frame(annotations) checkmate::assert_character(by, min.len = 1, names = "named") checkmate::assert_logical(keep_repertoires) + conflicts <- match.arg(conflicts) if (!remove_limit && ncol(annotations) >= 100) { rlang::abort(cli::format_inline(paste0( @@ -118,32 +254,45 @@ annotate_immundata <- function(idata, if (length(setdiff(by, colnames(ann_tbl)))) { cli_abort("Column(s) '{setdiff(by, colnames(ann_tbl))}' not found in annotations.") } - if (any(names(by) %in% colnames(ann_tbl))) { - same_col_names <- intersect(names(by), colnames(ann_tbl)) - if (!all(by[names(by)] == names(by))) { - # We don't care about the very same column names, we are try to mitigate risk when there is a collision after (!) the renaming - cli_abort("Column(s) '{names(by)[names(by) %in% colnames(ann_tbl)]}', reserved for joining with ImmunData, are found in the annotations. Can't rename the table, please make sure the names are unique and not already presented in annotations.") - } - } if (!all(names(by) %in% colnames(idata$annotations))) { cli_abort("Column(s) '{names(by)[! names(by) %in% colnames(idata$annotations)]}' are not found in ImmunData. Please double-check the column names: {.code colnames(idata$annotations)}.") } + annotation_value_cols <- setdiff(colnames(ann_tbl), unname(by)) + collisions <- intersect(annotation_value_cols, colnames(idata$annotations)) + if (length(collisions) > 0 && conflicts == "error") { + cli_abort( + "Annotation column(s) collide with existing ImmunData annotation columns: {.field {collisions}}. Please rename them before calling {.fn annotate_immundata}." + ) + } + if (length(collisions) > 0 && conflicts == "replace") { + # `imd_group_id` is an annotation-level grouping label and is intentionally + # replaceable. Other protected system and schema columns define receptor, + # repertoire, or strata state and cannot be replaced safely. + protected_collisions <- setdiff(collisions, imd_schema("group")) + assert_mutable_annotation_columns(idata, protected_collisions) + } + ann_tbl <- ann_tbl |> rename(all_of(by)) - new_annotations <- idata$annotations |> - left_join(ann_tbl, by = names(by)) + existing_annotations <- idata$annotations + if (conflicts == "replace") { + existing_annotations <- existing_annotations |> + select(-all_of(collisions)) + } - new_idata <- ImmunData$new( - schema = idata$schema_receptor, - annotations = new_annotations - ) + new_annotations <- existing_annotations |> + left_join(ann_tbl, by = names(by)) - if (keep_repertoires && !is.null(idata$schema_repertoire)) { - new_idata |> agg_repertoires(idata$schema_repertoire) + if (keep_repertoires) { + clone_with_annotations(idata, new_annotations) } else { - new_idata + ImmunData$new( + schema = idata$schema_receptor, + annotations = drop_repertoire_state(new_annotations), + provenance = get_provenance(idata) + ) } } @@ -159,7 +308,8 @@ annotate_receptors <- function(idata, annotations, annot_col = imd_schema("receptor"), keep_repertoires = TRUE, - remove_limit = FALSE) { + remove_limit = FALSE, + conflicts = c("error", "replace")) { if (annot_col == "") { annotations[["imd_row_names"]] <- rownames(annotations) annot_col <- "imd_row_names" @@ -171,7 +321,8 @@ annotate_receptors <- function(idata, annotations = annotations, by = match_col, keep_repertoires = keep_repertoires, - remove_limit = remove_limit + remove_limit = remove_limit, + conflicts = conflicts ) } @@ -182,7 +333,8 @@ annotate_barcodes <- function(idata, annotations, annot_col = "", keep_repertoires = TRUE, - remove_limit = FALSE) { + remove_limit = FALSE, + conflicts = c("error", "replace")) { if (annot_col == "") { annotations[["imd_row_names"]] <- rownames(annotations) annot_col <- "imd_row_names" @@ -194,7 +346,8 @@ annotate_barcodes <- function(idata, annotations = annotations, by = match_col, keep_repertoires = keep_repertoires, - remove_limit = remove_limit + remove_limit = remove_limit, + conflicts = conflicts ) } @@ -205,7 +358,8 @@ annotate_chains <- function(idata, annotations, annot_col = imd_schema("chain"), keep_repertoires = TRUE, - remove_limit = FALSE) { + remove_limit = FALSE, + conflicts = c("error", "replace")) { if (annot_col == "") { annotations[["imd_row_names"]] <- rownames(annotations) annot_col <- "imd_row_names" @@ -217,6 +371,7 @@ annotate_chains <- function(idata, annotations = annotations, by = match_col, keep_repertoires = keep_repertoires, - remove_limit = remove_limit + remove_limit = remove_limit, + conflicts = conflicts ) } diff --git a/R/operations_compute_collect.R b/R/operations_compute_collect.R new file mode 100644 index 0000000..3e92a4c --- /dev/null +++ b/R/operations_compute_collect.R @@ -0,0 +1,53 @@ +#' @title Compute ImmunData annotations +#' +#' @description +#' Materializes the annotation table of an `ImmunData` object via +#' [dplyr::compute()] and returns a new `ImmunData`. +#' +#' @param x ImmunData object. +#' @param ... Additional arguments passed to [dplyr::compute()] for +#' `x$annotations`. +#' +#' @return A new `ImmunData` object with computed annotations and the input +#' repertoire, strata, schema, and provenance state preserved. +#' +#' @concept operations +#' @exportS3Method dplyr::compute +compute.ImmunData <- function(x, ...) { + checkmate::assert_r6(x, "ImmunData") + + new_annotations <- x$annotations |> + compute(...) + + clone_with_annotations(x, new_annotations) +} + +#' @title Collect ImmunData annotations +#' +#' @description +#' Collects annotations from an `ImmunData` object and returns them as a tibble. +#' Factor columns are converted to character. +#' +#' @param x ImmunData object. +#' @param ... Additional arguments passed to [dplyr::collect()] for +#' `x$annotations`. +#' +#' @return A tibble with collected annotations. +#' +#' @concept operations +#' @exportS3Method dplyr::collect +collect.ImmunData <- function(x, ...) { + checkmate::assert_r6(x, "ImmunData") + + annotations <- x$annotations |> + collect(...) + + if (is.data.frame(annotations)) { + factor_cols <- vapply(annotations, is.factor, logical(1)) + if (any(factor_cols)) { + annotations[factor_cols] <- lapply(annotations[factor_cols], as.character) + } + } + + as_tibble(annotations) +} diff --git a/R/operations_count.R b/R/operations_count.R index 2ecd466..70f3821 100644 --- a/R/operations_count.R +++ b/R/operations_count.R @@ -1,16 +1,63 @@ -#' @title Count the number of chains in ImmunData +#' @title Count chain rows in ImmunData #' -#' @param x ImmunData object. -#' @param ... Not used. -#' @param wt Not used. -#' @param sort Not used. -#' @param name Not used. +#' @description +#' Use `count()` to find how many chain rows are stored in an [ImmunData] +#' object. #' -#' @concept operations +#' Use this method for a quick check of dataset size. The unit counted is one +#' retained chain row. Each retained cell with a paired receptor usually +#' contributes two rows, one for each chain. The same receptor can therefore +#' contribute two rows for every cell carrying it. For bulk data with an +#' abundance column, this method counts table rows rather than the summed +#' sequence abundance. +#' +#' The function returns a one-row duckplyr table. The original object is not +#' changed. +#' +#' @details +#' This method currently provides only the total row count. The grouping, +#' weighting, sorting, and result-name arguments of [dplyr::count()] are +#' accepted for method compatibility but are not applied. +#' +#' The calculation runs on the duckplyr annotation table and can remain in +#' DuckDB. Use [dplyr::pull()] or [dplyr::collect()] to bring the small result +#' into R. +#' +#' @param x An [ImmunData] object. +#' @param ... Additional arguments. Accepted for compatibility with +#' [dplyr::count()], but currently ignored. +#' @param wt Any value or `NULL`. Accepted for compatibility with +#' [dplyr::count()], but currently ignored. +#' @param sort A logical value. Accepted for compatibility with +#' [dplyr::count()], but currently ignored. +#' @param name A character string or `NULL`. Accepted for compatibility with +#' [dplyr::count()], but currently ignored. The result column is always named +#' `n`. +#' +#' @return A one-row duckplyr table with an integer column named `n`. This value +#' is the number of rows in the chain-level annotation table. #' +#' @seealso [dplyr::count()], [dplyr::collect()], [ImmunData] +#' +#' @examples +#' library(immundata) +#' library(dplyr) +#' +#' options(immundata.verbose = FALSE) +#' idata <- get_test_idata() +#' +#' idata |> count() +#' # Expected result: +#' # n +#' # 1902 +#' +#' # The result means that the object contains 1,902 retained chain rows. +#' # It does not mean that it contains 1,902 unique receptors. +#' +#' @concept operations #' @exportS3Method dplyr::count count.ImmunData <- function(x, ..., wt = NULL, sort = FALSE, name = NULL) { - checkmate::check_r6(x, "ImmunData") + checkmate::assert_r6(x, "ImmunData") x$annotations |> count() } diff --git a/R/operations_dimnames.R b/R/operations_dimnames.R new file mode 100644 index 0000000..285c147 --- /dev/null +++ b/R/operations_dimnames.R @@ -0,0 +1,53 @@ +#' @title Get Annotation Dimnames from ImmunData +#' +#' @description +#' Returns dimension names for an `ImmunData` object so that +#' `colnames(idata)` maps to annotation column names. +#' +#' @param x ImmunData object. +#' +#' @return A list with `NULL` row names and annotation column names. +#' +#' @concept operations +#' @export +dimnames.ImmunData <- function(x) { + checkmate::assert_r6(x, "ImmunData") + + list(NULL, colnames(x$annotations)) +} + +#' @title Prevent Renaming ImmunData via dimnames +#' +#' @description +#' Disallows replacing dimension names on `ImmunData`. +#' +#' @param x ImmunData object. +#' @param value Not used. +#' +#' @concept operations +#' @exportS3Method base::`dimnames<-` ImmunData +`dimnames<-.ImmunData` <- function(x, value) { + checkmate::assert_r6(x, "ImmunData") + + cli::cli_abort( + "Renaming columns via {.code colnames<-} or {.code dimnames<-} is not allowed for {.code ImmunData}." + ) +} + +#' @title Prevent Renaming ImmunData via names +#' +#' @description +#' Disallows replacing names on `ImmunData`. +#' +#' @param x ImmunData object. +#' @param value Not used. +#' +#' @concept operations +#' @exportS3Method base::`names<-` ImmunData +`names<-.ImmunData` <- function(x, value) { + checkmate::assert_r6(x, "ImmunData") + + cli::cli_abort( + "Renaming via {.code names<-} is not allowed for {.code ImmunData}." + ) +} diff --git a/R/operations_downsample.R b/R/operations_downsample.R new file mode 100644 index 0000000..664da2e --- /dev/null +++ b/R/operations_downsample.R @@ -0,0 +1,347 @@ +#' @title Reduce repertoires to a common sampling depth +#' +#' @description +#' Use `downsample_immundata()` to reduce every repertoire to the same number or +#' fraction of observed cells or bulk sequence counts before comparing +#' repertoires. So, it is just a downsampling. +#' +#' Use this function when different sequencing depths could affect a comparison +#' of repertoire diversity or composition. In single-cell data, the sampling +#' unit is a cell barcode and all selected chains from that cell stay together. +#' In bulk data with abundance values, the sampling unit is one sequence count. +#' +#' The function returns a new [ImmunData] object. The original object is not +#' changed. +#' +#' @section Meaning of `n` for single-cell data: +#' +#' * `0 < n < 1` keeps `floor(n * number of cells)` cells from each repertoire. +#' * `n >= 1` keeps `n` cells from each repertoire. +#' +#' Cell barcodes are sampled without replacement. For paired receptors, all +#' retained chains belonging to a selected cell stay together. +#' +#' @section Meaning of `n` for bulk data: +#' +#' * `0 < n < 1` keeps `floor(n * total abundance)` sequence counts from each +#' repertoire. +#' * `n >= 1` keeps a total abundance of `n` from each repertoire. +#' +#' Counts are sampled without replacement according to their observed +#' abundance. A retained receptor can therefore have a smaller abundance than +#' it had before downsampling. For example, `n = 1000` makes the total retained +#' abundance equal to 1000 in every repertoire that originally contained at +#' least 1000 counts. +#' +#' If a requested whole-number `n` is larger than a repertoire, that repertoire +#' is returned unchanged and the function gives a warning. If a fraction is so +#' small that it selects zero units in any repertoire, the function stops and +#' asks for a larger value. +#' +#' @section Repertoire and strata summaries: +#' +#' When repertoires are defined, the function recalculates receptor counts, +#' proportions, repertoire sizes, and the number of repertoires containing each +#' receptor. Existing strata are also rebuilt, and their labels are retained. +#' When repertoires are not defined, the complete dataset is treated as one +#' sampling group and no repertoire summary is added. +#' +#' @section Backend and storage: +#' +#' Chain-level selection and reconstruction use the duckplyr annotation table. +#' The small table of sampling units is collected into R for random sampling. +#' The function does not overwrite the stored input object. Use +#' [write_immundata()] to save the returned object. +#' +#' @param idata An [ImmunData] object. For comparisons between repertoires, its +#' repertoires should already be defined with [read_repertoires()] or +#' [agg_repertoires()]. +#' @param n A number. Sampling depth. Use a value strictly between 0 and 1 for a +#' fraction, or a whole number greater than or equal to 1 for an absolute +#' number of cells or bulk sequence counts. +#' @param seed A non-negative integer or `NULL`. Used to reproduce the same +#' random sample. The default is `NULL`. +#' +#' @return A new [ImmunData] object containing the sampled chain observations. +#' If the input has repertoires or strata, their summaries are recalculated +#' for the sampled data. +#' +#' @seealso [agg_repertoires()], [filter_immundata()], [write_immundata()] +#' +#' @examples +#' library(immundata) +#' library(dplyr) +#' +#' options(immundata.verbose = FALSE) +#' +#' # Create two small bulk T-cell repertoires with different total abundances. +#' bulk_data <- tibble( +#' Sample = c("Tumor", "Tumor", "Blood", "Blood"), +#' cdr3_aa = c("CASSA", "CASSB", "CASSA", "CASSC"), +#' v_call = c("TRBV1", "TRBV2", "TRBV1", "TRBV3"), +#' abundance = c(20L, 5L, 4L, 6L) +#' ) +#' bulk_file <- tempfile(fileext = ".tsv") +#' readr::write_tsv(bulk_data, bulk_file) +#' +#' idata <- read_repertoires( +#' path = bulk_file, +#' schema = c("cdr3_aa", "v_call"), +#' count_col = "abundance", +#' repertoire_schema = "Sample", +#' preprocess = NULL, +#' postprocess = NULL, +#' rename_columns = NULL, +#' output_folder = tempfile("immundata-downsample-") +#' ) +#' +#' before <- idata$repertoires |> +#' select(Sample, n_barcodes) |> +#' rename(before = n_barcodes) +#' +#' sampled <- downsample_immundata(idata, n = 5, seed = 42) +#' +#' before |> +#' left_join( +#' sampled$repertoires |> +#' select(Sample, n_barcodes) |> +#' rename(after = n_barcodes), +#' by = "Sample" +#' ) |> +#' arrange(Sample) +#' # Expected result: +#' # Sample before after +#' # Blood 10 5 +#' # Tumor 25 5 +#' +#' # Each returned repertoire has five sequence counts. `idata` still has its +#' # original repertoire sizes of 10 and 25. +#' +#' @concept filtering +#' @export +downsample_immundata <- function(idata, n, seed = NULL) { + checkmate::assert_r6(idata, "ImmunData") + checkmate::assert_number(n, lower = 0, finite = TRUE) + checkmate::assert_integerish(seed, len = 1, null.ok = TRUE, lower = 0) + + if (n > 1 && abs(n - round(n)) > sqrt(.Machine$double.eps)) { + cli::cli_abort("When `n > 1`, `n` must be an integer count.") + } + + if (!is.null(seed)) { + set.seed(seed) + } + + receptor_col <- imd_schema("receptor") + barcode_col <- imd_schema("barcode") + chain_count_col <- imd_schema("chain_count") + count_col <- imd_schema("count") + repertoire_col <- imd_schema("repertoire") + prop_col <- imd_schema("proportion") + n_repertoires_col <- imd_schema("n_repertoires") + n_receptors_col <- imd_schema("n_receptors") + n_barcodes_col <- imd_schema("n_barcodes") + + annotations_base <- idata$annotations |> + select(-any_of(c(count_col, prop_col, n_repertoires_col, n_receptors_col, n_barcodes_col))) + + has_repertoire <- !is.null(idata$schema_repertoire) + if (has_repertoire && !(repertoire_col %in% colnames(annotations_base))) { + cli::cli_abort( + "Repertoire aggregation is incomplete: {.field {repertoire_col}} is missing from {.field idata$annotations}." + ) + } + unit_cols <- c(if (has_repertoire) repertoire_col else character(0), barcode_col) + + is_count_mode <- annotations_base |> + summarise( + any_non_unit = any(!!rlang::sym(chain_count_col) != 1) + ) |> + collect() |> + pull("any_non_unit") + + unit_base <- annotations_base |> + select(all_of(unique(c(unit_cols, receptor_col, chain_count_col)))) |> + summarise( + .by = all_of(c(unit_cols, receptor_col)), + !!chain_count_col := dplyr::first(!!rlang::sym(chain_count_col)) + ) + + unit_table <- if (is_count_mode) { + unit_base |> + summarise( + .by = all_of(unit_cols), + !!count_col := sum(!!rlang::sym(chain_count_col)) + ) |> + collect() + } else { + unit_base |> + summarise( + .by = all_of(unit_cols), + !!count_col := 1L + ) |> + collect() + } + + if (nrow(unit_table) == 0) { + cli::cli_abort("No barcode units available for downsampling.") + } + + if (!is.null(seed)) { + # Keep deterministic sampling order across lazy backend materialization. + unit_table <- unit_table[ + do.call(order, unit_table[unit_cols]), , + drop = FALSE + ] + } + + group_ids <- if (has_repertoire) { + as.character(unit_table[[repertoire_col]]) + } else { + rep("__all__", nrow(unit_table)) + } + + split_groups <- split(unit_table, group_ids) + sampled_units_list <- vector("list", length(split_groups)) + n_clipped <- 0L + + for (i in seq_along(split_groups)) { + group_df <- split_groups[[i]] + total_count <- sum(group_df[[count_col]]) + target_raw <- if (n < 1) floor(total_count * n) else as.integer(round(n)) + target <- min(as.integer(target_raw), as.integer(total_count)) + + if (target_raw > total_count) { + n_clipped <- n_clipped + 1L + } + + if (target < 1) { + cli::cli_abort("No barcode units were selected. Increase `n`.") + } + + if (!is_count_mode) { + sampled_units_list[[i]] <- group_df[sample.int(nrow(group_df), size = target, replace = FALSE), , drop = FALSE] + next + } + + if (target >= total_count) { + sampled_units_list[[i]] <- group_df + next + } + + sampled_counts <- draw_weighted_counts(group_df[[count_col]], target) + out <- group_df[sampled_counts > 0, , drop = FALSE] + out[[count_col]] <- sampled_counts[sampled_counts > 0] + sampled_units_list[[i]] <- out + } + + if (n_clipped > 0L && n > 1) { + cli::cli_warn("Requested `n` exceeds available units in {n_clipped} repertoire(s). Those repertoires were returned unchanged.") + } + + sampled_units <- dplyr::bind_rows(sampled_units_list) + + if (nrow(sampled_units) == 0) { + cli::cli_abort("No barcode units were selected. Increase `n`.") + } + + if (!is_count_mode) { + sampled_keys <- duckdb_tibble(sampled_units |> + select(all_of(unit_cols))) + new_annotations <- annotations_base |> + semi_join(sampled_keys, by = unit_cols) + } else { + n_duplicate_units <- annotations_base |> + summarise( + .by = all_of(unit_cols), + n_rows = n() + ) |> + filter(.data$n_rows > 1) |> + summarise(n_dups = n()) |> + collect() |> + pull("n_dups") + + if (length(n_duplicate_units) == 0) { + n_duplicate_units <- 0 + } + + if (n_duplicate_units > 0) { + cli::cli_warn("Detected duplicated unit rows in count mode ({n_duplicate_units}). Collapsing to one row per unit before downsampling join.") + } + + sampled_units_chain <- sampled_units |> + select(all_of(c(unit_cols, count_col))) + colnames(sampled_units_chain)[colnames(sampled_units_chain) == count_col] <- chain_count_col + + sampled_tbl <- duckdb_tibble(sampled_units_chain) + unit_annotation_cols <- setdiff(colnames(annotations_base), c(unit_cols, chain_count_col)) + + unit_annotations <- annotations_base |> + select(-all_of(chain_count_col)) |> + summarise( + .by = all_of(unit_cols), + dplyr::across(all_of(unit_annotation_cols), dplyr::first) + ) + + new_annotations <- unit_annotations |> + inner_join(sampled_tbl, by = unit_cols) + } + + new_idata <- ImmunData$new( + schema = idata$schema_receptor, + annotations = new_annotations, + provenance = get_provenance(idata) + ) + + if (is.null(idata$schema_repertoire)) { + return(new_idata) + } + + rebuild_repertoire_and_strata(new_idata, idata) +} + +draw_weighted_counts <- function(weights, size) { + weights <- as.integer(round(weights)) + out <- integer(length(weights)) + total <- sum(weights) + + if (size <= 0 || total <= 0) { + return(out) + } + + if (size >= total) { + return(weights) + } + + remaining_draws <- as.integer(size) + remaining_total <- as.integer(total) + + if (length(weights) == 1) { + out[1] <- remaining_draws + return(out) + } + + # Exact weighted sampling without replacement via sequential hypergeometric draws. + for (i in seq_len(length(weights) - 1)) { + wi <- as.integer(weights[i]) + + if (wi <= 0 || remaining_draws <= 0) { + out[i] <- 0L + } else { + out[i] <- as.integer( + stats::rhyper( + nn = 1, + m = wi, + n = remaining_total - wi, + k = remaining_draws + ) + ) + remaining_draws <- remaining_draws - out[i] + } + + remaining_total <- remaining_total - wi + } + + out[length(weights)] <- remaining_draws + out +} diff --git a/R/operations_external_annotate_anndatar.R b/R/operations_external_annotate_anndatar.R new file mode 100644 index 0000000..00b3913 --- /dev/null +++ b/R/operations_external_annotate_anndatar.R @@ -0,0 +1,60 @@ +#' @title Annotate an AnnData object from ImmunData (by barcode) +#' +#' @description +#' Copy selected columns from `idata$annotations` to `adata$obs`, matching by +#' cell barcode (`adata$obs_names`). +#' +#' @param idata An [immundata::ImmunData] object. +#' @param adata An [anndataR::AbstractAnnData] object. +#' @param cols Character vector with column names to transfer from +#' `idata$annotations`. +#' +#' @return The updated AnnData object. +#' +#' @export +annotate_anndata <- function(idata, + adata, + cols) { + checkmate::assert_r6(idata, "ImmunData") + checkmate::assert_true(inherits(adata, "AbstractAnnData")) + checkmate::assert_character(cols, min.len = 1, any.missing = FALSE) + + obs_names <- adata$obs_names + if (is.null(obs_names) || length(obs_names) == 0) { + cli::cli_abort("`adata$obs_names` is missing or empty. Expected cell barcodes in obs_names.") + } + if (any(is.na(obs_names)) || any(obs_names == "")) { + cli::cli_abort("`adata$obs_names` contains NA/empty values. Expected valid barcodes.") + } + if (anyDuplicated(obs_names) > 0) { + cli::cli_abort("`adata$obs_names` must be unique (duplicate barcodes found).") + } + + ann <- idata$annotations + bcsym <- immundata::imd_schema_sym("barcode") + bcname <- immundata::imd_schema("barcode") + + missing_cols <- setdiff(c(bcname, cols), colnames(ann)) + if (length(missing_cols) > 0) { + cli::cli_abort( + "Column(s) {cli::col_cyan(missing_cols)} not found in idata$annotations." + ) + } + + # TODO: I use distinct() here, any edge cases? + # Keep first record per barcode (same idea as annotate_seurat) + df <- ann |> + dplyr::select(barcode = !!bcsym, dplyr::all_of(cols)) |> + dplyr::distinct(.data$barcode, .keep_all = TRUE) |> + collect() + + obs <- adata$obs + idx <- match(obs_names, df$barcode) + + for (nm in cols) { + obs[[nm]] <- df[[nm]][idx] + } + + adata$obs <- obs + adata +} diff --git a/R/operations_filter.R b/R/operations_filter.R index 83e8522..09f08c3 100644 --- a/R/operations_filter.R +++ b/R/operations_filter.R @@ -1,144 +1,167 @@ -#' @title Filter ImmunData by receptor features, barcodes or any annotations +#' @title Keep selected rows or receptors in ImmunData #' #' @description -#' Provides flexible filtering options for an `ImmunData` object. +#' Use `filter()` to keep selected rows in an [ImmunData] object. For example, +#' you can keep rows from one response group, rows using a selected V gene, or +#' receptors containing a CDR3 sequence similar to a reference sequence. #' -#' `filter()` is the main function, allowing filtering based on receptor features -#' (e.g., CDR3 sequence) using various matching methods (exact, regex, fuzzy) and/or -#' standard `dplyr`-style filtering on annotation columns. +#' The function returns a new [ImmunData] object. The original object is not +#' changed. #' -#' `filter_barcodes()` is a convenience function to filter by specific cell barcodes. +#' This function is a direct implementation of [dplyr::filter]. Alternative +#' function name is `filter_immundata`. #' -#' `filter_receptors()` is a convenience function to filter by specific receptor identifiers. +#' Use `filter_barcodes()` to keep selected cell barcodes and +#' `filter_receptors()` to keep selected receptor identifiers. #' #' @details -#' For `filter`: -#' * User-provided `dplyr`-style filters (`...`) are applied *before* any sequence-based -#' filtering defined in `seq_options`. -#' * Sequence filtering compares values in the `query_col` of the annotations table -#' against the provided `patterns`. -#' * Supported sequence matching methods are: -#' * `"exact"`: Keeps rows where `query_col` exactly matches any of the `patterns`. -#' * `"regex"`: Keeps rows where `query_col` matches any of the regular expressions -#' in `patterns`. -#' * `"lev"` (Levenshtein distance): Keeps rows where the edit distance between -#' `query_col` and any pattern is less than or equal to `max_dist`. -#' * `"hamm"` (Hamming distance): Keeps rows where the Hamming distance (for -#' equal length strings) between `query_col` and any pattern is less than -#' or equal to `max_dist`. -#' * The filtering operations act on the `$annotations` table. A new `ImmunData` -#' object is created containing only the rows (and corresponding receptors) -#' that pass the filter(s). -#' * If `keep_repertoires = TRUE` (and repertoire data exists in the input), -#' the repertoire-level summaries (`$repertoires` table) are recalculated based -#' on the filtered annotations. Otherwise, the `$repertoires` table in the -#' output will be `NULL`. +#' You can filter an [ImmunData] object in three ways: #' -#' For `filter_barcodes` and `filter_receptors`: -#' * These functions provide a simpler interface for common filtering tasks based on -#' cell barcodes or receptor IDs, respectively. They use efficient `semi_join` -#' operations internally. +#' * Supply conditions in `...` to filter using annotation columns. Refer to +#' columns directly by name. For example, `Response == "FR"` keeps rows from +#' the `FR` response group. +#' * Supply `seq_options`, created with [make_seq_options()], to find receptors +#' containing a sequence that matches one or more reference sequences or +#' patterns. +#' * Use `filter_barcodes()` or `filter_receptors()` when you already have the +#' identifiers that you want to keep. #' -#' @param idata,.data An `ImmunData` object. -#' @param ... For `filter`, these are regular `dplyr`-style filtering -#' expressions (e.g., `V_gene == "IGHV1-1"`, `chain == "IGH"`) applied to the -#' `$annotations` table *before* sequence filtering. Ignored by `filter_barcodes` -#' and `filter_receptors`. -#' @param .by Not used. -#' @param .preserve Not used. -#' @param seq_options For `filter`, an optional named list specifying sequence-based -#' filtering options. Use [make_seq_options()] for convenient creation. -#' The list can contain: -#' * `query_col` (Character scalar): The name of the column in `$annotations` -#' containing sequences to compare (e.g., `"CDR3_aa"`, `"FR1_nt"`). -#' * `patterns` (Character vector): A vector of sequences or regular expressions -#' to match against `query_col`. -#' * `method` (Character scalar): The matching method. One of `"exact"`, -#' `"regex"`, `"lev"` (Levenshtein distance), or `"hamm"` (Hamming distance). -#' Defaults typically handled by `make_seq_options`. -#' * `max_dist` (Numeric scalar): For fuzzy methods (`"lev"`, `"hamm"`), the -#' maximum allowed distance. Rows with distance <= `max_dist` are kept. -#' Defaults typically handled by `make_seq_options`. -#' * `name_type` (Character scalar): Determines column names in intermediate distance -#' calculations if applicable (`"index"` or `"pattern"`). Passed through to -#' internal annotation functions. Defaults typically handled by `make_seq_options`. -#' If `seq_options` is `NULL` (the default), no sequence-based filtering is performed. -#' @param keep_repertoires Logical scalar. If `TRUE` (the default) and the input -#' `idata` has repertoire information (`idata$schema_repertoire` is not `NULL`), -#' the repertoire summaries will be recalculated based on the filtered data using -#' [agg_repertoires()]. If `FALSE`, or if no repertoire schema exists, the -#' returned `ImmunData` object will not contain repertoire summaries (`$repertoires` -#' will be `NULL`). -#' @param barcodes For `filter_barcodes`, a vector of cell identifiers (barcodes) -#' to keep. Can be character, integer, or numeric. -#' @param receptors For `filter_receptors`, a vector of receptor identifiers -#' to keep. Can be character, integer, or numeric. +#' Conditions in `...` are applied before sequence matching. Sequence matching +#' then identifies receptors from the remaining rows. When one chain matches, +#' all remaining chains belonging to the same receptor are kept. A chain removed +#' by a condition in `...` is not added back by sequence matching. #' -#' @return A new `ImmunData` object containing only the filtered annotations -#' (and potentially recalculated repertoire summaries). The schema remains the same. +#' Sequence matching methods are: #' -#' @seealso [make_seq_options()], [dplyr::filter()], [agg_repertoires()], [ImmunData] +#' * `"exact"`: the sequence must be identical to one of the references. +#' * `"regex"`: the sequence must match a regular-expression pattern. This is +#' an advanced option for matching text patterns. +#' * `"lev"`: the Levenshtein distance counts the substitutions, insertions, or +#' deletions needed to change one sequence into the other. +#' * `"hamm"`: the Hamming distance counts different positions between +#' sequences of the same length. Sequences of different lengths do not match. +#' +#' For `"lev"` and `"hamm"`, provide `max_dist`. A sequence is accepted when +#' its distance from at least one reference is less than or equal to this value. +#' A distance of `0` means an exact match, and smaller values mean more similar +#' sequences. +#' +#' By default, existing repertoire summaries are recalculated from the filtered +#' data. Existing strata are also rebuilt, and their labels are retained. Set +#' `keep_repertoires = FALSE` to return an object without repertoire or strata +#' summaries. +#' +#' @param idata,.data An [ImmunData] object. +#' @param ... One or more conditions used to keep rows. Refer to annotation +#' columns directly by name. Multiple conditions are combined with `&`. +#' Conditions are applied before sequence matching. +#' @param .by,.preserve Accepted for compatibility with [dplyr::filter()], but +#' currently not used for [ImmunData] objects. +#' @param seq_options Options for matching sequences with reference sequences or +#' patterns. Create these options with [make_seq_options()]. If `NULL`, the +#' default, no sequence matching is performed. +#' @param keep_repertoires If `TRUE`, the default, existing repertoire and strata +#' summaries are recalculated from the filtered data. If `FALSE`, the returned +#' object does not contain these summaries. +#' @param barcodes A character, integer, or numeric vector of cell barcodes to +#' keep with `filter_barcodes()`. +#' @param receptors A character, integer, or numeric vector of receptor +#' identifiers to keep with `filter_receptors()`. +#' +#' @return A new [ImmunData] object containing the selected rows and receptors. +#' If requested, repertoire and strata summaries are recalculated for the +#' selected data. +#' +#' @seealso [dplyr::filter()], [make_seq_options()], [mutate_immundata()], +#' [agg_repertoires()], [ImmunData] #' #' @examples -#' # Basic setup (assuming idata_test is a valid ImmunData object) -#' # print(idata_test) +#' library(immundata) +#' library(dplyr) +#' +#' options(immundata.verbose = FALSE) +#' +#' # Load data included with immundata +#' idata <- get_test_idata() +#' +#' # Keep rows from one response group +#' fr_response <- idata |> +#' filter(Response == "FR") +#' +#' fr_response |> +#' collect() |> +#' summarise( +#' n_rows = n(), +#' n_receptors = n_distinct(imd_receptor_id) +#' ) +#' # Expected result: +#' # n_rows n_receptors +#' # 955 871 +#' +#' # Keep receptors containing one reference CDR3 sequence +#' reference_cdr3 <- "ASFPVLSPYNEQF" #' -#' # --- filter examples --- -#' \dontrun{ -#' # Example 1: dplyr-style filtering on annotations -#' filtered_heavy <- filter(idata_test, chain == "IGH") -#' print(filtered_heavy) +#' exact_match <- idata |> +#' filter( +#' seq_options = make_seq_options( +#' query_col = "cdr3_aa", +#' patterns = reference_cdr3, +#' method = "exact" +#' ) +#' ) #' -#' # Example 2: Exact sequence matching on CDR3 amino acid sequence -#' cdr3_patterns <- c("CARGLGLVFYGMDVW", "CARDNRGAVAGVFGEAFYW") -#' seq_opts_exact <- make_seq_options(query_col = "CDR3_aa", patterns = cdr3_patterns) -#' filtered_exact_cdr3 <- filter(idata_test, seq_options = seq_opts_exact) -#' print(filtered_exact_cdr3) +#' exact_match |> +#' collect() |> +#' select(cdr3_aa, v_call, Response) +#' # Expected result: +#' # cdr3_aa v_call Response +#' # ASFPVLSPYNEQF TRBV28*01 FR #' -#' # Example 3: Combining dplyr-style and fuzzy sequence matching (Levenshtein) -#' seq_opts_lev <- make_seq_options( -#' query_col = "CDR3_aa", -#' patterns = "CARGLGLVFYGMDVW", -#' method = "lev", -#' max_dist = 1 -#' ) -#' filtered_combined <- filter(idata_test, -#' chain == "IGH", -#' C_gene == "IGHG1", -#' seq_options = seq_opts_lev -#' ) -#' print(filtered_combined) +#' # Keep receptors within four sequence changes of the reference +#' similar_sequences <- idata |> +#' filter( +#' seq_options = make_seq_options( +#' query_col = "cdr3_aa", +#' patterns = reference_cdr3, +#' method = "lev", +#' max_dist = 4 +#' ) +#' ) #' -#' # Example 4: Regex matching on V gene -#' v_gene_pattern <- "^IGHV[13]-" # Keep only IGHV1 or IGHV3 families -#' seq_opts_regex <- make_seq_options( -#' query_col = "V_gene", -#' patterns = v_gene_pattern, -#' method = "regex" -#' ) -#' filtered_regex_v <- filter(idata_test, seq_options = seq_opts_regex) -#' print(filtered_regex_v) +#' similar_sequences |> +#' collect() |> +#' distinct(cdr3_aa) |> +#' arrange(cdr3_aa) +#' # Expected result: +#' # cdr3_aa +#' # ASFPVLSPYNEQF +#' # ASSPDSPSYNEQF +#' # ASSPGLAAYNEQF +#' # ASSPTLYNEQF #' -#' # Example 5: Filtering without recalculating repertoires -#' filtered_no_rep <- filter(idata_test, chain == "IGK", keep_repertoires = FALSE) -#' print(filtered_no_rep) # $repertoires should be NULL -#' } +#' # Keep two selected cell barcodes +#' selected_barcodes <- c("S1_1", "S1_2") #' -#' # --- filter_barcodes example --- -#' \dontrun{ -#' # Assuming 'cell1_barcode' and 'cell5_barcode' exist in idata_test$annotations$cell_id -#' specific_barcodes <- c("cell1_barcode", "cell5_barcode") -#' filtered_cells <- filter_barcodes(idata_test, barcodes = specific_barcodes) -#' print(filtered_cells) -#' } +#' selected_cells <- idata |> +#' filter_barcodes(selected_barcodes) #' -#' # --- filter_receptors example --- -#' \dontrun{ -#' # Assuming receptor IDs 101 and 205 exist in idata_test$annotations$receptor_id -#' specific_receptors <- c(101, 205) # Or character IDs if applicable -#' filtered_recs <- filter_receptors(idata_test, receptors = specific_receptors) -#' print(filtered_recs) -#' } +#' selected_cells |> +#' collect() |> +#' distinct(imd_barcode) +#' # Expected result: +#' # imd_barcode +#' # S1_1 +#' # S1_2 +#' +#' # The same approach can keep selected receptor identifiers +#' selected_receptors <- idata |> +#' collect() |> +#' distinct(imd_receptor_id) |> +#' slice_head(n = 2) |> +#' pull(imd_receptor_id) +#' +#' selected_receptors_data <- idata |> +#' filter_receptors(selected_receptors) #' #' @concept filtering #' @export @@ -169,11 +192,14 @@ filter_immundata <- function(idata, ..., seq_options = NULL, keep_repertoires = # Exact # if (seq_options$method == "exact") { - new_annotations <- new_annotations |> filter(!!col_sym %in% seq_options$patterns) + filtered_universe <- new_annotations - keep_ids <- new_annotations |> select({{ receptor_id }}) + keep_ids <- filtered_universe |> + filter(!!col_sym %in% seq_options$patterns) |> + select(all_of(receptor_id)) |> + distinct() - new_annotations <- idata$annotations |> + new_annotations <- filtered_universe |> semi_join(keep_ids, by = receptor_id) } @@ -207,11 +233,16 @@ filter_immundata <- function(idata, ..., seq_options = NULL, keep_repertoires = # # Select only those receptors which passed the filer # - # TODO: Refactor, but I'm not sure how to do it properly. Simply split to separte functions + ? + # TODO: Refactor, but I'm not sure how to do it properly. Simply split to separate functions + ? # TODO: looks like a case for from receptors to annotations if (seq_options$method != "exact") { + keep_ids <- new_annotations |> + semi_join(distance_data, by = seq_options$query_col) |> + select(all_of(receptor_id)) |> + distinct() + new_annotations <- new_annotations |> - semi_join(distance_data, by = seq_options$query_col) + semi_join(keep_ids, by = receptor_id) } } @@ -225,13 +256,19 @@ filter_immundata <- function(idata, ..., seq_options = NULL, keep_repertoires = semi_join(keep_ids, by = receptor_id) } + keep_repertoires <- keep_repertoires && !is.null(idata$schema_repertoire) + if (!keep_repertoires) { + new_annotations <- drop_repertoire_state(new_annotations) + } + new_idata <- ImmunData$new( schema = idata$schema_receptor, - annotations = new_annotations + annotations = new_annotations, + provenance = get_provenance(idata) ) - if (keep_repertoires && !is.null(idata$schema_repertoire)) { - new_idata |> agg_repertoires(idata$schema_repertoire) + if (keep_repertoires) { + rebuild_repertoire_and_strata(new_idata, idata) } else { new_idata } @@ -260,13 +297,19 @@ filter_barcodes <- function(idata, barcodes, keep_repertoires = TRUE) { new_annotations <- idata$annotations |> semi_join(barcodes_table, by = barcode_col_id) + keep_repertoires <- keep_repertoires && !is.null(idata$schema_repertoire) + if (!keep_repertoires) { + new_annotations <- drop_repertoire_state(new_annotations) + } + new_idata <- ImmunData$new( schema = idata$schema_receptor, - annotations = new_annotations + annotations = new_annotations, + provenance = get_provenance(idata) ) - if (keep_repertoires && !is.null(idata$schema_repertoire)) { - new_idata |> agg_repertoires(idata$schema_repertoire) + if (keep_repertoires) { + rebuild_repertoire_and_strata(new_idata, idata) } else { new_idata } @@ -290,13 +333,19 @@ filter_receptors <- function(idata, receptors, keep_repertoires = TRUE) { new_annotations <- idata$annotations |> semi_join(receptors_table, by = receptors_col_id) + keep_repertoires <- keep_repertoires && !is.null(idata$schema_repertoire) + if (!keep_repertoires) { + new_annotations <- drop_repertoire_state(new_annotations) + } + new_idata <- ImmunData$new( schema = idata$schema_receptor, - annotations = new_annotations + annotations = new_annotations, + provenance = get_provenance(idata) ) - if (keep_repertoires && !is.null(idata$schema_repertoire)) { - new_idata |> agg_repertoires(idata$schema_repertoire) + if (keep_repertoires) { + rebuild_repertoire_and_strata(new_idata, idata) } else { new_idata } diff --git a/R/operations_mutate.R b/R/operations_mutate.R index f997e6b..8350bcb 100644 --- a/R/operations_mutate.R +++ b/R/operations_mutate.R @@ -1,162 +1,302 @@ -#' @title Modify or Add Columns to ImmunData Annotations +#' @title Add or change annotation columns in ImmunData #' #' @description -#' Applies transformations to the `$annotations` table within an `ImmunData` -#' object, similar to `dplyr::mutate`. It allows adding new columns or modifying -#' existing non-schema columns using standard `dplyr` expressions. Additionally, -#' it can add new columns based on sequence comparisons (exact match, regular -#' expression matching, or distance calculation) against specified patterns. +#' Use `mutate()` to add information to each row of an [ImmunData] object. For +#' example, you can calculate CDR3 length, mark sequences of interest, or compare +#' receptor sequences with reference sequences. +#' +#' The function returns a new [ImmunData] object. The original object is not +#' changed. +#' +#' This function is a direct implementation of [dplyr::mutate]. Alternative +#' function name is `mutate_immundata`. #' #' @details -#' The function operates in two main steps: -#' 1. **Standard Mutations (`...`)**: Applies the standard `dplyr::mutate`-style -#' expressions provided in `...` to the `$annotations` table. You can create -#' new columns or modify existing ones, but you *cannot* modify columns -#' defined in the core `ImmunData` schema (e.g., `receptor_id`, `cell_id`). -#' An error will occur if you attempt to do so. -#' 2. **Sequence-based Annotations (`seq_options`)**: If `seq_options` is provided, -#' the function calculates sequence similarities or distances and adds corresponding -#' new columns to the `$annotations` table. -#' * `method = "exact"`: Adds boolean columns (TRUE/FALSE) indicating whether the -#' `query_col` value exactly matches each `pattern`. Column names are generated -#' using a prefix (e.g., `sim_exact_`) and the pattern or its index. -#' * `method = "regex"`: Uses `annotate_tbl_regex` to add columns indicating -#' matches for each regular expression pattern against the `query_col`. The -#' exact nature of the added columns depends on `annotate_tbl_regex` (e.g., -#' boolean flags or captured groups). -#' * `method = "lev"` or `method = "hamm"`: Uses `annotate_tbl_distance` to -#' calculate Levenshtein or Hamming distances between the `query_col` and -#' each `pattern`, adding columns containing these numeric distances. -#' `max_dist` is ignored in this context (internally treated as `NA`) as -#' all distances are calculated and added, not used for filtering. -#' * The naming of the new sequence-based columns depends on the `name_type` -#' option within `seq_options` and internal helper functions like -#' `make_pattern_columns`. Prefixes like `sim_exact_`, `sim_regex_`, -#' `dist_lev_`, `dist_hamm_` are typically used based on the schema. -#' -#' The `$repertoires` table, if present in the input `idata`, is copied to the -#' output object without modification. This function only affects the `$annotations` -#' table. -#' -#' @param idata,.data An `ImmunData` object. -#' @param ... `dplyr::mutate`-style named expressions (e.g., `new_col = existing_col * 2`, -#' `category = ifelse(value > 10, "high", "low")`). These are applied first. -#' **Important**: You cannot use names for new or modified columns that conflict -#' with the core `ImmunData` schema columns (retrieved via `imd_schema()`). -#' @param seq_options Optional named list specifying sequence-based annotation options. -#' Use [make_seq_options()] for convenient creation. See `filter_immundata` -#' documentation (`?filter_immundata`) or the details section here for the list -#' structure (`query_col`, `patterns`, `method`, `name_type`). `max_dist` is -#' ignored for mutation. If `NULL` (the default), no sequence-based columns are added. -#' -#' @return A *new* `ImmunData` object with the `$annotations` table modified according -#' to the provided expressions and `seq_options`. The `$repertoires` table (if present) -#' is carried over unchanged from the input `idata`. -#' -#' @seealso [dplyr::mutate()], [make_seq_options()], [filter_immundata()], [ImmunData], -#' `vignette("immundata-classes", package = "immunarch")` (replace with actual package name if different) +#' You can use `mutate()` in three ways: +#' +#' * Supply named calculations in `...` to create annotation columns from +#' existing data. For example, `cmv_specific = cdr3_aa %in% cmv_cdr3s` adds a +#' column containing `TRUE` or `FALSE` for each row. +#' * Supply `.by` to perform calculations separately for temporary groups. The +#' number of rows does not change. A group statistic is repeated for all rows +#' in that group. +#' * Supply `seq_options`, created with [make_seq_options()], to compare a +#' sequence column with one or more reference sequences or patterns. One result +#' column is added for each reference. +#' +#' Named calculations in `...` are performed before sequence comparisons. +#' +#' Most grouped calculations are translated directly to DuckDB. Some group +#' statistics, such as `n_distinct()`, are not available as DuckDB window +#' calculations when a large dataset must stay on disk. In that case, `mutate()` +#' automatically calculates one summary row per group and joins the values back +#' to the annotation rows. This remains lazy and does not load the full dataset +#' into R memory. +#' +#' The automatic fallback works when every calculation in the call produces one +#' value per group. If a call combines a row-level calculation with a group +#' statistic that needs the fallback, use two `mutate()` calls. Also use a second +#' call when a later calculation refers to a group statistic created by the +#' fallback. See the examples below. +#' +#' Columns used to identify receptors or repertoires, and identifiers managed by +#' `ImmunData`, are protected. This prevents accidental changes that would make +#' the object inconsistent. You can add new columns and change other annotation +#' columns. +#' +#' Sequence comparison methods are: +#' +#' * `"exact"`: `TRUE` when the sequence is identical to the reference. +#' * `"regex"`: `TRUE` when the sequence matches a regular-expression pattern. +#' This is an advanced option for matching text patterns. +#' * `"lev"`: the number of substitutions, insertions, or deletions needed to +#' change one sequence into the other. +#' * `"hamm"`: the number of different positions between sequences of the same +#' length. Sequences with different lengths receive `NA`. +#' +#' For the distance methods, `0` means an exact match and smaller values mean +#' more similar sequences. With `name_type = "index"`, the result columns have +#' short names such as `imd_sim_exact_1` or `imd_sim_lev_1`. With +#' `name_type = "pattern"`, each column name includes its reference pattern. +#' +#' `max_dist` is used by [filter_immundata()] but has no effect here because +#' `mutate()` reports every calculated distance. +#' +#' Existing repertoire and strata summaries are carried to the new object +#' without modification. +#' +#' @param idata,.data An [ImmunData] object. +#' @param ... One or more named calculations in the form +#' `new_column = calculation`. Refer to existing columns directly by name. You +#' can add new annotation columns or change columns that are not protected. +#' @param .by Optional columns used to form temporary groups for this operation. +#' For example, `.by = Response` calculates separately for each response, and +#' `.by = c(Response, imd_group_id)` uses each response and receptor-cluster +#' combination. The grouping applies only to this `mutate()` call. +#' @param seq_options Options for comparing sequences with reference sequences or +#' patterns. Create these options with [make_seq_options()]. If `NULL`, the +#' default, no sequence comparisons are performed. +#' +#' @return A new [ImmunData] object containing the added or changed annotation +#' columns. Existing repertoire and strata summaries are preserved. +#' +#' @seealso [dplyr::mutate()], [make_seq_options()], [filter_immundata()], +#' [annotate_receptors()], [agg_repertoires()], [ImmunData] #' #' @examples -#' # Basic setup (assuming idata_test is a valid ImmunData object) -#' # print(idata_test) -#' -#' \dontrun{ -#' # Example 1: Add a simple derived column -#' idata_mut1 <- mutate(idata_test, V_family = substr(V_gene, 1, 5)) -#' print(idata_mut1$annotations) -#' -#' # Example 2: Add multiple columns and modify one (if 'custom_score' exists) -#' # Note: Avoid modifying core schema columns like 'V_gene' itself. -#' idata_mut2 <- mutate(idata_test, -#' V_basic = gsub("-.*", "", V_gene), -#' J_len = nchar(J_gene), -#' custom_score = custom_score * 1.1 -#' ) # Fails if custom_score doesn't exist -#' print(idata_mut2$annotations) -#' -#' # Example 3: Add boolean columns for exact CDR3 matches -#' cdr3_patterns <- c("CARGLGLVFYGMDVW", "CARDNRGAVAGVFGEAFYW") -#' seq_opts_exact <- make_seq_options( -#' query_col = "CDR3_aa", -#' patterns = cdr3_patterns, -#' method = "exact", -#' name_type = "pattern" -#' ) # Name cols by pattern -#' idata_mut_exact <- mutate(idata_test, seq_options = seq_opts_exact) -#' # Look for new columns like 'sim_exact_CARGLGLVFYGMDVW' -#' print(idata_mut_exact$annotations) -#' -#' # Example 4: Add Levenshtein distance columns for a CDR3 pattern -#' seq_opts_lev <- make_seq_options( -#' query_col = "CDR3_aa", -#' patterns = "CARGLGLVFYGMDVW", -#' method = "lev", -#' name_type = "index" -#' ) # Name col like 'dist_lev_1' -#' idata_mut_lev <- mutate(idata_test, seq_options = seq_opts_lev) -#' # Look for new column 'dist_lev_1' (or similar based on schema) -#' print(idata_mut_lev$annotations) -#' -#' # Example 5: Combine standard mutation and sequence annotation -#' seq_opts_regex <- make_seq_options( -#' query_col = "V_gene", -#' patterns = c(ighv1 = "^IGHV1-", ighv3 = "^IGHV3-"), -#' method = "regex", -#' name_type = "pattern" +#' library(immundata) +#' library(dplyr) +#' +#' options(immundata.verbose = FALSE) +#' +#' # Load data included with immundata +#' idata <- get_test_idata() +#' +#' # Add the length of each CDR3 amino acid sequence +#' idata_with_length <- idata |> +#' mutate(cdr3_length = dd$length(cdr3_aa)) +#' +#' idata_with_length |> +#' collect() |> +#' select(cdr3_aa, cdr3_length) |> +#' slice_head(n = 3) +#' # Expected result: +#' # cdr3_aa cdr3_length +#' # ASFPVLSPYNEQF 13 +#' # ASRAGAGTGELF 12 +#' # ASSPGQGLDTQY 12 +#' +#' # Compare CDR3 sequences with one reference sequence +#' reference_cdr3 <- "ASFPVLSPYNEQF" +#' +#' idata_with_matches <- idata |> +#' mutate( +#' seq_options = make_seq_options( +#' query_col = "cdr3_aa", +#' patterns = reference_cdr3, +#' method = "exact" +#' ) +#' ) +#' +#' idata_with_matches |> +#' collect() |> +#' count(imd_sim_exact_1) +#' # Expected result: +#' # imd_sim_exact_1 n +#' # FALSE 1901 +#' # TRUE 1 +#' +#' # Calculate Levenshtein distance from the reference sequence +#' idata_with_distance <- idata |> +#' mutate( +#' seq_options = make_seq_options( +#' query_col = "cdr3_aa", +#' patterns = reference_cdr3, +#' method = "lev" +#' ) +#' ) +#' +#' idata_with_distance |> +#' collect() |> +#' select(cdr3_aa, imd_sim_lev_1) |> +#' arrange(imd_sim_lev_1, cdr3_aa) |> +#' slice_head(n = 3) +#' # Expected result: +#' # cdr3_aa imd_sim_lev_1 +#' # ASFPVLSPYNEQF 0 +#' # ASSPDSPSYNEQF 4 +#' # ASSPGLAAYNEQF 4 +#' +#' # Mark selected sequences +#' cmv_cdr3s <- c( +#' "ASFPVLSPYNEQF", +#' "ASRAGAGTGELF" #' ) -#' idata_mut_combo <- mutate(idata_test, -#' chain_upper = toupper(chain), -#' seq_options = seq_opts_regex +#' +#' marked_sequences <- idata |> +#' mutate( +#' cmv_specific = cdr3_aa %in% cmv_cdr3s +#' ) +#' +#' marked_sequences |> +#' collect() |> +#' count(cmv_specific) +#' # Expected result: +#' # cmv_specific n +#' # FALSE 1900 +#' # TRUE 2 +#' +#' # Mark selected receptor identities +#' cmv_hits <- tibble( +#' imd_receptor_id = c(1L, 105L), +#' cmv_specific = TRUE #' ) -#' # Look for 'chain_upper' and regex match columns (e.g., 'sim_regex_ighv1') -#' print(idata_mut_combo) -#' } +#' +#' marked_receptors <- idata |> +#' annotate_receptors(cmv_hits) |> +#' mutate( +#' cmv_specific = coalesce(cmv_specific, FALSE) +#' ) +#' +#' marked_receptors |> +#' collect() |> +#' count(cmv_specific) +#' # Expected result: +#' # cmv_specific n +#' # FALSE 1898 +#' # TRUE 4 +#' +#' # Add response-level statistics to every annotation row +#' # `.by` means: calculate separately for each response. +#' response_stats <- idata |> +#' mutate( +#' response_n_rows = n(), +#' response_n_receptors = n_distinct(imd_receptor_id), +#' .by = Response +#' ) +#' +#' response_stats |> +#' collect() |> +#' distinct(Response, response_n_rows, response_n_receptors) |> +#' arrange(Response) +#' # Expected result: +#' # Response response_n_rows response_n_receptors +#' # FR 955 871 +#' # PR 947 867 +#' +#' # A grouped calculation can also produce a different value for every row. +#' response_centered <- idata |> +#' mutate( +#' centered_counts = counts - mean(counts, na.rm = TRUE), +#' .by = Response +#' ) +#' +#' # Do not combine that row-level calculation with a statistic that needs the +#' # automatic summary fallback in the same call: +#' # idata |> +#' # mutate( +#' # centered_counts = counts - mean(counts, na.rm = TRUE), +#' # response_n_receptors = n_distinct(imd_receptor_id), +#' # .by = Response +#' # ) +#' +#' # Use two mutate calls instead. The work remains lazy in DuckDB. +#' response_details <- idata |> +#' mutate( +#' centered_counts = counts - mean(counts, na.rm = TRUE), +#' .by = Response +#' ) |> +#' mutate( +#' response_n_receptors = n_distinct(imd_receptor_id), +#' .by = Response +#' ) +#' +#' # Also use a second call when a new calculation uses a statistic created by +#' # the fallback. +#' response_details <- idata |> +#' mutate( +#' response_n_receptors = n_distinct(imd_receptor_id), +#' .by = Response +#' ) |> +#' mutate( +#' twice_response_n_receptors = response_n_receptors * 2 +#' ) #' #' @concept mutation #' @export mutate_immundata <- function(idata, ..., + .by = NULL, seq_options = NULL) { checkmate::assert_r6(idata, "ImmunData") checkmate::assert_list(seq_options, null.ok = TRUE) dots <- rlang::enquos(..., .named = TRUE) # keep names exactly as passed - bad <- names(dots)[names(dots) %in% imd_schema()] + by <- rlang::enquo(.by) + assert_mutable_annotation_columns(idata, names(dots)) - if (length(bad)) { - cli::cli_abort( - "You cannot create or overwrite system columns. Offending names: {.val {bad}}" + sequence_annotation_cols <- NULL + if (!is.null(seq_options)) { + seq_options <- check_seq_options(seq_options, mode = "mutate") + sequence_col_prefix <- switch(seq_options$method, + exact = imd_schema("sim_exact"), + regex = imd_schema("sim_regex"), + lev = imd_schema("sim_lev"), + hamm = imd_schema("sim_hamm") + ) + sequence_annotation_cols <- make_pattern_columns( + patterns = seq_options$patterns, + col_prefix = sequence_col_prefix, + name_type = seq_options$name_type ) + + assert_mutable_annotation_columns(idata, sequence_annotation_cols) } # Run "basic" mutate first new_annotations <- idata$annotations if (length(dots) > 0) { - new_annotations <- new_annotations |> mutate(!!!dots) + new_annotations <- mutate_annotations_by( + annotations = new_annotations, + dots = dots, + by = by + ) } receptor_id <- imd_schema("receptor") # Run the sequence-based mutations if (!is.null(seq_options)) { - seq_options <- check_seq_options(seq_options, mode = "mutate") - col_sym <- rlang::sym(seq_options$query_col) # # Exact # if (seq_options$method == "exact") { - dist_cols <- make_pattern_columns( - patterns = seq_options$patterns, - col_prefix = imd_schema("sim_exact"), - name_type = seq_options$name_type - ) - for (p_index in seq_along(seq_options$patterns)) { p_seq <- seq_options$patterns[p_index] new_annotations <- new_annotations |> - mutate(!!rlang::sym(dist_cols[p_index]) := !!col_sym == p_seq) + mutate(!!rlang::sym(sequence_annotation_cols[p_index]) := !!col_sym == p_seq) } } else { # @@ -194,18 +334,74 @@ mutate_immundata <- function(idata, } } - new_idata <- ImmunData$new( - schema = idata$schema_receptor, - annotations = new_annotations, - repertoires = idata$repertoires - ) - - new_idata + clone_with_annotations(idata, new_annotations) } #' @rdname mutate_immundata #' @exportS3Method dplyr::mutate -mutate.ImmunData <- function(.data, ..., seq_options = NULL) { - mutate_immundata(idata = .data, ..., seq_options = seq_options) +mutate.ImmunData <- function(.data, ..., .by = NULL, seq_options = NULL) { + mutate_immundata( + idata = .data, + ..., + .by = {{ .by }}, + seq_options = seq_options + ) +} + + +is_unsupported_duckplyr_window_error <- function(error) { + parent <- error$parent + + inherits(error, "rlang_error") && + !is.null(parent) && + grepl( + "stingy duckplyr frame", + conditionMessage(error), + fixed = TRUE + ) && + grepl( + "not supported in window functions", + conditionMessage(parent), + fixed = TRUE + ) +} + +mutate_annotations_by <- function(annotations, dots, by) { + mutated <- tryCatch( + annotations |> + mutate(!!!dots, .by = !!by), + error = identity + ) + + if (!inherits(mutated, "error")) { + return(mutated) + } + + if (!is_unsupported_duckplyr_window_error(mutated) || rlang::quo_is_null(by)) { + rlang::cnd_signal(mutated) + } + + by_names <- names(annotations |> select(!!by)) + if (length(intersect(names(dots), by_names)) > 0) { + rlang::cnd_signal(mutated) + } + + stats <- tryCatch( + annotations |> + summarise(!!!dots, .by = !!by), + error = identity + ) + + if (inherits(stats, "error")) { + rlang::cnd_signal(mutated) + } + + value_names <- setdiff(names(stats), by_names) + desired_order <- base::union(names(annotations), value_names) + + annotations |> + select(-any_of(value_names)) |> + left_join(stats, by = by_names, na_matches = "na") |> + select(all_of(desired_order)) } diff --git a/R/operations_print.R b/R/operations_print.R index fb9e20b..d913fe7 100644 --- a/R/operations_print.R +++ b/R/operations_print.R @@ -1,3 +1,51 @@ +#' @title Display the contents and biological definitions of ImmunData +#' +#' @description +#' Use `print()` to inspect the receptor table, chain annotations, and biological +#' schemas stored in an [ImmunData] object. +#' +#' Use this method for a quick overview after reading, filtering, or aggregating +#' repertoire data. It displays the units available in the object: receptors, +#' chain rows, repertoires, and strata. It also shows the feature and chain +#' definitions used to construct receptors. +#' +#' Printing is read-only. It does not collect the complete dataset into R and +#' does not change the original object. The object is returned invisibly so it +#' can still be assigned or used in a pipeline. +#' +#' @details +#' A section is shown only when that information is available. An object without +#' repertoire definitions, for example, has no repertoire schema or repertoire +#' summary section. Duckplyr prints a preview of large tables rather than every +#' row. +#' +#' @param x An [ImmunData] object to display. +#' @param ... Additional arguments. Currently not used. +#' +#' @return `x`, invisibly. The displayed output is a human-readable overview; +#' no data are modified. +#' +#' @seealso [ImmunData], [dplyr::collect()], [dplyr::count()] +#' +#' @examples +#' library(immundata) +#' +#' options(immundata.verbose = FALSE) +#' idata <- get_test_idata() +#' +#' print(idata) +#' # Expected output contains these sections: +#' # ImmunData +#' # Receptors +#' # Annotations +#' # Receptor schema +#' # Repertoire schema +#' # List of repertoires +#' +#' # `Receptors` previews distinct biological receptor definitions. +#' # `Annotations` previews the retained chain rows and sample information. +#' # The schema sections explain how receptors and repertoires were defined. +#' #' @concept operations #' @export print.ImmunData <- function(x, ...) { @@ -45,11 +93,25 @@ print.ImmunData <- function(x, ...) { cli::cli_bullets(schema) } + if (!is.null(x$schema_strata)) { + cli::cat_line() + cli::cli_h2("{cli::col_br_blue('Strata schema:')}") + schema <- x$schema_strata + names(schema) <- rep(">", times = length(schema)) + cli::cli_bullets(schema) + } + if (!is.null(x$repertoires)) { cli::cat_line() cli::cli_h2("{cli::col_br_cyan('List of repertoires:')}") print(x$repertoires) } + if (!is.null(x$strata)) { + cli::cat_line() + cli::cli_h2("{cli::col_br_blue('List of strata:')}") + print(x$strata) + } + invisible(x) } diff --git a/R/operations_utils.R b/R/operations_utils.R index 50ac670..c957431 100644 --- a/R/operations_utils.R +++ b/R/operations_utils.R @@ -1,262 +1,93 @@ -#' @title Build a `seq_options` list for sequence‑based receptor filtering -#' -#' @description -#' A convenience wrapper that validates the common arguments for -#' **`filter_receptors()`** and returns them in the required list form. -#' -#' @param query_col Character(1). Name of the receptor column to compare -#' (e.g. `"cdr3_aa"`). -#' @param patterns Character vector of sequences or regular expressions to -#' search for. -#' @param method One of `"exact"`, `"regex"`, `"lev"` (Levenshtein), or -#' `"hamm"` (Hamming). Defaults to `"exact"`. -#' @param max_dist Numeric distance threshold for `"lev"` / `"hamm"` -#' filtering. Use `NA` (default) to keep all rows after annotation. -#' @param name_type Passed straight to `annotate_tbl_distance()`; either -#' `"index"` (default) or `"pattern"`. -#' -#' @return A named list suitable for the `seq_options` argument of -#' [filter_receptors()]. -#' -#' @seealso [filter_receptors()], [annotate_receptors()] -#' -#' @concept utils -#' @export -make_seq_options <- function(query_col, - patterns, - method = c("exact", "lev", "hamm", "regex"), - max_dist = NA, - name_type = c("index", "pattern")) { - checkmate::assert_character(query_col, len = 1) - checkmate::assert_character(patterns, min.len = 1) - - list( - query_col = query_col, - patterns = patterns, - method = match.arg(method), - max_dist = max_dist, - name_type = match.arg(name_type) +drop_repertoire_state <- function(annotations) { + repertoire_state_cols <- c( + imd_schema("repertoire"), + imd_schema("strata"), + imd_schema("strata_name"), + imd_schema("count"), + imd_schema("proportion"), + imd_schema("n_receptors"), + imd_schema("n_barcodes"), + imd_schema("n_repertoires") ) -} - -check_seq_options <- function(seq_options, mode = NULL) { - checkmate::check_list(seq_options, null.ok = FALSE) - checkmate::check_choice(mode, choices = c("filter", "mutate"), null.ok = FALSE) - - if (!is.null(seq_options$patterns) && - length(seq_options$patterns) > 0 && - !is.null(seq_options$query_col)) { - defaults <- list(method = "exact", max_dist = NA, name_type = "index") - - seq_options <- utils::modifyList(defaults, seq_options) - - seq_options$method <- match.arg(seq_options$method, c("exact", "regex", "lev", "hamm")) - if (mode == "filter" && - is.na(seq_options$max_dist) && - seq_options$method %in% c("lev", "hamm")) { - cli::cli_abort("You passed `seq_options` to `filter`, but didn't provide `max_dist` for filtering. Either provide `max_dist` or use `left_join` to annotate receptors with distances to patterns.") - } - - seq_options - } else { - cli::cli_abort("Missing fields in `seq_options`, please use {.run immundata::make_seq_options()} to create the options") - } + annotations |> select(-any_of(repertoire_state_cols)) } +rebuild_repertoire_and_strata <- function(idata, source_idata) { + checkmate::assert_r6(idata, "ImmunData") + checkmate::assert_r6(source_idata, "ImmunData") -make_pattern_columns <- function(patterns, - col_prefix, - name_type = c("pattern", "index")) { - checkmate::assert_character(patterns, min.len = 1) - checkmate::assert_character(col_prefix, max.len = 1) - name_type <- match.arg(name_type) - - sapply(seq_along(patterns), function(p_index) { - p_seq <- patterns[[p_index]] + rebuilt <- agg_repertoires(idata, source_idata$schema_repertoire) - if (name_type == "pattern") { - safe_name <- gsub("[^A-Za-z0-9]", "_", p_seq) # just in case - col_name_out <- paste0(col_prefix, safe_name) - } else if (name_type == "index") { - col_name_out <- paste0(col_prefix, p_index) - } else { - # TODO: what the heck - stop("!") - } - - col_name_out - }) -} - - -#' @keywords internal -annotate_tbl_distance <- function(tbl_data, - query_col, - patterns, - method = c("lev", "hamm"), - max_dist = NA, - name_type = c("pattern", "index")) { - checkmate::assert_character(query_col, len = 1) - checkmate::assert_character(patterns, min.len = 1) - - method <- match.arg(method) - name_type <- match.arg(name_type) - - uniq <- tbl_data |> - distinct(!!rlang::sym(query_col)) |> - as_tbl() - - # TODO: settings for kmers - if (!is.na(max_dist)) { - uniq <- uniq |> - mutate( - kmer_left = dbplyr::sql(cli::format_inline("{query_col}[:3]")), - kmer_right = dbplyr::sql(cli::format_inline("{query_col}[-2:]")) - ) - } - - if (method == "lev") { - col_prefix <- imd_schema("sim_lev") - } else if (method == "hamm") { - col_prefix <- imd_schema("sim_hamm") + if (is.null(source_idata$schema_strata) || is.null(source_idata$strata)) { + return(rebuilt) } - dist_cols <- make_pattern_columns( - patterns = patterns, - col_prefix = col_prefix, - name_type = name_type - ) - - # TODO: Optimize it via SQL instead of cycles - if it is even needed... - # TODO: lump together multiple patterns in batches - for (i in seq_along(patterns)) { - p <- patterns[[i]] - col_name_out <- dist_cols[i] - - # - # 1) Levenshtein distance - # - if (method == "lev") { - if (!is.na(max_dist)) { - len_p <- nchar(p) - sql_expr <- cli::format_inline( - "CASE WHEN ", - " kmer_left = {query_col}[:3] AND kmer_right = {query_col}[-2:] AND", - " length({query_col}) >= {len_p - max_dist} AND length({query_col}) <= {len_p + max_dist}", - " THEN levenshtein({query_col}, '{p}')", - " ELSE NULL END" - ) - } else { - len_p <- nchar(p) - sql_expr <- cli::format_inline( - "levenshtein({query_col}, '{p}')" - ) - } - - uniq <- uniq |> - mutate({{ col_name_out }} := dbplyr::sql(sql_expr)) - } - - # - # 2) Hamming distance - # - else { - len_p <- nchar(p) - sql_expr <- cli::format_inline( - "CASE WHEN length({query_col}) = {len_p}", - " THEN hamming({query_col}, '{p}')", - " ELSE NULL END" + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + rebuilt <- rebuilt |> agg_strata(source_idata$schema_strata) + + old_strata_labels <- source_idata$strata |> + duckplyr::as_duckdb_tibble() |> + select(all_of(c(source_idata$schema_strata, strata_name_col))) + rebuilt_strata_labels <- rebuilt$strata |> + duckplyr::as_duckdb_tibble() |> + select(all_of(c(strata_col, source_idata$schema_strata))) |> + left_join( + old_strata_labels, + by = source_idata$schema_strata, + na_matches = "na" + ) |> + collect() + + strata_names <- rebuilt_strata_labels[[strata_name_col]] + if (all(!is.na(strata_names))) { + rebuilt <- rename_strata( + rebuilt, + names = stats::setNames( + as.character(strata_names), + as.character(rebuilt_strata_labels[[strata_col]]) ) - - uniq <- uniq |> - mutate({{ col_name_out }} := dbplyr::sql(sql_expr)) - } - } - - # - # TODO: In case of max_dist, pre-optimize for levenshtein by filtering out too short or too long distances - # TODO: benchmark 1 - distinct vs no distinct - # TODO: benchmark 2 - pre-optimize vs no optimize - - # TODO: fun experiment - compute for patterns, then filter out, then compute again, and so on. - # filter out -> filter out those who has <= max_dist (!) because we already found them and just need to store - - # TODO: Benchmarks - # 1) distinct vs non-distinct - # 2) pre-optimize vs no optimization for levenshtein - # 3) step-by-step filtering out "good" sequences - # 4) precompute sequence length before (!) any filtering, on data loading, and don't compute it here - - # TODO: max dist. Left join - compute. Right join - filter - - if (is.na(max_dist)) { - uniq <- uniq |> - as_duckdb_tibble() - } else { - sql_expr <- sprintf("LEAST(%s) <= %d", paste(dist_cols, collapse = ", "), max_dist) - - uniq <- uniq |> - filter(dbplyr::sql(sql_expr)) |> - as_duckdb_tibble() + ) } - uniq |> compute() + rebuilt } +protected_annotation_columns <- function(idata) { + unique(c( + unname(unlist(imd_schema(), use.names = FALSE)), + imd_receptor_features(idata$schema_receptor), + idata$schema_repertoire + )) +} -#' @keywords internal -annotate_tbl_regex <- function(tbl_data, - query_col, - patterns, - filter_out = FALSE, - name_type = c("index", "pattern")) { - checkmate::assert_character(query_col, len = 1) - checkmate::assert_character(patterns, min.len = 1) - checkmate::assert_logical(filter_out) - - name_type <- match.arg(name_type) - - uniq <- tbl_data |> distinct(!!rlang::sym(query_col)) - - col_prefix <- imd_schema("sim_regex") - - dist_cols <- make_pattern_columns( - patterns = patterns, - col_prefix = col_prefix, - name_type = name_type - ) - - # TODO: Optimize it via SQL instead of cycles - if it is even needed... - for (i in seq_along(patterns)) { - p <- patterns[[i]] - col_name_out <- dist_cols[i] +assert_mutable_annotation_columns <- function(idata, columns) { + protected_cols <- protected_annotation_columns(idata) + bad <- intersect(unique(columns), protected_cols) - # annotate with DuckDB regexp_matches() - uniq <- uniq |> - mutate(!!col_name_out := dd$regexp_matches(!!rlang::sym(query_col), p)) + if (length(bad) == 0) { + return(invisible(TRUE)) } - tbl_data <- tbl_data |> left_join(uniq, by = query_col) - if (filter_out) { - # TODO: need to replace it with if_else when it is available in duckplyr - sql_expr <- paste(dist_cols, collapse = " OR ") + system_cols <- unique(unname(unlist(imd_schema(), use.names = FALSE))) + bad_system <- intersect(bad, system_cols) + bad_schema <- setdiff(bad, system_cols) + messages <- "You cannot create or overwrite protected ImmunData columns." - tbl_data |> - as_tbl() |> - filter(dbplyr::sql(sql_expr)) |> - as_duckdb_tibble() |> - compute() # TODO: We need a compute here because sometimes duckplyr can't find the table - } else { - tbl_data + if (length(bad_system) > 0) { + messages <- c( + messages, + "x" = "You cannot create or overwrite system columns. Offending names: {.val {bad_system}}" + ) } -} - -to_sym <- function(val) { - if (length(val) == 1) { - rlang::sym(val) - } else { - rlang::syms(val) + if (length(bad_schema) > 0) { + messages <- c( + messages, + "x" = "You cannot create or overwrite receptor or repertoire schema columns. Offending names: {.val {bad_schema}}" + ) } + + cli::cli_abort(messages) } diff --git a/R/operations_utils_levenshtein.R b/R/operations_utils_levenshtein.R deleted file mode 100644 index 961de42..0000000 --- a/R/operations_utils_levenshtein.R +++ /dev/null @@ -1,22 +0,0 @@ -filter_by_levenshtein <- function(uniq_seq_tbl, - query_col, - patterns, - pattern_cols, - backend = c("duckdb", "stringdist", "hybrid")) { -} - - -filter_by_levenshtein <- function(uniq_seq_tbl, - query_col, - patterns, - pattern_cols, - max_dist = 2, - kmer_left = 3, - kmer_right = 2, - backend = c("duckdb", "stringdist", "hybrid")) { - checkmate::assert_character(patterns) - checkmate::assert_character(pattern_cols, len = length(patterns)) - checkmate::assert_integer(max_dist, lower = 1) - checkmate::assert_integer(kmer_left, lower = 0) - checkmate::assert_integer(kmer_left, lower = 0) -} diff --git a/R/test_utils.R b/R/test_utils.R deleted file mode 100644 index 30dcc38..0000000 --- a/R/test_utils.R +++ /dev/null @@ -1,47 +0,0 @@ -get_test_idata_tsv_no_metadata <- function(schema = c("cdr3_aa", "v_call")) { - sample_files <- c( - system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata"), - system.file("extdata/tsv", "sample_1k_2k.tsv", package = "immundata") - ) - read_repertoires( - path = sample_files, - schema = schema, - output_folder = tempfile() - ) -} - -get_test_idata_tsv_with_metadata <- function(schema = c("cdr3_aa", "v_call")) { - md_path <- system.file("extdata/tsv", "metadata.tsv", package = "immundata") - md <- read_metadata(md_path) - - sample_files <- c( - system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata"), - system.file("extdata/tsv", "sample_1k_2k.tsv", package = "immundata") - ) - read_repertoires( - path = sample_files, - schema = schema, - metadata = md, - output_folder = tempfile() - ) -} - -#' Get test datasets from `immundata` -#' @keywords internal -#' @export -get_test_idata <- function() { - get_test_idata_tsv_with_metadata() -} - -#' Get test datasets from `immundata` -#' @keywords internal -#' @export -get_test_immundata <- function() { - get_test_idata_tsv_with_metadata() -} - -get_test_idata_tsv_metadata <- function() { - md_path <- system.file("extdata/tsv", "metadata_samples.tsv", package = "immundata") - - read_metadata(md_path) -} diff --git a/R/utils_schema.R b/R/utils_schema.R new file mode 100644 index 0000000..e1e7114 --- /dev/null +++ b/R/utils_schema.R @@ -0,0 +1,207 @@ +#' @title Define which chain observations form the same receptor +#' +#' @description +#' Use `make_receptor_schema()` to define a biological receptor from sequence +#' features and one or two receptor chains. +#' +#' Use this function when reading single-cell data with one selected chain, when +#' pairing chains such as TRA-TRB, or when accepting alternative light chains +#' such as IGK or IGL. The unit being defined is the receptor. Creating a schema +#' does not change any data or an existing [ImmunData] object. +#' +#' @section When observations are the same receptor: +#' +#' `features` names the fields that define chain identity. Common choices are +#' the CDR3 amino acid sequence, V gene, and J gene. Two observations represent +#' the same receptor only when the relevant chain loci and every selected +#' feature match. +#' +#' * With one chain, such as `chains = "TRB"`, only that locus is used. Two TRB +#' observations are the same receptor when all their selected feature values +#' match. +#' * With a strict pair, such as `chains = c("TRA", "TRB")`, chains are first +#' paired within each cell barcode. Receptors from two cells are the same only +#' when every selected TRA feature and every selected TRB feature match. +#' * With an alternative second chain, such as +#' `chains = c("IGH", "IGK|IGL")`, each receptor must contain IGH and exactly +#' one of IGK or IGL. Cells containing both IGK and IGL are excluded. The +#' light-chain locus and all selected heavy- and light-chain features must +#' match for two observations to be the same receptor. +#' +#' A barcode determines which chains belong to the same cell; it does not by +#' itself define receptor identity across cells. During single-cell import, +#' [read_repertoires()] uses `umi_col` to choose one chain when a cell contains +#' several observations from the same locus. +#' +#' Use `chains = NULL` for chain-agnostic bulk or pre-filtered data. In that +#' case, only the values in `features` define receptor identity. +#' +#' @section Validate a schema: +#' +#' `assert_receptor_schema()` stops with an error if `schema` is not accepted. +#' Use it inside another function when invalid input must stop the calculation. +#' `test_receptor_schema()` returns one `TRUE` or `FALSE` value and is useful in +#' conditional code. +#' +#' @section Backend and storage: +#' +#' A receptor schema is a small R list containing `features` and `chains`. It +#' stores no sequence data. [read_repertoires()] and [agg_receptors()] apply the +#' schema to chain observations using duckplyr. +#' +#' @param features A non-empty character vector. Column names containing the +#' chain fields that must match, such as +#' `c("junction_aa", "v_call", "j_call")`. Use names as they appear after any +#' input-column renaming. +#' @param chains A character vector of length one or two, or `NULL`. Use one +#' value, such as `"TRB"`, to keep one chain; two values, such as +#' `c("TRA", "TRB")`, to define a strict pair; or the `"IGK|IGL"` syntax in +#' the second value to accept either alternative. The default is `NULL`, which +#' does not select loci. +#' @param schema A non-empty character vector or receptor-schema list. An object +#' to check. A schema created by `make_receptor_schema()` is accepted. A +#' character vector supplies feature names for a chain-agnostic schema. +#' +#' @return `make_receptor_schema()` returns a list with character elements +#' `features` and `chains`; `chains` is `NULL` when loci are not selected. +#' `assert_receptor_schema()` returns `TRUE` for accepted input and otherwise +#' stops with an error. `test_receptor_schema()` returns one logical value. +#' +#' @seealso [read_repertoires()], [agg_receptors()], [imd_schema()] +#' +#' @examples +#' # Single-chain TCR: compare TRB observations by CDR3, V gene, and J gene. +#' trb_schema <- make_receptor_schema( +#' features = c("junction_aa", "v_call", "j_call"), +#' chains = "TRB" +#' ) +#' trb_schema +#' # Expected result: +#' # $features: "junction_aa" "v_call" "j_call" +#' # $chains: "TRB" +#' +#' # Paired alpha-beta TCR: all selected fields must match on both TRA and TRB. +#' ab_tcr_schema <- make_receptor_schema( +#' features = c("junction_aa", "v_call", "j_call"), +#' chains = c("TRA", "TRB") +#' ) +#' ab_tcr_schema +#' # The result defines one receptor as a matched TRA-TRB pair from one cell. +#' +#' # BCR: require IGH and accept either an IGK or IGL light chain. +#' bcr_schema <- make_receptor_schema( +#' features = c("junction_aa", "v_call", "j_call"), +#' chains = c("IGH", "IGK|IGL") +#' ) +#' bcr_schema +#' # The result accepts IGH-IGK and IGH-IGL receptors, while keeping the two +#' # light-chain loci biologically distinct. +#' +#' test_receptor_schema(bcr_schema) +#' # Expected result: TRUE +#' +#' @rdname make_receptor_schema +#' @concept utils +#' @export +make_receptor_schema <- function(features, chains = NULL) { + checkmate::assert_character(features, min.len = 1, any.missing = FALSE) + checkmate::assert_character( + chains, + min.len = 1, + max.len = 2, + any.missing = FALSE, + null.ok = TRUE + ) + + list(features = features, chains = chains) +} + + +#' @rdname make_receptor_schema +#' @export +assert_receptor_schema <- function(schema) { + # TODO: globals.R with schema list + + if (checkmate::test_character(schema, min.len = 1, any.missing = FALSE)) { + schema <- make_receptor_schema(features = schema) + } else { + checkmate::assert_list(schema, len = 2, null.ok = FALSE) + checkmate::assert_names( + names(schema), + permutation.of = c("features", "chains") + ) + checkmate::assert_character( + schema[["features"]], + min.len = 1, + any.missing = FALSE + ) + checkmate::assert_character( + schema[["chains"]], + min.len = 1, + max.len = 2, + any.missing = FALSE, + null.ok = TRUE + ) + } + + receptor_chains <- imd_receptor_chains(schema) + + if (!is.null(receptor_chains)) { + # Validate chain syntax rules + if (length(receptor_chains) > 2) { + cli::cli_abort("Schema can have at most 2 chain elements. Found {length(receptor_chains)}: [{paste(receptor_chains, collapse=', ')}]") + } + + if (length(receptor_chains) >= 1) { + # Check first chain doesn't contain pipe + if (grepl("\\|", receptor_chains[1])) { + cli::cli_abort("The first chain in the schema cannot contain '|' character. Found: '{receptor_chains[1]}'. The OR syntax is only allowed in the second chain.") + } + } + + if (length(receptor_chains) == 2) { + # Check if second chain contains OR syntax (|) + if (grepl("\\|", receptor_chains[2])) { + # Split and validate the alternatives + relaxed_chain_alternatives <- trimws(unlist(strsplit(receptor_chains[2], "\\|"))) + + # Validate the alternatives + if (length(relaxed_chain_alternatives) != 2) { + cli::cli_abort("Relaxed pairing syntax requires exactly 2 alternatives separated by '|'. Found {length(relaxed_chain_alternatives)} in '{receptor_chains[2]}'") + } + + # Check for empty alternatives + if (any(relaxed_chain_alternatives == "")) { + cli::cli_abort("Empty chain name found in '{receptor_chains[2]}'. Both alternatives must be valid chain names.") + } + + # Check for duplicate alternatives + if (length(unique(relaxed_chain_alternatives)) != length(relaxed_chain_alternatives)) { + cli::cli_abort("Duplicate chain names found in '{receptor_chains[2]}'. Alternatives must be different.") + } + + # Check that alternatives are different from the required chain + if (receptor_chains[1] %in% relaxed_chain_alternatives) { + cli::cli_abort("The required chain '{receptor_chains[1]}' cannot also be an alternative in '{receptor_chains[2]}'") + } + } else { + # Strict pairing - check for accidental spaces or typos + if (grepl("[\\s,;]", receptor_chains[2])) { + cli::cli_warn("Found potential separator characters in '{receptor_chains[2]}'. For relaxed pairing, use the pipe character '|' to separate alternatives (e.g., 'IGL|IGK')") + } + } + } + } + + TRUE +} + + +#' @rdname make_receptor_schema +#' @export +test_receptor_schema <- function(schema) { + isTRUE(tryCatch( + suppressWarnings(assert_receptor_schema(schema)), + error = function(...) FALSE + )) +} diff --git a/R/utils_seq.R b/R/utils_seq.R new file mode 100644 index 0000000..df2334f --- /dev/null +++ b/R/utils_seq.R @@ -0,0 +1,248 @@ +#' @title Create options for comparing receptor sequences +#' +#' @description +#' Create sequence comparison options for the `seq_options` argument of +#' [filter_immundata()] or [mutate_immundata()]. Use these options to compare a +#' sequence column with one or more reference sequences or patterns. +#' +#' @param query_col Name of the sequence column to compare, such as `"cdr3_aa"`. +#' @param patterns One or more reference sequences or regular-expression +#' patterns. +#' @param method Comparison method: `"exact"`, `"regex"`, `"lev"` +#' (Levenshtein distance), or `"hamm"` (Hamming distance). The default is +#' `"exact"`. +#' @param max_dist Maximum distance accepted by [filter_immundata()] when +#' `method = "lev"` or `method = "hamm"`. A value is required when filtering +#' with either distance method. This argument has no effect on +#' [mutate_immundata()], which reports every calculated distance. +#' @param name_type How result columns created by [mutate_immundata()] are named. +#' `"index"`, the default, creates short numbered names. `"pattern"` includes +#' the reference pattern in each name. This argument does not change which +#' receptors are kept by [filter_immundata()]. +#' +#' @return A named list for the `seq_options` argument of [filter_immundata()] or +#' [mutate_immundata()]. +#' +#' @seealso [filter_immundata()], [mutate_immundata()], [annotate_receptors()] +#' +#' @concept utils +#' @export +make_seq_options <- function(query_col, + patterns, + method = c("exact", "lev", "hamm", "regex"), + max_dist = NA, + name_type = c("index", "pattern")) { + checkmate::assert_character(query_col, len = 1) + checkmate::assert_character(patterns, min.len = 1) + + list( + query_col = query_col, + patterns = patterns, + method = match.arg(method), + max_dist = max_dist, + name_type = match.arg(name_type) + ) +} + +check_seq_options <- function(seq_options, mode = NULL) { + checkmate::assert_list(seq_options, null.ok = FALSE) + checkmate::assert_choice(mode, choices = c("filter", "mutate"), null.ok = FALSE) + + if (!is.null(seq_options$patterns) && + length(seq_options$patterns) > 0 && + !is.null(seq_options$query_col)) { + defaults <- list(method = "exact", max_dist = NA, name_type = "index") + + seq_options <- utils::modifyList(defaults, seq_options) + + seq_options$method <- match.arg(seq_options$method, c("exact", "regex", "lev", "hamm")) + + if (mode == "filter" && + is.na(seq_options$max_dist) && + seq_options$method %in% c("lev", "hamm")) { + cli::cli_abort("You passed `seq_options` to `filter`, but didn't provide `max_dist` for filtering. Either provide `max_dist` or use `left_join` to annotate receptors with distances to patterns.") + } + + seq_options + } else { + cli::cli_abort("Missing fields in `seq_options`, please use {.run immundata::make_seq_options()} to create the options") + } +} + +make_pattern_columns <- function(patterns, + col_prefix, + name_type = c("pattern", "index")) { + checkmate::assert_character(patterns, min.len = 1) + checkmate::assert_character(col_prefix, max.len = 1) + name_type <- match.arg(name_type) + + sapply(seq_along(patterns), function(p_index) { + p_seq <- patterns[[p_index]] + + if (name_type == "pattern") { + safe_name <- gsub("[^A-Za-z0-9]", "_", p_seq) # just in case + col_name_out <- paste0(col_prefix, safe_name) + } else if (name_type == "index") { + col_name_out <- paste0(col_prefix, p_index) + } else { + # TODO: what the heck + stop("!") + } + + col_name_out + }) +} + + +#' @keywords internal +annotate_tbl_distance <- function(tbl_data, + query_col, + patterns, + method = c("lev", "hamm"), + max_dist = NA, + name_type = c("pattern", "index")) { + checkmate::assert_character(query_col, len = 1) + checkmate::assert_character(patterns, min.len = 1) + + method <- match.arg(method) + name_type <- match.arg(name_type) + + uniq <- tbl_data |> + distinct(!!rlang::sym(query_col)) + + query_col_expr <- rlang::sym(query_col) + + if (method == "lev") { + col_prefix <- imd_schema("sim_lev") + } else if (method == "hamm") { + col_prefix <- imd_schema("sim_hamm") + } + + dist_cols <- make_pattern_columns( + patterns = patterns, + col_prefix = col_prefix, + name_type = name_type + ) + + # TODO: Optimize it via SQL instead of cycles - if it is even needed... + # TODO: lump together multiple patterns in batches + for (i in seq_along(patterns)) { + p <- patterns[[i]] + col_name_out <- dist_cols[i] + + # + # 1) Levenshtein distance + # + if (method == "lev") { + if (!is.na(max_dist)) { + len_p <- nchar(p) + uniq <- uniq |> + mutate( + {{ col_name_out }} := dplyr::if_else( + dd$length(!!query_col_expr) >= len_p - max_dist & + dd$length(!!query_col_expr) <= len_p + max_dist, + dd$levenshtein(!!query_col_expr, p), + NA_real_ + ) + ) + } else { + uniq <- uniq |> + mutate( + {{ col_name_out }} := dd$levenshtein(!!query_col_expr, p) + ) + } + } + + # + # 2) Hamming distance + # + else { + len_p <- nchar(p) + uniq <- uniq |> + mutate( + {{ col_name_out }} := dplyr::if_else( + dd$length(!!query_col_expr) == len_p, + dd$hamming(!!query_col_expr, p), + NA_real_ + ) + ) + } + } + + # + # TODO: benchmark 1 - distinct vs no distinct + # TODO: benchmark 2 - pre-optimize vs no optimize + + # TODO: fun experiment - compute for patterns, then filter out, then compute again, and so on. + # filter out -> filter out those who has <= max_dist (!) because we already found them and just need to store + + # TODO: Benchmarks + # 1) distinct vs non-distinct + # 2) pre-optimize vs no optimization for levenshtein + # 3) step-by-step filtering out "good" sequences + # 4) precompute sequence length before (!) any filtering, on data loading, and don't compute it here + + if (!is.na(max_dist)) { + within_max_dist <- lapply( + dist_cols, + function(col) rlang::expr(!!rlang::sym(col) <= !!max_dist) + ) |> + Reduce( + f = function(left, right) rlang::expr((!!left) | (!!right)) + ) + + uniq <- uniq |> + filter(!!within_max_dist) + } + + uniq |> + compute(name = basename(tempfile(pattern = "immundata_"))) +} + + +#' @keywords internal +annotate_tbl_regex <- function(tbl_data, + query_col, + patterns, + filter_out = FALSE, + name_type = c("index", "pattern")) { + checkmate::assert_character(query_col, len = 1) + checkmate::assert_character(patterns, min.len = 1) + checkmate::assert_logical(filter_out) + + name_type <- match.arg(name_type) + + uniq <- tbl_data |> distinct(!!rlang::sym(query_col)) + + col_prefix <- imd_schema("sim_regex") + + dist_cols <- make_pattern_columns( + patterns = patterns, + col_prefix = col_prefix, + name_type = name_type + ) + + # TODO: Optimize it via SQL instead of cycles - if it is even needed... + for (i in seq_along(patterns)) { + p <- patterns[[i]] + col_name_out <- dist_cols[i] + + # annotate with DuckDB regexp_matches() + uniq <- uniq |> + mutate(!!col_name_out := dd$regexp_matches(!!rlang::sym(query_col), p)) + } + + tbl_data <- tbl_data |> left_join(uniq, by = query_col) + if (filter_out) { + # TODO: need to replace it with if_else when it is available in duckplyr + sql_expr <- paste(dist_cols, collapse = " OR ") + + tbl_data |> + as_tbl() |> + filter(dbplyr::sql(sql_expr)) |> + as_duckdb_tibble() |> + compute() # TODO: We need a compute here because sometimes duckplyr can't find the table + } else { + tbl_data + } +} diff --git a/R/utils_test.R b/R/utils_test.R new file mode 100644 index 0000000..4c7111b --- /dev/null +++ b/R/utils_test.R @@ -0,0 +1,30 @@ +#' Get test datasets from `immundata` +#' @keywords internal +#' @export +get_test_idata <- function() { + manifest_path <- system.file( + "extdata/parquet", + "manifest.csv", + package = "immundata" + ) + manifest <- read_manifest(manifest_path) + + sample_files <- c( + system.file( + "extdata/parquet", + "sample_0_1k.parquet", + package = "immundata" + ), + system.file( + "extdata/parquet", + "sample_1k_2k.parquet", + package = "immundata" + ) + ) + read_repertoires( + path = sample_files, + schema = c("cdr3_aa", "v_call"), + manifest = manifest, + output_folder = tempfile() + ) +} diff --git a/R/zzz.R b/R/zzz.R index 5fc6986..44052e0 100644 --- a/R/zzz.R +++ b/R/zzz.R @@ -1,7 +1,13 @@ -.onAttach <- function(libname, pkgname) { - packageStartupMessage("Loading immundata version ", packageVersion(pkgname)) -} - -.onUnload <- function(libpath) { - message("Unloading immundata") +.onLoad <- function(libname, pkgname) { + # has_duckplyr <- requireNamespace("duckplyr", quietly = TRUE) + # has_duckdb <- requireNamespace("duckdb", quietly = TRUE) + # duckdb_ge_150 <- has_duckdb && + # utils::packageVersion("duckdb") >= base::package_version("1.5.0") + # + # if (has_duckplyr && duckdb_ge_150) { + # try( + # duckplyr::db_exec("SET disabled_optimizers = 'top_n_window_elimination'"), + # silent = TRUE + # ) + # } } diff --git a/README.md b/README.md index 2975e4a..6ddef9f 100644 --- a/README.md +++ b/README.md @@ -1,5 +1,5 @@
-

🦋 immundata – Data layer for large-scale multi-modal immune repertoires in R

+

🦋 immundata – Data layer for large-scale multi-modal immune repertoires in R

--- @@ -21,18 +21,24 @@ CRAN Downloads (last week) - + + Conda Version + + + Conda Total Downloads + + GitHub Issues

- Tutorials - | API reference | - Ecosystem + Tutorials ↗ | Publication (coming soon...)

@@ -91,7 +97,7 @@ It is the data-engineering backbone powered by [Arrow](https://arrow.apache.org/ - 🧬 [Workflow Explained](#-workflow-explained) - 💾 [Ingestion](#-ingestion) - [Load AIRR data](#load-airr-data) - - [Working with metadata table files](#working-with-metadata-table-files) + - [Working with manifest files](#working-with-manifest-files) - [Receptor schema](#receptor-schema) - [Repertoire schema](#repertoire-schema) - [Pre‑ and post‑processing strategies](#pre--and-post‑processing-strategies) @@ -173,8 +179,8 @@ Replace `system.file` calls with your local file paths to run the code on your d ```r library(immundata) -# Metadata table with additional sample-level information -md_path <- system.file("extdata/tsv", "metadata.tsv", package = "immundata") +# Manifest with additional sample-level information +manifest_path <- system.file("extdata/tsv", "manifest.csv", package = "immundata") # Two sample files samples <- c( @@ -182,13 +188,13 @@ samples <- c( system.file("extdata/tsv", "sample_1k_2k.tsv", package = "immundata") ) -# Read the metadata table -md <- read_metadata(md_path) +# Read the manifest +manifest <- read_manifest(manifest_path) -# Pass the file paths and the metadata table to the function to read the dataset into R +# Pass the file paths and the manifest to the function to read the dataset into R imdata <- read_repertoires(path = samples, schema = c("cdr3_aa", "v_call"), - metadata = md, + manifest = manifest, output_folder = "./immundata-quick-start") # Print the resultant object in the detailed yet manageable format @@ -238,7 +244,7 @@ Before we go into more details for each of the phase, there are three straightfo The function `agg_receptors()` lets you declare what *one receptor* means in your study. You choose a schema – perhaps "pair chains that share a barcode and have complementary α and β loci" or "group every IGH with whatever IGL shares the same CDR3 amino-acid sequence." The function re-aggregates the data and returns a new `ImmunData` object, so you keep the previous receptor definition intact; every receptor now has a stable identifier and can be traced back to its constituent chains and barcode. There is no need to touch the downstream pipeline – just change the input. - The function `agg_repertoires()` states how receptors should be bundled into biologically meaningful cohorts: all receptors from a biopsy, from a therapy responder, from a single-cell-defined cluster, or any combination of metadata columns. The result is a physical `idata$repertoires` table with basic statistics (numbers of chains, barcodes, and unique receptors), again preserving direct links to the receptors it aggregates. + The function `agg_repertoires()` states how receptors should be bundled into biologically meaningful cohorts: all receptors from a biopsy, from a therapy responder, from a single-cell-defined cluster, or any combination of manifest-derived annotation columns. The result is a physical `idata$repertoires` table with basic statistics (numbers of chains, barcodes, and unique receptors), again preserving direct links to the receptors it aggregates. Because these aggregation steps live in your pipeline rather than being buried inside helper functions, they deliver two major pay-offs: @@ -272,7 +278,7 @@ And now, let's dive into how you work with `immundata`. └───────┘ │ ▼ - read_metadata() ──── Read metadata + read_manifest() ──── Read manifest │ ▼ read_repertoires() ──┬─ Read repertoire files (!) @@ -299,11 +305,11 @@ Steps marked with `(!)` are non-optional. The goal of the **ingestion phase** is to turn a folder of AIRR-seq files into an immutable on-disk `ImmunData` dataset. - 1) **Read metadata:** + 1) **Read manifest:** - `read_metadata()` pulls in any sample- or donor-level information, such as therapy arm, HLA type, age, etc., and stores it in a data frame that we can pass to the main reading functions `read_repertoires`. Attaching this context early means every chain you read later already "knows" which patient or time-point it belongs to. + `read_manifest()` pulls in any sample- or donor-level information, such as therapy arm, HLA type, age, etc., and stores it in a data frame that we can pass to the main reading functions `read_repertoires`. Attaching this context early means every chain you read later already "knows" which patient or time-point it belongs to. - You can safely skip it if you don't have per-sample pr per-donor metadata. + You can safely skip it if you don't have per-sample or per-donor manifest annotations. 2) **Read repertoire files:** @@ -323,7 +329,7 @@ The goal of the **ingestion phase** is to turn a folder of AIRR-seq files into a 5) **Aggregate repertoires #1:** - If you already know how to group chains into receptors, perhaps by `"Sample"` or `"Donor"` columns from the metadata, you can pass `repertoire_schema = c("Sample")` to `read_repertoires()`. Otherwise, skip and define repertoires later (common in single-cell workflows where you need cluster labels first). + If you already know how to group chains into receptors, perhaps by `"Sample"` or `"Donor"` columns from the manifest, you can pass `repertoire_schema = c("Sample")` to `read_repertoires()`. Otherwise, skip and define repertoires later (common in single-cell workflows where you need cluster labels first). 3) **Write data on disk:** @@ -485,31 +491,31 @@ Transformation is a loop of annotation → modification and computation → visu Behind the scenes, `read_repertoires()` expands the glob with `Sys.glob(...)`, merges the data, and produces a single `ImmunData`. - 4. **Use a metadata file:** + 4. **Use a manifest file:** Sometimes you need more control over the data source (e.g. consistent sample naming, extra columns). In that case: - 1. **Load metadata** with `read_metadata()`. + 1. **Load a manifest** with `read_manifest()`. - 2. **Pass** the resulting data frame to `read_repertoires(path = "", ..., metadata = md_table)`. Mind the `""` string we pass to the function. It indicates that we should take file paths from the input metadata table. + 2. **Pass** the resulting data frame to `read_repertoires(path = "", ..., manifest = manifest_table)`. Mind the `""` string we pass to the function. It indicates that we should take file paths from the input manifest. An example code: ```r library(immundata) - md_path <- system.file("extdata/tsv", "metadata.tsv", package = "immundata") + manifest_path <- system.file("extdata/tsv", "manifest.csv", package = "immundata") - md_table <- read_metadata(md_path) + manifest_table <- read_manifest(manifest_path) - print(md_table) + print(manifest_table) ``` ``` - # The column "File" stores the file paths. If you have a different column name - # for files, use the `metadata_file_col = ""` argument. + # The column "file" stores the file paths. If you have a different column name + # for files, use the `manifest_file_col = ""` argument. # A tibble: 2 × 5 - File Therapy Response Prefix filename + file Therapy Response Prefix imd_filename 1 /.../immundata-/inst/extd… ICI FR S1_ /Users/… 2 /.../immundata-/inst/extd… CAR-T PR S2_ /Users/… @@ -517,19 +523,19 @@ Transformation is a loop of annotation → modification and computation → visu ```r idata <- read_repertoires( - path = "", - metadata = md_table, + path = "", + manifest = manifest_table, schema = c("cdr3_aa", "v_call") ) print(idata) ``` - This approach **unifies** sample-level metadata (e.g. donor ID, timepoint) with your repertoire data inside a single `ImmunData`. + This approach **unifies** sample-level annotations (e.g. donor ID, timepoint) with your repertoire data inside a single `ImmunData`. - You can pass the metadata table separately along with the list of files as we did in the previous examples without the "" directive, but in that case you would need to check the correctness of all filepaths by yourself. Which could be quite cumbersome, to say the least. + You can pass the manifest separately along with the list of files as we did in the previous examples without the "" directive, but in that case you would need to check the correctness of all filepaths by yourself. Which could be quite cumbersome, to say the least. - The more information on how to work with metadata files, please read the next section. + For more information on how to work with manifest files, please read the next section. 5. **Convert from `immunarch` lists:** @@ -549,43 +555,43 @@ Transformation is a loop of annotation → modification and computation → visu print(idata) ``` -### Working with metadata table files +### Working with manifest files -Metadata tables store the sample-level information. When `immundata` loads the metadata, it annotates every receptor from a given sample (or file) with the corresponding metadata fields. For example, if a sample has "Therapy" = "CAR‑T", all receptors from that sample receive the same "Therapy" value. You can then aggregate receptors by donor, tissue, or any other field and run your analysis on those repertoires (see the next sections for aggregations). +Manifest files store repertoire file paths plus sample-level information. When `immundata` loads a manifest, it annotates every receptor from a given sample (or file) with the corresponding manifest fields. For example, if a sample has "Therapy" = "CAR-T", all receptors from that sample receive the same "Therapy" value. You can then aggregate receptors by donor, tissue, or any other field and run your analysis on those repertoires (see the next sections for aggregations). > [!WARNING] -> In the current version, "metadata" and "repertoire schema" is the same, meaning you can't get -> a metadata field to `idata$repertoires` if you haven't define repertoires using that field. +> In the current version, manifest annotations and "repertoire schema" are tightly linked, meaning you can't get +> a manifest field to `idata$repertoires` if you haven't define repertoires using that field. > I will implement it in the next versions; for now, please consider using `dplyr::left_join` to -> merge metadata and the repertoires table together. +> merge manifest annotations and the repertoires table together. ```r library(immundata) -md_path <- system.file("extdata/tsv", "metadata.tsv", package = "immundata") -md_table <- read_metadata(md_path) +manifest_path <- system.file("extdata/tsv", "manifest.csv", package = "immundata") +manifest_table <- read_manifest(manifest_path) ``` ``` Rows: 2 Columns: 4 ── Column specification ───────────────────────────────────────────────────────── -Delimiter: "\t" -chr (4): File, Therapy, Response, Prefix +Delimiter: "," +chr (4): file, Therapy, Response, Prefix ℹ Use `spec()` to retrieve the full column specification for this data. ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message. -ℹ Found 2/2 repertoire files from the metadata on the disk -✔ Metadata parsed successfully +ℹ Found 2/2 repertoire files from the manifest on disk +✔ Manifest parsed successfully ``` ```r -print(md_table) +print(manifest_table) ``` ``` # A tibble: 2 × 5 - File Therapy Response Prefix filename + file Therapy Response Prefix imd_filename 1 /.../immundata-/inst/extd… ICI FR S1_ /Users/… 2 /.../immundata-/inst/extd… CAR-T PR S2_ /Users/… @@ -686,11 +692,12 @@ schema <- make_receptor_schema( Cheat-sheet for arguments to `read_repertoires`: -| Situation | `barcode_col` | `locus_col` | `umi_col` | `chains` | -| ---------------------------------------- | ------------- | ----------- | --------- | ---------------- | -| Bulk data, no locus filtering | no | no | no | omit / `NULL` | -| Analyse TRA only | **yes**¹ | **yes** | no | `"TRA"` | -| Pair TRA+TRB, pick best chain per cell | **yes** | **yes** | **yes** | `c("TRA","TRB")` | +| Situation | `count_col` |`barcode_col` | `locus_col` | `umi_col` | `schema$chains` | +| ---------------------------------------- | ----------- | ------------ | ----------- | --------- | ---------------- | +| Bulk data without counts, no locus | no | no | no | no | no | +| Bulk data with counts, no locus | *yes* | no | no | no | no | +| Analyse TRA only | no | **yes**¹ | **yes** | no | `"TRA"` | +| Pair TRA+TRB, pick best chain per cell | no | **yes** | **yes** | **yes** | `c("TRA","TRB")` | ¹ If you pass barcodes, they're stored but used for counting only. @@ -700,7 +707,7 @@ To compute repertoire‑level statistics such as gene‑segment usage, the Jacca Just like with receptors, you can pass a schema – a character vector of column names – to specify how receptors are grouped into repertoires. -For the bulk data, usually, you rely on the metadata table. It could be useful when you want to aggregate together receptors from the same donor or tissue, and then analyse it. Or you may want to filter out non-responders to analyse the responders only. +For the bulk data, usually, you rely on the manifest. It could be useful when you want to aggregate together receptors from the same donor or tissue, and then analyse it. Or you may want to filter out non-responders to analyse the responders only. > [!NOTE] > Don't confuse grouping of immune repertoires with grouping in plots. @@ -717,14 +724,14 @@ The true power of regrouping repertoires opens up when you work with single-cell library(immundata) inp_file <- system.file("extdata/single_cell", "lt6.csv.gz", package = "immundata") -md_file <- system.file("extdata/single_cell", "metadata.tsv", package = "immundata") -md_table <- read_metadata(md_file) +manifest_file <- system.file("extdata/single_cell", "manifest.csv", package = "immundata") +manifest_table <- read_manifest(manifest_file) schema <- make_receptor_schema(features = c("cdr3", "v_call"), chains = c("TRA", "TRB")) idata <- read_repertoires( path = inp_file, schema = schema, - metadata = md_table, + manifest = manifest_table, barcode_col = "barcode", # required for pairing locus_col = "locus", # column that says "TRA" / "TRB" umi_col = "umis", # choose chain with max UMIs per locus @@ -764,14 +771,23 @@ print(idata) 2. **Barcode prefix** - Provide a column named "Prefix" to the metadata so `make_default_postprocessing()` can automatically add this prefix to barcodes to make barcodes unique in your resultant dataset. + Provide a column named "Prefix" to the manifest so `make_default_postprocessing()` can automatically add this prefix to barcodes to make barcodes unique in your resultant dataset. ### Managing the output and intermediate ImmunData files > [!CAUTION] > 🚧 Under construction. 🚧 -By default, `read_repertoires()` writes the created Parquet files into a directory named `immundata_`. Consider passing `output_folder` to `read_repertoires()` if you want to specify the output path. +By default, `read_repertoires()` writes the persistent ImmunData snapshot into +a directory named `immundata-`. Consider passing +`output_folder` to choose another location. + +For CSV, TSV, and compressed text input, `read_repertoires()` first combines +the source files into one temporary Parquet file. The intermediate is created +under `tempdir()` and deleted when the function exits. Pass +`prematerialize_folder` to use another temporary-storage directory, or set +`prematerialize = FALSE` to disable this step. Original source paths remain in +`imd_filename` and ingestion provenance. ### Writing ImmunData objects on disk @@ -787,7 +803,7 @@ Why you might need it - to save intermediate files, e.g., after computing levens - `ImmunData$receptors` – a virtual table created on demand from `$annotations`. One row per receptor as defined by your `$schema_receptor`; guaranteed to have the stable key `imd_receptor_id`. This is the aggregated view into your dataset, meaning that all fields from receptor features (cdr3, v_call) are unique with respect to row, i.e., each row is unique. -- `ImmunData$annotations` – the main table that holds all the data. One row per chain (or per cell barcode in case of single-chained data). Holds every AIRR field (cdr3, v_call, umis, etc.) plus any metadata you imported (sample_id, tissue, distances to patterns). +- `ImmunData$annotations` – the main table that holds all the data. One row per chain (or per cell barcode in case of single-chained data). Holds every AIRR field (cdr3, v_call, umis, etc.) plus any manifest annotations you imported (sample_id, tissue, distances to patterns). - `ImmunData$repertoires` – a physical table produced by agg_repertoires(). Each row is a repertoire (sample, donor, cluster) and carries pre-computed counts: number of receptors, barcodes, chains. @@ -805,14 +821,14 @@ Example: library(immundata) inp_files <- paste0(system.file("extdata/single_cell", "", package = "immundata"), "/*.csv.gz") -md_file <- system.file("extdata/single_cell", "metadata.tsv", package = "immundata") -md_table <- read_metadata(md_file) +manifest_file <- system.file("extdata/single_cell", "manifest.csv", package = "immundata") +manifest_table <- read_manifest(manifest_file) cells_file <- system.file("extdata/single_cell", "cells.tsv.gz", package = "immundata") cells <- readr::read_tsv(cells_file) schema <- make_receptor_schema(features = c("cdr3", "v_call"), chains = c("TRB")) -idata <- read_repertoires(path = inp_files, schema = schema, metadata = md_table, barcode_col = "barcode", locus_col = "locus", umi_col = "umis", preprocess = make_default_preprocessing("10x"), repertoire_schema = "Tissue") +idata <- read_repertoires(path = inp_files, schema = schema, manifest = manifest_table, barcode_col = "barcode", locus_col = "locus", umi_col = "umis", preprocess = make_default_preprocessing("10x"), repertoire_schema = "Tissue") print(idata) ``` @@ -843,7 +859,7 @@ Printed ImmunData `idata`: ── Annotations: ── # A duckplyr data frame: 23 variables - barcode locus v_call d_call j_call c_gene productive cdr3 cdr3_nt reads umis filename imd_barcode + barcode locus v_call d_call j_call c_gene productive cdr3 cdr3_nt reads umis imd_filename imd_barcode 1 AAACCTGA… TRB TRBV2 None TRBJ2… TRBC2 True CASS… TGTGCC… 13736 11 /Users/… LB6_AAACCT… 2 AAACCTGC… TRB TRBV30 TRBD1 TRBJ2… TRBC2 True CAWS… TGTGCC… 4062 5 /Users/… LB6_AAACCT… @@ -856,7 +872,7 @@ Printed ImmunData `idata`: 9 AAACGGGA… TRB TRBV7… TRBD2 TRBJ1… TRBC1 True CASS… TGTGCC… 4956 4 /Users/… LB6_AAACGG… 10 AAACGGGA… TRB TRBV2 TRBD1 TRBJ2… TRBC2 True CASP… TGTGCC… 5625 4 /Users/… LB6_AAACGG… # ℹ more rows -# ℹ 10 more variables: imd_chain_id , imd_receptor_id , imd_n_chains , File , +# ℹ 10 more variables: imd_chain_id , imd_receptor_id , imd_n_chains , file , # Tissue , Prefix , imd_count , imd_repertoire_id , imd_proportion , # n_repertoires # ℹ Use `print(n = ...)` to see more rows @@ -895,14 +911,14 @@ Before running the code in the following subsections, execute the code below. Mi library(immundata) inp_files <- paste0(system.file("extdata/single_cell", "", package = "immundata"), "/*.csv.gz") -md_file <- system.file("extdata/single_cell", "metadata.tsv", package = "immundata") -md_table <- read_metadata(md_file) +manifest_file <- system.file("extdata/single_cell", "manifest.csv", package = "immundata") +manifest_table <- read_manifest(manifest_file) cells_file <- system.file("extdata/single_cell", "cells.tsv.gz", package = "immundata") cells <- readr::read_tsv(cells_file) schema <- make_receptor_schema(features = c("cdr3", "v_call"), chains = c("TRB")) -idata <- read_repertoires(path = inp_files, schema = schema, metadata = md_table, barcode_col = "barcode", locus_col = "locus", umi_col = "umis", preprocess = make_default_preprocessing("10x"), repertoire_schema = "Tissue") +idata <- read_repertoires(path = inp_files, schema = schema, manifest = manifest_table, barcode_col = "barcode", locus_col = "locus", umi_col = "umis", preprocess = make_default_preprocessing("10x"), repertoire_schema = "Tissue") ``` ### Filter @@ -1019,6 +1035,89 @@ The key functions for this are `mutate` (`dplyr`-compatible) / `mutate_immundata ```r idata |> mutate(found_pattern = if_else(cdr3 == "CASSVHPQYF", 1, 0)) ``` + + 4. **Add statistics for groups without removing receptor rows** + + Use `.by` to calculate separately for each group. Unlike + `summarise()`, `mutate()` keeps every annotation row and repeats the + group result for rows in the same group. + + ```r + example_idata <- get_test_idata() + + response_stats <- example_idata |> + mutate( + response_n_rows = n(), + response_n_receptors = n_distinct(imd_receptor_id), + .by = Response + ) + + response_stats |> + collect() |> + distinct(Response, response_n_rows, response_n_receptors) |> + arrange(Response) + # Response response_n_rows response_n_receptors + # FR 955 871 + # PR 947 867 + ``` + + `n()` counts annotation rows. Use + `n_distinct(imd_receptor_id)` when you need the number of different + receptors. + + 5. **Split complex grouped calculations when needed** + + A grouped `mutate()` can calculate a different value for every row: + + ```r + response_centered <- example_idata |> + mutate( + centered_counts = counts - mean(counts, na.rm = TRUE), + .by = Response + ) + ``` + + Some group statistics, including `n_distinct()`, use an automatic + summary-and-join calculation for large datasets. Do not combine such a + statistic with a row-level calculation in the same call: + + ```r + # This call is not supported: + example_idata |> + mutate( + centered_counts = counts - mean(counts, na.rm = TRUE), + response_n_receptors = n_distinct(imd_receptor_id), + .by = Response + ) + ``` + + Use two calls instead. The calculations remain lazy in DuckDB: + + ```r + response_details <- example_idata |> + mutate( + centered_counts = counts - mean(counts, na.rm = TRUE), + .by = Response + ) |> + mutate( + response_n_receptors = n_distinct(imd_receptor_id), + .by = Response + ) + ``` + + Use a second call also when the next calculation uses a statistic that + was just created: + + ```r + response_details <- example_idata |> + mutate( + response_n_receptors = n_distinct(imd_receptor_id), + .by = Response + ) |> + mutate( + twice_response_n_receptors = response_n_receptors * 2 + ) + ``` --- ## 📈 Analysis @@ -1135,8 +1234,8 @@ S3 methods etc. By design, **`immundata`** data-loading pipeline is **three** steps, rather than one giant function. This promotes modularity, easier debugging, and flexible usage: -1. **(Optionally) Load the metadata** via `read_metadata()`. - - This ensures your metadata has the correct file paths, absolute or relative. +1. **(Optionally) Load the manifest** via `read_manifest()`. + - This ensures your manifest has the correct file paths, absolute or relative. 2. **Load the repertoire files** from disk via `read_repertoires()`. - This function unifies your data (be it 1 file or 100 files) and **outputs** a Parquet file: - **`annotations.parquet`** (cell-level data, sample metadata, etc.) @@ -1148,7 +1247,7 @@ By design, **`immundata`** data-loading pipeline is **three** steps, rather than Why split it up? -- **Modularity**: If something breaks, you can debug whether it's in metadata parsing or the actual repertoire table creation. +- **Modularity**: If something breaks, you can debug whether it's in manifest parsing or the actual repertoire table creation. - **Reusability**: It is straightforward to share one folder with two `immundata` files. - **Performance**: Once your data is in `immundata` format, you can load it in future sessions in **constant time** without merging or parsing again. @@ -1205,11 +1304,22 @@ If you are looking for prioritized support and setting up your data pipelines, c ## 🤔 FAQ -1. **Q: Why all the function names or ImmunData fields are so long? I want to write `idata$rec` instead of `idata$receptors`.** +1. **Q: Why did `metadata.tsv` become `manifest.csv`?** + + A: `metadata.tsv` was an old input-table format for listing repertoire files and per-sample annotations. It was confusing because the word "metadata" already means several different things in `immundata` workflows: + + - input file manifests: a table that tells `read_repertoires()` which repertoire files to read and what sample-level annotations to attach; + - biological or clinical metadata: donor, tissue, therapy, response, HLA, time point, and similar fields; + - cell-level metadata: barcode-level annotations from single-cell or spatial objects, such as cluster labels or gene expression summaries; + - ImmunData snapshot metadata: the internal `metadata.json` file that stores schemas, provenance, lineage, package version, and snapshot IDs. + + Calling the input manifest `metadata.tsv` mixed these concepts together. We now use the industry-standard manifest pattern: a `manifest.csv` file with a `file` column plus any additional annotation columns. In code, use `read_manifest()` and `read_repertoires(path = "", manifest = manifest_table)`. Snapshot `metadata.json` remains unchanged because it is metadata about the saved `ImmunData` object, not a list of input repertoire files. + +2. **Q: Why all the function names or ImmunData fields are so long? I want to write `idata$rec` instead of `idata$receptors`.** A: Two major reasons – improving the code readability and motivation to leverage the autocomplete tools. Please consider using `tab` for leveraging autocomplete. It accelerates things x10-20. -2. **Q: How does `immundata` works under the hood, in simpler terms?** +3. **Q: How does `immundata` works under the hood, in simpler terms?** A: Picture a three-layer sandwich: @@ -1226,11 +1336,11 @@ If you are looking for prioritized support and setting up your data pipelines, c 2. [DuckDB – embedded analytical database](https://duckdb.org/) 3. [duckplyr – API/implementation details](https://duckplyr.tidyverse.org/index.html) -3. **Q: Why do you need to create Parquet files with receptors and annotations?** +4. **Q: Why do you need to create Parquet files with receptors and annotations?** A: Those are intermediate files, optimized for future data operations, and working with them significantly accelerates `immundata`. I will post a benchmark soon. -4. **Q: Why does `immundata` support only the AIRR standard?!** +5. **Q: Why does `immundata` support only the AIRR standard?!** A: The short answer is because a single, stable schema beats a zoo of drifting ones. @@ -1240,7 +1350,7 @@ If you are looking for prioritized support and setting up your data pipelines, c `immundata` does not and will not explicitly support other formats. This is both a practical stance and communication of crucial values, put into `immundata` as part of a broader ecosystem of AIRR tools. The domain is already too complex, and we need to work together to make this complexity manageable. A healthy ecosystem is not the same as a complex ecosystem. -5. **Q: Why is it so complex? Why do we need to use `dplyr` instead of plain R?** +6. **Q: Why is it so complex? Why do we need to use `dplyr` instead of plain R?** A: The short answer is: @@ -1249,7 +1359,7 @@ If you are looking for prioritized support and setting up your data pipelines, c - better data skills thanks to thinking in immutable transformations, - in most cases you don't really need complex transformations, so we can optimize 95% of all AIRR data operations behind the scenes. -6. **Q: How do I use `dplyr` operations that `duckplyr` doesn't support yet?** +7. **Q: How do I use `dplyr` operations that `duckplyr` doesn't support yet?** A: Let's consider several use cases. diff --git a/_pkgdown.yml b/_pkgdown.yml deleted file mode 100644 index 513a4d9..0000000 --- a/_pkgdown.yml +++ /dev/null @@ -1,54 +0,0 @@ -url: https://immunomind.github.io/immundata/ - -template: - bootstrap: 5 - bslib: - bootswatch: flatly - pkgdown-nav-height: 100px - params: - ganalytics: G-YHL2JHZTFV - -authors: - Vadim I. Nazarov: - href: https://www.linkedin.com/in/vdnaz - -navbar: - structure: - left: [tutorial, migration, reference] - right: [search, lightswitch, im_link, github] - components: - tutorial: - text: "" - href: ... - migration: - text: "" - href: ... - im_link: - text: ImmunoMind - href: https://immunomind.com - github: - icon: fa-github - href: https://github.com/immunomind/immundata - aria-label: GitHub - -reference: -- title: "Ingestion" - contents: - - has_concept("ingestion") - - has_concept("processing") - - has_concept("aggregation") -- title: "ImmunData object" - contents: - - has_concept("core_immundata") -- title: "Transformation" - contents: - - has_concept("annotation") - - has_concept("filtering") - - has_concept("mutation") - - has_concept("operations") -- title: "Utils" - contents: - - has_concept("utils") -- title: "Schema" - contents: - - has_concept("schema") diff --git a/altdoc/mkdocs.yml b/altdoc/mkdocs.yml new file mode 100644 index 0000000..018a477 --- /dev/null +++ b/altdoc/mkdocs.yml @@ -0,0 +1,129 @@ +site_name: immundata +site_description: A unified data layer for large-scale single-cell, spatial, and bulk immunomics in R +site_url: https://immunomind.github.io/immundata/ +repo_url: https://github.com/immunomind/immundata +repo_name: immunomind/immundata + +theme: + name: material + icon: + logo: material/dna + repo: fontawesome/brands/github + font: + text: Roboto + code: Roboto Mono + palette: + - media: "(prefers-color-scheme: light)" + scheme: default + toggle: + icon: material/weather-night + name: Switch to dark mode + - media: "(prefers-color-scheme: dark)" + scheme: slate + toggle: + icon: material/weather-sunny + name: Switch to light mode + features: + - navigation.tabs + - navigation.sections + - navigation.top + - content.code.copy + - toc.follow + +plugins: + - search: + separator: '[\s\-\._]+' + - redirects: + redirect_maps: + "reference/ImmunData.md": "man/ImmunData.md" + "reference/agg_receptors.md": "man/agg_receptors.md" + "reference/agg_repertoires.md": "man/agg_repertoires.md" + "reference/agg_strata.md": "man/agg_strata.md" + "reference/annotate_anndata.md": "man/annotate_anndata.md" + "reference/annotate_immundata.md": "man/annotate_immundata.md" + "reference/annotate_seurat.md": "man/annotate_seurat.md" + "reference/collect.ImmunData.md": "man/collect.ImmunData.md" + "reference/compute.ImmunData.md": "man/compute.ImmunData.md" + "reference/count.ImmunData.md": "man/count.ImmunData.md" + "reference/dimnames-set-.ImmunData.md": "man/dimnames-set-.ImmunData.md" + "reference/dimnames.ImmunData.md": "man/dimnames.ImmunData.md" + "reference/downsample_immundata.md": "man/downsample_immundata.md" + "reference/filter_immundata.md": "man/filter_immundata.md" + "reference/from_immunarch.md": "man/from_immunarch.md" + "reference/imd_schema.md": "man/imd_schema.md" + "reference/make_receptor_schema.md": "man/make_receptor_schema.md" + "reference/make_seq_options.md": "man/make_seq_options.md" + "reference/mutate_immundata.md": "man/mutate_immundata.md" + "reference/names-set-.ImmunData.md": "man/names-set-.ImmunData.md" + "reference/preprocess_postprocess.md": "man/preprocess_postprocess.md" + "reference/read_immundata.md": "man/read_immundata.md" + "reference/read_manifest.md": "man/read_manifest.md" + "reference/read_repertoires.md": "man/read_repertoires.md" + "reference/rename_strata.md": "man/rename_strata.md" + "reference/write_immundata.md": "man/write_immundata.md" + +markdown_extensions: + - admonition + - attr_list + - footnotes + - mdx_truly_sane_lists + - pymdownx.details + - pymdownx.highlight: + anchor_linenums: true + - pymdownx.superfences + - toc: + permalink: true + toc_depth: 3 + +extra_css: + - stylesheets/extra.css + +extra: + analytics: + provider: google + property: G-YHL2JHZTFV + social: + - icon: fontawesome/brands/github + link: https://github.com/immunomind/immundata + - icon: fontawesome/solid/globe + link: https://immunomind.com + +use_directory_urls: false + +nav: + - Home: README.md + - News: NEWS.md + - Reference: + - Overview: reference/index.md + - Ingestion: + - read_repertoires: man/read_repertoires.md + - read_manifest: man/read_manifest.md + - Pre- and post-processing: man/preprocess_postprocess.md + - from_immunarch: man/from_immunarch.md + - read_immundata: man/read_immundata.md + - write_immundata: man/write_immundata.md + - Data model and aggregation: + - ImmunData: man/ImmunData.md + - make_receptor_schema: man/make_receptor_schema.md + - agg_receptors: man/agg_receptors.md + - agg_repertoires: man/agg_repertoires.md + - agg_strata: man/agg_strata.md + - rename_strata: man/rename_strata.md + - Transformation: + - filter_immundata: man/filter_immundata.md + - downsample_immundata: man/downsample_immundata.md + - annotate_immundata: man/annotate_immundata.md + - annotate_seurat: man/annotate_seurat.md + - annotate_anndata: man/annotate_anndata.md + - mutate_immundata: man/mutate_immundata.md + - Computation: + - count: man/count.ImmunData.md + - compute: man/compute.ImmunData.md + - collect: man/collect.ImmunData.md + - Utilities: + - Schema helpers: man/imd_schema.md + - Sequence options: man/make_seq_options.md + - dimnames: man/dimnames.ImmunData.md + - "dimnames<-": man/dimnames-set-.ImmunData.md + - "names<-": man/names-set-.ImmunData.md + - Tutorials ↗: https://immunomind.github.io/docs/tutorials/single_cell/ diff --git a/altdoc/pkgdown.yml b/altdoc/pkgdown.yml new file mode 100644 index 0000000..cf30d2c --- /dev/null +++ b/altdoc/pkgdown.yml @@ -0,0 +1,8 @@ +altdoc: 0.7.3 +pandoc: 3.7.0.2 +pkgdown: 2.1.3 +pkgdown_sha: ~ +last_built: 2026-08-13T12:34:52+0000 +urls: + reference: https://immunomind.github.io/immundata/man + article: https://immunomind.github.io/immundata/vignettes diff --git a/altdoc/preamble_man_qmd.yml b/altdoc/preamble_man_qmd.yml new file mode 100644 index 0000000..f7f071b --- /dev/null +++ b/altdoc/preamble_man_qmd.yml @@ -0,0 +1,8 @@ +--- +format: + md: + prefer-html: true +knitr: + opts_chunk: + comment: "#>" +--- diff --git a/altdoc/preamble_vignettes_qmd.yml b/altdoc/preamble_vignettes_qmd.yml new file mode 100644 index 0000000..7c75bd1 --- /dev/null +++ b/altdoc/preamble_vignettes_qmd.yml @@ -0,0 +1,9 @@ +--- +format: + md: + prefer-html: true +default-image-extension: "" +knitr: + opts_chunk: + comment: "#>" +--- diff --git a/altdoc/preamble_vignettes_rmd.yml b/altdoc/preamble_vignettes_rmd.yml new file mode 100644 index 0000000..b5c2f09 --- /dev/null +++ b/altdoc/preamble_vignettes_rmd.yml @@ -0,0 +1,7 @@ +--- +always_allow_html: true +default-image-extension: "" +knitr: + opts_chunk: + comment: "#>" +--- diff --git a/altdoc/reference/index.md b/altdoc/reference/index.md new file mode 100644 index 0000000..a8d0058 --- /dev/null +++ b/altdoc/reference/index.md @@ -0,0 +1,15 @@ +# Reference + +The reference documents the public `immundata` API. Start with +[`read_repertoires()`](../man/read_repertoires.md) to ingest repertoire files, +or [`ImmunData`](../man/ImmunData.md) for the package's core data structure. + +The API is organized around the lifecycle of an immune-repertoire dataset: + +- **Ingestion** reads raw repertoire files, manifests, and saved datasets. +- **Data model and aggregation** defines receptors, repertoires, and strata. +- **Transformation** filters, annotates, and modifies an `ImmunData` object. +- **Computation** materializes or summarizes results when needed. + +Use the navigation on the left or search at the top of the page to find a +specific function. diff --git a/altdoc/requirements.txt b/altdoc/requirements.txt new file mode 100644 index 0000000..3933459 --- /dev/null +++ b/altdoc/requirements.txt @@ -0,0 +1,4 @@ +mkdocs>=1.6,<2 +mkdocs-material>=9.5,<10 +mkdocs-redirects>=1.2,<2 +mdx-truly-sane-lists>=1.3,<2 diff --git a/altdoc/stylesheets/extra.css b/altdoc/stylesheets/extra.css new file mode 100644 index 0000000..c5843e0 --- /dev/null +++ b/altdoc/stylesheets/extra.css @@ -0,0 +1,40 @@ +/* Keep long R signatures and output readable without distorting the page. */ +pre.r-example { + display: block; + margin: 0 0 1rem; + overflow: auto; + border: 1px solid var(--md-default-fg-color--lightest); + border-radius: 0.35rem; + font-size: 0.72rem; +} + +.md-typeset pre > code { + font-size: 0.73rem; +} + +.md-typeset code { + font-size: inherit; +} + +.md-typeset h1 > code, +.md-typeset h2 > code, +.md-typeset h3 > code { + font-size: inherit; +} + +.md-typeset__table { + width: 100%; + overflow-x: auto; +} + +[data-md-color-scheme="default"] { + --md-primary-fg-color: #243447; + --md-accent-fg-color: #d97706; + --md-typeset-a-color: #087f8c; +} + +[data-md-color-scheme="slate"] { + --md-primary-fg-color: #182433; + --md-accent-fg-color: #f59e0b; + --md-typeset-a-color: #67c7d0; +} diff --git a/inst/extdata/ig/multiple_ig_loci.tsv.gz b/inst/extdata/ig/multiple_ig_loci.tsv.gz new file mode 100644 index 0000000..a7abd83 Binary files /dev/null and b/inst/extdata/ig/multiple_ig_loci.tsv.gz differ diff --git a/inst/extdata/parquet/manifest.csv b/inst/extdata/parquet/manifest.csv new file mode 100644 index 0000000..99029f8 --- /dev/null +++ b/inst/extdata/parquet/manifest.csv @@ -0,0 +1,3 @@ +file,Therapy,Response,Prefix +sample_0_1k.parquet,ICI,FR,S1_ +sample_1k_2k.parquet,CAR-T,PR,S2_ diff --git a/inst/extdata/parquet/sample_0_1k.parquet b/inst/extdata/parquet/sample_0_1k.parquet new file mode 100644 index 0000000..09e2971 Binary files /dev/null and b/inst/extdata/parquet/sample_0_1k.parquet differ diff --git a/inst/extdata/parquet/sample_1k_2k.parquet b/inst/extdata/parquet/sample_1k_2k.parquet new file mode 100644 index 0000000..e0d8952 Binary files /dev/null and b/inst/extdata/parquet/sample_1k_2k.parquet differ diff --git a/inst/extdata/single_cell/manifest.csv b/inst/extdata/single_cell/manifest.csv new file mode 100644 index 0000000..0e004c8 --- /dev/null +++ b/inst/extdata/single_cell/manifest.csv @@ -0,0 +1,4 @@ +file,Tissue,Prefix +lb6.csv.gz,Blood,LB6_ +ln6.csv.gz,Normal,LN6_ +lt6.csv.gz,Tumor,LT6_ diff --git a/inst/extdata/single_cell/metadata.tsv b/inst/extdata/single_cell/metadata.tsv deleted file mode 100644 index 7db5e57..0000000 --- a/inst/extdata/single_cell/metadata.tsv +++ /dev/null @@ -1,4 +0,0 @@ -File Tissue Prefix -lb6.csv.gz Blood LB6_ -ln6.csv.gz Normal LN6_ -lt6.csv.gz Tumor LT6_ diff --git a/inst/extdata/tsv/immundata-sample_0_1k/annotations.parquet b/inst/extdata/tsv/immundata-sample_0_1k/annotations.parquet index 13d437e..fd14685 100644 Binary files a/inst/extdata/tsv/immundata-sample_0_1k/annotations.parquet and b/inst/extdata/tsv/immundata-sample_0_1k/annotations.parquet differ diff --git a/inst/extdata/tsv/immundata-sample_0_1k/metadata.json b/inst/extdata/tsv/immundata-sample_0_1k/metadata.json index 6d54549..98a169f 100644 --- a/inst/extdata/tsv/immundata-sample_0_1k/metadata.json +++ b/inst/extdata/tsv/immundata-sample_0_1k/metadata.json @@ -1 +1,92 @@ -{"version":["0.0.3"],"receptor_schema":{"features":["cdr3_aa","v_call"],"chains":{}},"repertoire_schema":{}} +{ + "format_version": 2, + "package_version": "0.0.6", + "schema_receptor": { + "features": ["cdr3_aa", "v_call"], + "chains": null + }, + "schema_repertoire": null, + "producer": { + "function": "read_repertoires" + }, + "snapshot_id": "imd_20260403T110305Z_ctrax314", + "lineage": [ + { + "event": "ingestion", + "created_at": "2026-04-03T11:03:05Z", + "snapshot_id": "imd_20260403T110305Z_ctrax314", + "producer": { + "function": "read_repertoires" + }, + "inputs": { + "files": "/Users/vdn/Projects/immundata-rlang/inst/extdata/tsv/sample_0_1k.tsv", + "metadata_joined": false, + "enforce_schema": true + }, + "args": { + "barcode_col": null, + "count_col": null, + "locus_col": null, + "umi_col": null, + "metadata_file_col": "File" + }, + "column_lineage": { + "renamed": { + "requested": ["v_gene", "d_gene", "j_gene", "d_gene", "chain"], + "applied": [], + "not_found": ["v_gene", "d_gene", "j_gene", "d_gene", "chain"] + }, + "dropped": { + "applied": [] + } + }, + "pipeline": { + "preprocess": null, + "postprocess": "prefix_barcodes" + } + } + ], + "provenance": { + "home_path": "/Users/vdn/Projects/immundata-rlang/inst/extdata/tsv/immundata-sample_0_1k", + "current_path": "/Users/vdn/Projects/immundata-rlang/inst/extdata/tsv/immundata-sample_0_1k", + "snapshot_root": "/Users/vdn/Projects/immundata-rlang/inst/extdata/tsv/immundata-sample_0_1k/snapshots", + "snapshot_id": "imd_20260403T110305Z_ctrax314", + "lineage": [ + { + "event": "ingestion", + "created_at": "2026-04-03T11:03:05Z", + "snapshot_id": "imd_20260403T110305Z_ctrax314", + "producer": { + "function": "read_repertoires" + }, + "inputs": { + "files": "/Users/vdn/Projects/immundata-rlang/inst/extdata/tsv/sample_0_1k.tsv", + "metadata_joined": false, + "enforce_schema": true + }, + "args": { + "barcode_col": null, + "count_col": null, + "locus_col": null, + "umi_col": null, + "metadata_file_col": "File" + }, + "column_lineage": { + "renamed": { + "requested": ["v_gene", "d_gene", "j_gene", "d_gene", "chain"], + "applied": [], + "not_found": ["v_gene", "d_gene", "j_gene", "d_gene", "chain"] + }, + "dropped": { + "applied": [] + } + }, + "pipeline": { + "preprocess": null, + "postprocess": "prefix_barcodes" + } + } + ] + }, + "extensions": [] +} diff --git a/inst/extdata/tsv/manifest.csv b/inst/extdata/tsv/manifest.csv new file mode 100644 index 0000000..05814bc --- /dev/null +++ b/inst/extdata/tsv/manifest.csv @@ -0,0 +1,3 @@ +file,Therapy,Response,Prefix +sample_0_1k.tsv,ICI,FR,S1_ +sample_1k_2k.tsv,CAR-T,PR,S2_ diff --git a/inst/extdata/tsv/metadata.tsv b/inst/extdata/tsv/metadata.tsv deleted file mode 100644 index adf7380..0000000 --- a/inst/extdata/tsv/metadata.tsv +++ /dev/null @@ -1,3 +0,0 @@ -File Therapy Response Prefix -sample_0_1k.tsv ICI FR S1_ -sample_1k_2k.tsv CAR-T PR S2_ diff --git a/man/IMD_GLOBALS.Rd b/man/IMD_GLOBALS.Rd index 891b8da..7d49895 100644 --- a/man/IMD_GLOBALS.Rd +++ b/man/IMD_GLOBALS.Rd @@ -1,12 +1,8 @@ % Generated by roxygen2: do not edit by hand % Please edit documentation in R/globals.R -\docType{data} \name{IMD_GLOBALS} \alias{IMD_GLOBALS} \title{Internal Immundata Global Configuration} -\format{ -An object of class \code{list} of length 6. -} \usage{ IMD_GLOBALS } @@ -24,14 +20,13 @@ field names, default file names, and internal error messages. \item \code{cell}: Column name for cell barcode IDs. \item \code{receptor}: Column name for receptor unique identifiers. \item \code{repertoire}: Column name for repertoire group IDs. -\item \code{metadata_filename}: Column name for metadata files (internal). +\item \code{manifest_filename}: Column name for manifest file paths (internal). \item \code{count}: Column name for receptor count per group. -\item \code{filename}: Original column name used in user metadata. } \item \code{files}: Default file names used to store structured Immundata: \itemize{ -\item \code{receptors}: File name for receptor-level data (\code{receptors.parquet}). -\item \code{annotations}: File name for annotation-level data (\code{annotations.parquet}). +\item \code{metadata}: File name for schemas and small summary tables (\code{metadata.json}). +\item \code{annotations}: File name for chain-level data (\code{annotations.parquet}). } } } diff --git a/man/ImmunData.Rd b/man/ImmunData.Rd index bde3378..5f08c70 100644 --- a/man/ImmunData.Rd +++ b/man/ImmunData.Rd @@ -2,88 +2,278 @@ % Please edit documentation in R/core_immundata.R \name{ImmunData} \alias{ImmunData} -\title{ImmunData: A Unified Structure for Immune Receptor Repertoire Data} +\title{ImmunData: A data structure for storing adaptive immune receptor repertoire data} \description{ -\code{ImmunData} is an abstract R6 class for managing and transforming immune receptor repertoire data. -It supports flexible backends (e.g., Arrow, DuckDB, dbplyr) and lazy evaluation, -and provides tools for filtering, aggregation, and receptor-to-repertoire mapping. +\code{ImmunData} stores adaptive immune receptor repertoire (AIRR) data and the rules +used to turn observed sequences or cells into data units for analysis. Think AnnData +or SeuratObject, but for immune repertoires. + +You work with an \code{ImmunData} object after importing bulk or single-cell AIRR-seq +data. The major idea behind \code{ImmunData} is that because sequencing provides only +information about sequences and, for single-cell data, cell +barcodes, the your responsibility is to determine, which sequences you want to treat as the +same receptor, repertoire, or stratum (group of repertoires). You define these analysis units with +schemas. A schema is a stored set of column names and chain-selection rules +that tells \code{ImmunData} how to group observations. Those definitions are kept +inside \code{ImmunData} to ensure that downstream functions count, filter, and compare +the same units consistently. Repertoire and strata schemas can be changed later to re-aggregate +repertoires differently, e.g., merge receptors from different clusters into +per-patient clusters. Receptor schema is fixed once and for all, so if you want +to work with a different receptor definition, e.g., use "CDR3aa + V gene" instead of +just "CDR3aa" as a definiton for a unique receptor, you will need to create +a separate \code{ImmunData} object. + +\code{ImmunData} is immutable, meaning that functions that transform an \code{ImmunData} +object return a new object, and the original object is not changed. Due to multiple +optimisations on the backend, it does not mean that you re-create the whole +dataset each time you run a, let's stay, a filter. However, it does affect analysis workflow +significantly. You can read about it more on the website and in tutorials. } -\seealso{ -\code{\link[=read_repertoires]{read_repertoires()}}, \code{\link[=read_immundata]{read_immundata()}} +\section{From observed data to analysis units}{ + + +\code{ImmunData} connects observed records to user-defined analysis units: +\itemize{ +\item A \strong{chain observation} is an observed receptor-chain sequence, such as a +TRA, TRB, or IGH sequence. Chain observations form the main table. +\item A \strong{barcode} is an observed identifier for a cell in single-cell data. It +links chains found in the same cell. +\item A \strong{receptor} is a virtual analysis unit that you define. For example, you +may define it by CDR3 sequence alone, by CDR3 and V gene, or as a paired +TRA-TRB receptor. The receptor schema records which chain features and loci +must match for observations to receive the same receptor identifier. +\item A \strong{repertoire} is a virtual collection of receptors that you define from +annotation columns. For example, one repertoire may contain all receptors +from one sample, or from one donor at one time point. +\item A \strong{stratum} is a virtual collection of repertoires for a comparison. For +example, one stratum may contain all repertoires from one treatment arm. } -\concept{core_immundata} -\section{Public fields}{ -\if{html}{\out{
}} -\describe{ -\item{\code{schema_receptor}}{A named list describing how to interpret receptor-level data. -This includes the fields used for aggregation (e.g., \code{CDR3}, \code{V_gene}, \code{J_gene}), -and optionally unique identifiers for each receptor row. Used to ensure consistency -across processing steps.} -\item{\code{schema_repertoire}}{A named list defining how barcodes or annotations should be -grouped into repertoires. This may include sample-level metadata (e.g., \code{sample_id}, -\code{donor_id}) used to define unique repertoires.} +These definitions do not change the observed sequences. They determine how +observations are grouped and counted during analysis. The resulting +hierarchy is \verb{chain observations and barcodes -> receptors -> repertoires -> strata}. +} + +\section{Inspect and transform an object}{ + + +Print an object for a compact overview. Use \verb{$receptors} for the receptor +table, \verb{$repertoires} for one summary row per repertoire, and \verb{$strata} for +one row per stratum. Most analysis functions accept the complete +\code{ImmunData} object directly. + +Common transformations include: +\itemize{ +\item \code{\link[=filter_immundata]{filter_immundata()}} to keep selected chains, cells, or receptors; +\item \code{\link[=mutate_immundata]{mutate_immundata()}} to calculate annotation columns; +\item \code{\link[=annotate]{annotate()}} to add external biological information; +\item \code{\link[=agg_repertoires]{agg_repertoires()}} to define repertoires; and +\item \code{\link[=agg_strata]{agg_strata()}} to group repertoires into strata. } -\if{html}{\out{
}} } -\section{Active bindings}{ -\if{html}{\out{
}} -\describe{ -\item{\code{receptors}}{Accessor for the dynamically-created table with receptors.} -\item{\code{annotations}}{Accessor for the annotation-level table (\code{.annotations}).} +\section{Create an object}{ -\item{\code{repertoires}}{Get a table of repertoires and their basic statistics.} -\item{\code{metadata}}{Get a table of repertoires without their basic statistics.} +Create an \code{ImmunData} object with \code{\link[=read_repertoires]{read_repertoires()}}, or reopen a saved +object with \code{\link[=read_immundata]{read_immundata()}}. Do not call the \verb{$new()} constructor in +analysis code. Direct construction is reserved for package developers. } -\if{html}{\out{
}} + +\section{Lazy data and storage}{ + + +The chain-level table uses duckplyr and can remain on disk. Filtering, +mutation, and aggregation stay lazy when possible, so large datasets do not +need to be loaded fully into R memory. Downstream analysis functions in the +\code{immunarch} package are designed to accept lazy \code{ImmunData} objects. Pass the +object directly; you usually do not need to call \code{\link[dplyr:collect]{dplyr::collect()}}. Collect +data only when another function explicitly requires an in-memory data frame +or when you want to inspect a small table in R. + +Objects created by \code{\link[=read_repertoires]{read_repertoires()}} are backed by files in their output +folder. Keep that folder while you use the object. Use \code{\link[=write_immundata]{write_immundata()}} to +save a transformed object and \code{\link[=read_immundata]{read_immundata()}} to reopen it. } -\section{Methods}{ -\subsection{Public methods}{ -\itemize{ -\item \href{#method-ImmunData-new}{\code{ImmunData$new()}} -\item \href{#method-ImmunData-clone}{\code{ImmunData$clone()}} + +\examples{ +library(immundata) +library(dplyr) + +options(immundata.verbose = FALSE) + +# Load the small dataset included with immundata, then define one repertoire +# for each treatment-response group. +idata <- get_test_idata() |> + agg_repertoires(schema = "Response") + +idata$repertoires |> + select(Response, n_barcodes, n_receptors) |> + arrange(Response) +# Expected result: +# Response n_barcodes n_receptors +# FR 955 871 +# PR 947 867 + +# Under the current receptor definition, the full-response (FR) repertoire +# contains 955 chain observations grouped into 871 receptor units. + +# Keep only the full-response repertoire. filter() returns a new object. +fr_only <- idata |> + filter(Response == "FR") + +tibble( + original_repertoires = nrow(idata$repertoires), + filtered_repertoires = nrow(fr_only$repertoires) +) +# Expected result: +# original_repertoires filtered_repertoires +# 2 1 +# The original object still contains both repertoires. + } +\seealso{ +\code{\link[=read_repertoires]{read_repertoires()}}, \code{\link[=read_immundata]{read_immundata()}}, \code{\link[=write_immundata]{write_immundata()}}, +\code{\link[=agg_repertoires]{agg_repertoires()}}, \code{\link[=agg_strata]{agg_strata()}}, \code{\link[=filter_immundata]{filter_immundata()}}, +\code{\link[=mutate_immundata]{mutate_immundata()}}, \code{\link[=annotate]{annotate()}} } -\if{html}{\out{
}} -\if{html}{\out{}} -\if{latex}{\out{\hypertarget{method-ImmunData-new}{}}} -\subsection{Method \code{new()}}{ -Creates a new \code{ImmunData} object. -This constructor expects receptor-level and barcode-level data, -along with a receptor schema defining aggregation and identity fields. -\subsection{Usage}{ -\if{html}{\out{
}}\preformatted{ImmunData$new(schema, annotations, repertoires = NULL)}\if{html}{\out{
}} +\concept{core_immundata} +\section{Public fields}{ + \if{html}{\out{
}} + \describe{ + \item{\code{schema_receptor}}{A named list defining the virtual receptor unit. +The \code{features} element names the chain columns used to group +observations, such as CDR3 sequence and V gene. The \code{chains} element +selects one chain or a paired set of chains.} + + \item{\code{schema_repertoire}}{A character vector naming annotation columns +whose unique combinations define one repertoire, such as \code{sample_id} +or \code{c("donor_id", "timepoint")}. It is \code{NULL} when repertoires have +not been defined.} + + \item{\code{schema_strata}}{A character vector naming repertoire-level columns +whose unique combinations define one stratum, such as \code{treatment}. It +is \code{NULL} when strata have not been defined.} + } + \if{html}{\out{
}} } +\section{Active bindings}{ + \if{html}{\out{
}} + \describe{ + \item{\code{receptors}}{A derived duckplyr table of distinct receptors. For a +paired receptor, the selected chain features are shown side by side.} + + \item{\code{annotations}}{The lazy duckplyr table of retained chain +observations and their biological annotations. For most tasks, pass the +complete \code{ImmunData} object to a transformation function or use +\code{collect(idata)} to inspect this table in memory.} + + \item{\code{repertoires}}{A small table with one row per repertoire, the columns +that define it, and summary statistics such as \code{n_barcodes} and +\code{n_receptors}. It is \code{NULL} when repertoires have not been defined.} -\subsection{Arguments}{ -\if{html}{\out{
}} -\describe{ -\item{\code{schema}}{A character vector specifying the receptor schema (e.g., aggregate fields, ID columns).} + \item{\code{strata}}{A small table with one row per stratum, its label, and the +repertoire-level columns that define it. It is \code{NULL} when strata have +not been defined.} -\item{\code{annotations}}{A cell/barcode-level dataset mapping barcodes to receptor rows.} + \item{\code{provenance}}{Read-only named list describing the snapshot origin +and storage context carried by this object. Retrieve the complete list with +\code{idata$provenance}, or one field with, for example, +\code{idata$provenance$current_path}. The fields are: +\itemize{ +\item \code{home_path}: project home used for managed snapshots and artifacts. +The original ingestion snapshot is stored directly in this folder; +it is \code{NULL} for an object with no persisted home. +\item \code{current_path}: exact folder of the most recently loaded or written +snapshot. Transformations preserve this source path until the +transformed object is written as another snapshot; it is \code{NULL} for +an object that has never been loaded from or written to disk. +\item \code{snapshot_root}: derived managed-snapshot root, +\code{home_path/snapshots}, or \code{NULL} when \code{home_path} is \code{NULL}. +\item \code{artifacts_root}: derived project-level root for optional external +tool outputs, \code{home_path/artifacts}, or \code{NULL} when \code{home_path} is +\code{NULL}. +\item \code{artifacts_path}: derived namespace for artifacts associated with the +most recently loaded or written snapshot. It is +\code{artifacts_root/root} for the original +ingestion, \verb{artifacts_root//vNNN} for a managed snapshot, and +\verb{artifacts_root/by-id/} for a detached explicit snapshot. +External tools can append \verb{/} and create that directory; +artifact contents are not part of \code{ImmunData}. Write a transformed +object as a new snapshot before storing artifacts that should be +associated with the transformed data. +\item \code{snapshot_id}: unique identifier generated when the snapshot is +written; \code{NULL} for an in-memory object that has never been written. +\item \code{lineage}: ordered list of ingestion and snapshot events leading to +the current snapshot. +} -\item{\code{repertoires}}{A repertoire table, created inside the body of \link{agg_repertoires}.} +The accessor is read-only; assigning to \code{idata$provenance} is an error.} + } + \if{html}{\out{
}} } -\if{html}{\out{
}} +\section{Methods}{ +\subsection{Public methods}{ + \itemize{ + \item \href{#method-ImmunData-initialize}{\code{ImmunData$new()}} + \item \href{#method-ImmunData-clone}{\code{ImmunData$clone()}} + } } +\if{html}{\out{
}} +\if{html}{\out{}} +\if{latex}{\out{\hypertarget{method-ImmunData-initialize}{}}} +\subsection{\code{ImmunData$new()}}{ + Low-level constructor for package developers. Analysis code +must create an \code{ImmunData} object with \code{\link[=read_repertoires]{read_repertoires()}} or reopen one +with \code{\link[=read_immundata]{read_immundata()}}. + \subsection{Usage}{ + \if{html}{\out{
}} + \preformatted{ImmunData$new( + schema, + annotations, + repertoires = NULL, + provenance = NULL, + strata = NULL +)} + \if{html}{\out{
}} + } + \subsection{Arguments}{ + \if{html}{\out{
}} + \describe{ + \item{\code{schema}}{A character vector or named list. A character vector names +the features used to define a chain-agnostic receptor. A named list is +created by \code{\link[=make_receptor_schema]{make_receptor_schema()}} and can also select receptor chains.} + \item{\code{annotations}}{A duckplyr table. It contains retained chain +observations, receptor identifiers, and biological annotations.} + \item{\code{repertoires}}{A data frame or \code{NULL}. It contains one row per +repertoire and its summary statistics and is usually created by +\code{\link[=agg_repertoires]{agg_repertoires()}}.} + \item{\code{provenance}}{A list or \code{NULL}. It contains internal storage and +snapshot history.} + \item{\code{strata}}{A data frame or \code{NULL}. It contains one row per stratum, +its label, and the repertoire-level columns that define it.} + } + \if{html}{\out{
}} + } } + \if{html}{\out{
}} \if{html}{\out{}} \if{latex}{\out{\hypertarget{method-ImmunData-clone}{}}} -\subsection{Method \code{clone()}}{ -The objects of this class are cloneable with this method. -\subsection{Usage}{ -\if{html}{\out{
}}\preformatted{ImmunData$clone(deep = FALSE)}\if{html}{\out{
}} +\subsection{\code{ImmunData$clone()}}{ + The objects of this class are cloneable with this method. + \subsection{Usage}{ + \if{html}{\out{
}} + \preformatted{ImmunData$clone(deep = FALSE)} + \if{html}{\out{
}} + } + \subsection{Arguments}{ + \if{html}{\out{
}} + \describe{ + \item{\code{deep}}{Whether to make a deep clone.} + } + \if{html}{\out{
}} + } } -\subsection{Arguments}{ -\if{html}{\out{
}} -\describe{ -\item{\code{deep}}{Whether to make a deep clone.} -} -\if{html}{\out{
}} -} -} } diff --git a/man/agg_receptors.Rd b/man/agg_receptors.Rd index 7f49151..291fc4c 100644 --- a/man/agg_receptors.Rd +++ b/man/agg_receptors.Rd @@ -1,8 +1,8 @@ % Generated by roxygen2: do not edit by hand -% Please edit documentation in R/operations_agg.R +% Please edit documentation in R/operations_agg_receptors.R \name{agg_receptors} \alias{agg_receptors} -\title{Aggregates AIRR data into receptors} +\title{Group AIRR sequence rows into receptors} \usage{ agg_receptors( dataset, @@ -10,104 +10,117 @@ agg_receptors( barcode_col = NULL, count_col = NULL, locus_col = NULL, - umi_col = NULL + umi_col = NULL, + verbose = getOption("immundata.verbose", TRUE) ) } \arguments{ -\item{dataset}{A data frame or \code{duckplyr_df} containing sequence/clonotype data. -Must include columns specified in \code{schema} and potentially \code{barcode_col}, -\code{count_col}, \code{locus_col}, \code{umi_col}. Could be \code{idata$annotations}.} +\item{dataset}{A duckplyr table containing AIRR sequence data, with one row +per chain or bulk clonotype. It must contain the columns named in \code{schema} +and in any of \code{barcode_col}, \code{count_col}, \code{locus_col}, and \code{umi_col} that +are supplied.} -\item{schema}{Defines how a unique receptor is identified. Can be: +\item{schema}{Definition of receptor identity. Supply either: \itemize{ -\item A character vector of column names representing receptor features -(e.g., \code{c("v_call", "j_call", "junction_aa")}). -\item A list created by \code{make_receptor_schema()}, specifying both \code{features} -(character vector) and optionally \code{chains} (character vector of locus -names like \code{"TRA"}, \code{"TRB"}, \code{"IGH"}, \code{"IGK"}, \code{"IGL"}, max length 2). -Specifying \code{chains} triggers filtering by locus and enables pairing logic -if two chains are given. -}} +\item A character vector naming the features that must match, such as +\code{c("v_call", "j_call", "junction_aa")}. +\item An object created by \code{\link[=make_receptor_schema]{make_receptor_schema()}}. Its \code{features} define chain +identity, while its optional \code{chains} select one locus or define a pair +of loci. +} + +A schema can contain at most two chain entries. Use syntax such as +\code{c("IGH", "IGL|IGK")} to accept either IGH-IGL or IGH-IGK pairs.} -\item{barcode_col}{Character(1). The name of the column containing cell -identifiers (barcodes). Required for single-cell processing and chain pairing. -Default: \code{NULL}.} +\item{barcode_col}{Name of the column containing cell barcodes. Supply this +for single-cell data. \code{umi_col} is then also required, and \code{count_col} +cannot be supplied. If \code{imd_filename} is present, identical barcode values +from different source files are treated as different cells. The default is +\code{NULL}.} -\item{count_col}{Character(1). The name of the column containing counts -(e.g., UMI counts for bulk, clonotype frequency). Used for bulk data -processing. Default: \code{NULL}. Cannot be specified if \code{barcode_col} is set.} +\item{count_col}{Name of the column containing non-negative abundance values +in bulk repertoire data. These values are copied to \code{imd_n_chains}. +\code{count_col} cannot be used together with \code{barcode_col}. The default is +\code{NULL}.} -\item{locus_col}{Character(1). The name of the column specifying the chain locus -(e.g., "TRA", "TRB"). Required if \code{schema} includes \code{chains} for filtering -or pairing. Default: \code{NULL}.} +\item{locus_col}{Name of the column containing loci such as \code{"TRA"}, \code{"TRB"}, +or \code{"IGH"}. It is required when \code{schema} specifies one or more chains. The +column is renamed to the standard name \code{locus} when necessary. The default +is \code{NULL}.} -\item{umi_col}{Character(1). The name of the column containing UMI counts. -Required for \emph{paired-chain single-cell} data (\code{length(schema$chains) == 2}). -Used to select the most abundant chain per locus within a cell when multiple -chains of the same locus are present. Default: \code{NULL}.} +\item{umi_col}{Name of the column containing per-chain UMI or read counts. +It is required when \code{barcode_col} is supplied and is used to choose one +chain when a cell contains several chains from the same locus. The default +is \code{NULL}.} + +\item{verbose}{Whether to print information about the selected processing +mode and loci. Defaults to \code{getOption("immundata.verbose", TRUE)}.} } \value{ -A \code{duckplyr_df} (or data frame) representing the annotated sequences. -This table links each original sequence record (chain) to a defined receptor -and includes standardized columns: +A duckplyr table containing the retained input rows and these +package-standard columns: \itemize{ -\item \code{imd_receptor_id}: Integer ID unique to each distinct receptor signature. -\item \code{imd_barcode_id}: Integer ID unique to each cell/barcode (or row if no barcode). -\item \code{imd_chain_id}: Integer ID unique to each input row (chain). -\item \code{imd_chain_count}: Integer count associated with the chain (1 for SC/simple, -from \code{count_col} for bulk). -This output is typically assigned to the \verb{$annotations} field of an \code{ImmunData} object. +\item \code{imd_receptor_id}: links rows that belong to the same receptor. +\item \code{imd_barcode}: contains the input cell barcode, or a synthetic row-level +barcode for uncounted and bulk data. +\item \code{imd_chain_id}: identifies an individual retained chain row. +\item \code{imd_n_chains}: contains \code{1} for uncounted and single-cell data, or the +value from \code{count_col} for bulk data. +\item \code{imd_count}: initialized to \code{0}; receptor counts are calculated later by +\code{\link[=agg_repertoires]{agg_repertoires()}}. } } \description{ -Processes a table of immune receptor sequences (chains or clonotypes) to -identify unique receptors based on a specified schema. It assigns a unique -identifier (\code{imd_receptor_id}) to each distinct receptor signature and -returns an annotated table linking the original sequence data to these -receptor IDs. +\code{agg_receptors()} is a low-level function used during AIRR data ingestion. It +decides which sequence rows represent the same biological receptor and adds +package-standard identifiers and counts to the input table. -This function is a core component used within \code{\link[=read_repertoires]{read_repertoires()}} and handles -different input data structures: -\itemize{ -\item Simple tables (no counts, no cell IDs). -\item Bulk sequencing data (using a count column). -\item Single-cell data (using a barcode/cell ID column). For single-cell data, -it can perform chain pairing if the schema specifies multiple chains -(e.g., TRA and TRB). -} +A receptor can be one chain or a pair of chains from the same cell. The +\code{schema} argument defines which sequence features and loci make two receptors +identical. + +This function works with a prepared duckplyr table and returns a duckplyr +table. It does not accept or return an \link{ImmunData} object. Most analysis +workflows should provide the same arguments to \code{\link[=read_repertoires]{read_repertoires()}}, which +calls \code{agg_receptors()} during import. } \details{ -The function performs the following main steps: -\enumerate{ -\item \strong{Validation:} Checks inputs, schema validity, and existence of required columns. -\item \strong{Schema Parsing:} Determines receptor features and target chains from \code{schema}. -\item \strong{Locus Filtering:} If \code{schema$chains} is provided, filters the dataset -to include only rows matching the specified locus/loci. -\item \strong{Processing Logic (based on \code{barcode_col} and \code{count_col}):} -\itemize{ -\item \strong{Simple Table/Bulk (No Barcodes):} Assigns unique internal barcode/chain IDs. -Identifies unique receptors based on \code{schema$features}. Calculates -\code{imd_chain_count} (1 for simple table, from \code{count_col} for bulk). -\item \strong{Single-Cell (Barcodes Provided):} Uses \code{barcode_col} for \code{imd_barcode_id}. +The receptor features are the columns that define the identity of one chain. +Two chains with the same values in all feature columns receive the same +receptor identity in a single-chain analysis. Common features include V gene, +J gene, and CDR3 amino acid sequence. + +The function supports three input modes: \itemize{ -\item \strong{Single Chain:} (\code{length(schema$chains) <= 1}). Identifies unique -receptors based on \code{schema$features}. \code{imd_chain_count} is 1. -\item \strong{Paired Chain:} (\code{length(schema$chains) == 2}). Requires \code{locus_col} -and \code{umi_col}. Filters chains within each cell/locus group based -on max \code{umi_col}. Creates paired receptors by joining the two -specified loci for each cell based on \code{schema$features} from both. -Assigns a unique \code{imd_receptor_id} to each \emph{pair}. -\code{imd_chain_count} is 1 (representing the chain record). -} -} -\item \strong{Output:} Returns an annotated data frame containing original columns plus -internal identifiers (\code{imd_receptor_id}, \code{imd_barcode_id}, \code{imd_chain_id}) -and counts (\code{imd_chain_count}). +\item \strong{Uncounted sequence table:} If neither \code{barcode_col} nor \code{count_col} is +supplied, every input row is treated as one observed chain. A synthetic +barcode is created for each row, and \code{imd_n_chains} is set to \code{1}. +\item \strong{Bulk repertoire:} If \code{count_col} is supplied, every input row receives a +synthetic barcode and its abundance is copied to \code{imd_n_chains}. +\item \strong{Single-cell repertoire:} If \code{barcode_col} is supplied, rows are grouped +by cell. \code{umi_col} is required and \code{imd_n_chains} is set to \code{1} for every +retained cell-chain observation. } -Internal column names are typically managed by \code{immundata:::imd_schema()}. +When one chain is specified in \code{schema}, only that locus is retained. If a +cell contains several chains from that locus, the row with the highest value +in \code{umi_col} is retained. If the highest values are tied, the first row is +retained. + +When two chains are specified, only cells containing both requested loci are +retained. The selected chains are paired by barcode, and both rows receive +the same \code{imd_receptor_id}. Cells with incomplete pairs are excluded. + +A relaxed pair such as \code{c("IGH", "IGL|IGK")} requires IGH and exactly one of +the two alternative light-chain loci. Cells containing both IGL and IGK are +excluded. + +Numeric \code{imd_receptor_id} values identify receptors within the returned +table. The particular number assigned to a receptor is not a biological +identifier and may change when the data are aggregated again. } \seealso{ -\code{\link[=read_repertoires]{read_repertoires()}}, \code{\link[=make_receptor_schema]{make_receptor_schema()}}, \link{ImmunData} +\code{\link[=read_repertoires]{read_repertoires()}}, \code{\link[=make_receptor_schema]{make_receptor_schema()}}, \code{\link[=agg_repertoires]{agg_repertoires()}}, +\link{ImmunData} } \concept{aggregation} diff --git a/man/agg_repertoires.Rd b/man/agg_repertoires.Rd index 761e37e..e458ff8 100644 --- a/man/agg_repertoires.Rd +++ b/man/agg_repertoires.Rd @@ -1,101 +1,155 @@ % Generated by roxygen2: do not edit by hand -% Please edit documentation in R/operations_agg.R +% Please edit documentation in R/operations_agg_repertoires.R \name{agg_repertoires} \alias{agg_repertoires} -\title{Aggregate AIRR data into repertoires} +\title{Define biological repertoires and calculate receptor abundance} \usage{ -agg_repertoires(idata, schema = "repertoire_id") +agg_repertoires( + idata, + schema = "repertoire_id", + verbose = getOption("immundata.verbose", TRUE) +) } \arguments{ -\item{idata}{An \code{ImmunData} object, typically the output of \code{\link[=read_repertoires]{read_repertoires()}} -or \code{\link[=read_immundata]{read_immundata()}}. Must contain the \verb{$annotations} table with columns -specified in \code{schema} and internal columns like \code{imd_receptor_id} and -\code{imd_chain_count}.} - -\item{schema}{Character vector. Column name(s) in \code{idata$annotations} that -define a unique repertoire. For example, \code{c("SampleID")} or -\code{c("DonorID", "TimePoint")}. Columns must exist in \code{idata$annotations}. -Default: \code{"repertoire_id"} (assumes such a column exists).} +\item{idata}{An \link{ImmunData} object containing receptor observations and the +columns named in \code{schema}. This is usually created by \code{\link[=read_repertoires]{read_repertoires()}} +or \code{\link[=read_immundata]{read_immundata()}}.} + +\item{schema}{A non-empty character vector. One or more column names that +together define a repertoire. For example, \code{"Sample"} creates one +repertoire per sample, and +\code{c("Sample", "TimePoint")} creates one repertoire per sample and time-point +combination. The default is \code{"repertoire_id"}; this column must exist if +the default is used.} + +\item{verbose}{A logical value. Accepted for consistency with other +aggregation functions. It currently does not change the output. Defaults to +\code{getOption("immundata.verbose", TRUE)}.} } \value{ -A \strong{new} \code{ImmunData} object. Its \verb{$annotations} table includes the -added columns (\code{imd_repertoire_id}, \code{imd_count}, \code{imd_proportion}, \code{n_repertoires}). -Its \verb{$repertoires} slot contains the summary table linking \code{schema} columns -to \code{imd_repertoire_id}, \code{n_barcodes}, and \code{n_receptors}. +A new \link{ImmunData} object with repertoire definitions and abundance +statistics. Its repertoire summary contains the \code{schema} columns, +\code{imd_repertoire_id}, \code{n_barcodes}, and \code{n_receptors}. Its chain rows also +contain \code{imd_repertoire_id}, \code{imd_count}, \code{imd_proportion}, and +\code{n_repertoires}. } \description{ -Groups the annotation table of an \code{ImmunData} object by user-specified -columns to define distinct \emph{repertoires} (e.g., based on sample, donor, -time point). It then calculates summary statistics both per-repertoire and -per-receptor within each repertoire. +Use \code{agg_repertoires()} to define which receptor observations belong to the +same biological repertoire and calculate receptor abundance within each +repertoire. -Calculated \strong{per repertoire}: -\itemize{ -\item \code{n_barcodes}: Total number of unique cells/barcodes within the repertoire -(sum of \code{imd_chain_count}, effectively summing unique cells if input was SC, -or total counts if input was bulk). -\item \code{n_receptors}: Number of unique receptors (\code{imd_receptor_id}) found within -the repertoire. +Use this function after importing data without repertoire definitions, or +when you want to redefine repertoires using sample information. One +repertoire usually represents one biological sample. It can also represent +one sample and time-point combination. The columns in \code{schema} define these +groups. + +The unit being defined is the repertoire. The function does not remove chain +rows or redefine cells or receptors. It returns a new \link{ImmunData} object. The +original object is not changed. } +\details{ +The function calculates summaries at repertoire and receptor levels while +keeping the original chain rows. +} +\section{What the function calculates}{ + -Calculated \strong{per annotation row} (receptor within repertoire context): +The returned repertoire summary contains one row for each repertoire: \itemize{ -\item \code{imd_count}: Total count of a specific receptor (\code{imd_receptor_id}) within -the specific repertoire it belongs to in that row (sum of relevant -\code{imd_chain_count}). -\item \code{imd_proportion}: The proportion of the repertoire's total \code{n_barcodes} -accounted for by that specific receptor (\code{imd_count / n_barcodes}). -\item \code{n_repertoires}: The total number of distinct repertoires (across the entire -dataset) in which this specific receptor (\code{imd_receptor_id}) appears. +\item \code{imd_repertoire_id}: a new integer identifier for the repertoire. +\item \code{n_barcodes}: the number of observed cells for single-cell data, or the +total abundance for bulk data. +\item \code{n_receptors}: the number of distinct receptors in the repertoire. } -These statistics are added to the annotation table, and a summary table is -stored in the \verb{$repertoires} slot of the returned object. +The function also adds these values to each chain row: +\itemize{ +\item \code{imd_repertoire_id}: the repertoire containing the row. +\item \code{imd_count}: the number of cells carrying that receptor in single-cell +data, or its summed abundance in bulk data, within the repertoire. +\item \code{imd_proportion}: the receptor's fraction of the repertoire, calculated as +\code{imd_count / n_barcodes}. +\item \code{n_repertoires}: the number of repertoires in which the receptor occurs. } -\details{ -The function operates on the \code{idata$annotations} table: -\enumerate{ -\item \strong{Validation:} Checks \code{idata} and existence of \code{schema} columns. Removes -any pre-existing repertoire summary columns to prevent duplication. -\item \strong{Repertoire Definition:} Groups annotations by the \code{schema} columns. -Calculates total counts (\code{n_barcodes}) per group. Assigns a unique integer -\code{imd_repertoire_id} to each distinct repertoire group. This forms the -initial \code{repertoires_table}. -\item \strong{Receptor Counts & Proportion:} Calculates the sum of \code{imd_chain_count} -for each receptor within each repertoire (\code{imd_count}). Calculates the -proportion (\code{imd_proportion}) of each receptor within its repertoire. -\item \strong{Repertoire & Receptor Stats:} Counts unique receptors per repertoire -(\code{n_receptors}, added to \code{repertoires_table}). Counts the number of -distinct repertoires each unique receptor appears in (\code{n_repertoires}). -\item \strong{Join Results:} Joins the calculated \code{imd_count}, \code{imd_proportion}, and -\code{n_repertoires} back to the annotation table based on repertoire columns -and \code{imd_receptor_id}. -\item \strong{Return New Object:} Creates and returns a \emph{new} \code{ImmunData} object -containing the updated \verb{$annotations} table (with the added statistics) -and the \verb{$repertoires} slot populated with the \code{repertoires_table} -(containing \code{schema} columns, \code{imd_repertoire_id}, \code{n_barcodes}, \code{n_receptors}). + +Values calculated for a receptor are repeated on all chain rows belonging to +that receptor in the same repertoire. + +Calling \code{agg_repertoires()} again replaces previous repertoire definitions, +receptor counts, proportions, and related strata summaries. } -The original \code{idata} object remains unmodified. Internal column names are -typically managed by \code{immundata:::imd_schema()}. +\section{Backend and storage}{ + + +Large-table calculations run on the duckplyr annotation table. The annotation +data remain lazy when the input is lazy. The small repertoire summary is +collected and stored in the returned object. + +Aggregation can be expensive for a large dataset. After checking the result, +consider saving it so later analyses do not repeat the calculation. Use +\code{write_immundata(idata, tag = "by-sample")} to create a managed snapshot in +the object's project home. Managed snapshots are versioned, so another write +with the same tag creates a new version and keeps the earlier version. + +Use \code{write_immundata(idata, output_folder = "path/to/result")} when you need a +standalone saved state in a specific folder, for example to share it or to +choose a new storage location. Unlike a managed snapshot, writing to an +existing explicit folder replaces the ImmunData files in that folder. Both +forms materialize pending duckplyr calculations and return a disk-backed +object that can be reopened with \code{\link[=read_immundata]{read_immundata()}}. } + \examples{ -\dontrun{ -# Assume 'idata_raw' is an ImmunData object loaded via read_repertoires -# but *without* providing 'repertoire_schema' initially. -# It has $annotations but $repertoires is likely NULL or empty. -# Assume idata_raw$annotations has columns "SampleID" and "TimePoint". - -# Define repertoires based on SampleID and TimePoint -idata_aggregated <- agg_repertoires(idata_raw, schema = c("SampleID", "TimePoint")) - -# Explore the results -print(idata_aggregated) -print(idata_aggregated$repertoires) -print(head(idata_aggregated$annotations)) # Note the new columns -} +library(immundata) +library(dplyr) + +options(immundata.verbose = FALSE) + +# Create a small bulk T-cell receptor dataset from two biological samples +bulk_data <- tibble( + Sample = c("Tumor", "Tumor", "Blood", "Blood"), + cdr3_aa = c("CASSA", "CASSB", "CASSA", "CASSC"), + v_call = c("TRBV1", "TRBV2", "TRBV1", "TRBV3"), + abundance = c(20L, 5L, 4L, 6L) +) + +bulk_file <- tempfile(fileext = ".tsv") +readr::write_tsv(bulk_data, bulk_file) + +# Import receptors without defining repertoires +idata <- read_repertoires( + path = bulk_file, + schema = c("cdr3_aa", "v_call"), + count_col = "abundance", + repertoire_schema = NULL, + output_folder = tempfile("immundata-example-") +) + +# Define one repertoire for each biological sample +sample_repertoires <- idata |> + agg_repertoires(schema = "Sample") + +sample_repertoires$repertoires |> + select(Sample, n_barcodes, n_receptors) |> + arrange(Sample) +# Expected result: +# Sample n_barcodes n_receptors +# Blood 10 2 +# Tumor 25 2 + +# For example, CASSA forms 80\% of the Tumor repertoire and 40\% of the +# Blood repertoire. It occurs in two repertoires. + +# For a large dataset, save the result as a managed snapshot so this +# aggregation does not need to run again. +saved_repertoires <- write_immundata( + sample_repertoires, + tag = "by-sample" +) } \seealso{ -\code{\link[=read_repertoires]{read_repertoires()}} (which can call this function), \link{ImmunData} class. +\code{\link[=read_repertoires]{read_repertoires()}}, \code{\link[=agg_strata]{agg_strata()}}, \code{\link[=write_immundata]{write_immundata()}}, \link{ImmunData} } \concept{aggregation} diff --git a/man/agg_strata.Rd b/man/agg_strata.Rd new file mode 100644 index 0000000..a4d1ef7 --- /dev/null +++ b/man/agg_strata.Rd @@ -0,0 +1,88 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/operations_agg_strata.R +\name{agg_strata} +\alias{agg_strata} +\title{Group repertoires into biological strata} +\usage{ +agg_strata(idata, schema, prefix = "Strata") +} +\arguments{ +\item{idata}{An \link{ImmunData} object with repertoires already defined. Use +\code{\link[=agg_repertoires]{agg_repertoires()}} first if the object does not contain repertoires.} + +\item{schema}{A non-empty character vector. One or more repertoire-level +columns that define a stratum. For example, use \code{"Therapy"} for treatment +arms or \code{c("Tissue", "Disease")} for each tissue and disease combination. +The columns must be present in \code{idata$repertoires}.} + +\item{prefix}{A non-empty character string. Prefix for the +automatic stratum labels. The default, \code{"Strata"}, produces labels such as +\code{"Strata1"} and \code{"Strata2"}. You can use \code{\link[=rename_strata]{rename_strata()}} to assign meaningful +labels later instead.} +} +\value{ +A new \link{ImmunData} object in which every repertoire belongs to one +stratum. The \verb{$strata} table lists the strata, their defining biological +values, and their automatic labels. Repertoire definitions and summary +statistics are preserved. +} +\description{ +Use \code{agg_strata()} to place sample repertoires into biological comparison +groups, such as treatment arms, tissues, or disease groups. + +Use this function after \code{\link[=agg_repertoires]{agg_repertoires()}} when several repertoires should be +analysed as one group. A \emph{stratum} contains every repertoire with the same +value, or the same combination of values, in \code{schema}. + +The unit being grouped is a whole repertoire. The function returns a new +\link{ImmunData} object. The original object is not changed. +} +\details{ +If \code{schema} contains several columns, a separate stratum is created for each +observed combination. For example, \code{c("Tissue", "Therapy")} can define +separate blood and tumour strata within each treatment arm. + +Calling \code{agg_strata()} again replaces the existing strata with groups defined +by the new \code{schema}. +} +\section{Identifiers and storage}{ + + +\code{imd_strata_id} is an internal identifier and can change when strata are +rebuilt. It is added to the repertoire table and to the underlying chain +annotations. \code{strata_name} is stored only in the smaller repertoire and +strata tables. + +Calling \code{\link[=agg_repertoires]{agg_repertoires()}} again rebuilds the repertoires, so it removes the +existing strata. Call \code{agg_strata()} again after redefining repertoires. +} + +\examples{ +library(immundata) +library(dplyr) + +options(immundata.verbose = FALSE) + +# Define sample repertoires using the biological metadata in the test data +idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) + +# Group the sample repertoires into treatment arms +treatment_groups <- idata |> + agg_strata(schema = "Therapy") + +treatment_groups$repertoires |> + select(Therapy, Response, imd_strata_id, strata_name) |> + arrange(imd_strata_id) +# Expected result: +# Therapy Response imd_strata_id strata_name +# CAR-T PR 1 Strata1 +# ICI FR 2 Strata2 + +# Each repertoire is now assigned to its treatment stratum. Any additional +# repertoire with the same Therapy value would receive the same stratum ID. +} +\seealso{ +\code{\link[=agg_repertoires]{agg_repertoires()}}, \code{\link[=rename_strata]{rename_strata()}}, \link{ImmunData} +} +\concept{aggregation} diff --git a/man/annotate_anndata.Rd b/man/annotate_anndata.Rd new file mode 100644 index 0000000..8157572 --- /dev/null +++ b/man/annotate_anndata.Rd @@ -0,0 +1,23 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/operations_external_annotate_anndatar.R +\name{annotate_anndata} +\alias{annotate_anndata} +\title{Annotate an AnnData object from ImmunData (by barcode)} +\usage{ +annotate_anndata(idata, adata, cols) +} +\arguments{ +\item{idata}{An \link{ImmunData} object.} + +\item{adata}{An \link[anndataR:AbstractAnnData]{anndataR::AbstractAnnData} object.} + +\item{cols}{Character vector with column names to transfer from +\code{idata$annotations}.} +} +\value{ +The updated AnnData object. +} +\description{ +Copy selected columns from \code{idata$annotations} to \code{adata$obs}, matching by +cell barcode (\code{adata$obs_names}). +} diff --git a/man/annotate_immundata.Rd b/man/annotate_immundata.Rd index 0352b2b..9be3f4a 100644 --- a/man/annotate_immundata.Rd +++ b/man/annotate_immundata.Rd @@ -6,24 +6,33 @@ \alias{annotate_receptors} \alias{annotate_barcodes} \alias{annotate_chains} -\title{Annotate ImmunData object} +\title{Add external information to ImmunData} \usage{ annotate_immundata( idata, annotations, by, keep_repertoires = TRUE, - remove_limit = FALSE + remove_limit = FALSE, + conflicts = c("error", "replace") ) -annotate(idata, annotations, by, keep_repertoires = TRUE, remove_limit = FALSE) +annotate( + idata, + annotations, + by, + keep_repertoires = TRUE, + remove_limit = FALSE, + conflicts = c("error", "replace") +) annotate_receptors( idata, annotations, annot_col = imd_schema("receptor"), keep_repertoires = TRUE, - remove_limit = FALSE + remove_limit = FALSE, + conflicts = c("error", "replace") ) annotate_barcodes( @@ -31,7 +40,8 @@ annotate_barcodes( annotations, annot_col = "", keep_repertoires = TRUE, - remove_limit = FALSE + remove_limit = FALSE, + conflicts = c("error", "replace") ) annotate_chains( @@ -39,106 +49,253 @@ annotate_chains( annotations, annot_col = imd_schema("chain"), keep_repertoires = TRUE, - remove_limit = FALSE + remove_limit = FALSE, + conflicts = c("error", "replace") ) } \arguments{ -\item{idata}{An \code{ImmunData} R6 object containing repertoire and annotation data.} +\item{idata}{An \link{ImmunData} object.} + +\item{annotations}{A data frame containing the information to add. It must +contain the columns used for matching and at most one row for each matching +identifier or combination of identifiers.} -\item{annotations}{A data frame containing the annotations to be joined.} +\item{by}{A named character vector describing how columns are matched. Names +are columns in \code{idata}; values are the corresponding columns in +\code{annotations}. For example, \code{c(Response = "response_code")}.} -\item{by}{A named character vector specifying the columns to join by. The names of the -vector should be the column names in \code{idata$annotations} and the values should be -the corresponding column names in the \code{annotations} data frame.} +\item{keep_repertoires}{Whether to preserve existing repertoire and strata +summaries without recalculation. The default is \code{TRUE}. If \code{FALSE}, these +summaries and their derived annotation columns are removed.} -\item{keep_repertoires}{Logical. If \code{TRUE} (default) and the \code{ImmunData} object -contains repertoire data (\code{idata$schema_repertoire} is not NULL), the repertoires -will be re-aggregated after joining the annotations. Set to \code{FALSE} if you do not -want to re-aggregate repertoires immediately.} +\item{remove_limit}{Whether to allow an annotation table with 100 or more +columns. The default is \code{FALSE}, which stops the operation for such tables. +Set to \code{TRUE} only when the wide join is intentional.} -\item{remove_limit}{Logical. If \code{FALSE} (default), a warning will be issued if the -\code{annotations} data frame has 100 or more columns, suggesting potential performance -issues. Set to \code{TRUE} to disable this warning and allow joining of annotations -with an arbitrary number of columns. Use with caution, as joining wide dataframes -can be memory-intensive and slow.} +\item{conflicts}{How to handle new annotation columns whose names already +exist in \code{idata}. \code{"error"}, the default, stops the operation. \code{"replace"} +replaces existing columns that are not protected by \code{ImmunData}.} -\item{annot_col}{A character vector specifying the column with receptor, barcode or chain identifiers -to annotate a corresponding receptors, barode or chains in \code{idata}.} +\item{annot_col}{Name of the identifier column in \code{annotations}. For +\code{annotate_receptors()} and \code{annotate_chains()}, the default is the standard +\code{ImmunData} receptor or chain identifier. For \code{annotate_barcodes()}, the +default \code{""} uses the row names of \code{annotations}. Supplying an +explicit barcode column is usually clearer.} } \value{ -A new \code{ImmunData} object with the annotations joined to the \code{annotations} slot. +A new \link{ImmunData} object containing the added annotation columns. +Existing repertoire and strata summaries are preserved when +\code{keep_repertoires = TRUE}. } \description{ -Joins additional annotation data to the annotations slot of an \code{ImmunData} object. +Use the \verb{annotate_*()} functions to add information stored in another data +frame to an \link{ImmunData} object. For example, you can add cell types from a +single-cell analysis, antigen labels for receptors, or clinical information +for samples. -This function allows you to add extra information to your repertoire data by joining a -dataframe of annotations based on specified columns. It supports joining by -one or more columns. +When matching identifiers are unique, the functions keep every row in +\code{idata}. When a row has no match in \code{annotations}, the new columns contain +\code{NA}. Each function returns a new \link{ImmunData} object. The original object is +not changed. } \details{ -The function performs a left join operation, keeping all rows from -\code{idata$annotations} and adding matching columns from the \code{annotations} data frame. -If there are multiple matches in \code{annotations} for a row in \code{idata$annotations}, -all combinations will be returned, potentially increasing the number of rows -in the resulting annotations table. +The functions differ in how they select the columns used for matching. The +rules for duplicate identifiers, column conflicts, and preserved summaries +are the same for all functions. +} +\section{Choose a function}{ + + +Use the function that matches the type of information you want to add: +\itemize{ +\item \code{\link[=annotate_barcodes]{annotate_barcodes()}} matches cell or barcode identifiers. +\item \code{\link[=annotate_receptors]{annotate_receptors()}} matches receptor identifiers. All rows belonging to +a matched receptor receive the new information. +\item \code{\link[=annotate_chains]{annotate_chains()}} matches chain identifiers. +\item \code{annotate()} matches any one or more columns that you specify in \code{by}. +} + +The first three functions select the correct \code{ImmunData} identifier for you. +\code{annotate_immundata()} is an alternative name for \code{annotate()}. +} + +\section{How matching works}{ + + +For \code{annotate_barcodes()}, \code{annotate_receptors()}, and \code{annotate_chains()}, +\code{annot_col} names the identifier column in \code{annotations}. For example, +\code{annot_col = "barcode"} matches the \code{barcode} column in \code{annotations} with +the barcode identifier in \code{idata}. + +For general matching, supply \code{by} in the form +\code{c(immundata_column = "annotation_column")}. For example, +\code{by = c(Response = "response_code")} matches the \code{Response} column in +\code{idata} with the \code{response_code} column in \code{annotations}. To match columns +with the same name, use a value such as \code{by = c(Response = "Response")}. +You can include more than one pair of columns in \code{by}. + +Columns from \code{annotations} that are not used for matching are added to the +result. Rows in \code{idata} without a match receive \code{NA}. Rows in \code{annotations} +without a match are ignored. +} + +\section{Annotation identifiers must be unique}{ + + +\code{annotations} must contain at most one row for each identifier, or each +combination of identifiers when matching several columns. For example, a +barcode annotation table must contain at most one row per barcode. + +The function does not check this rule because the annotation table may be +very large. If an identifier occurs several times, the corresponding rows in +\code{idata} are repeated. This can make receptor counts, proportions, and other +summaries incorrect. +} + +\section{Existing annotation columns}{ -The function uses \code{checkmate} to validate the input types and structure. -A check is performed to ensure that the columns specified in \code{by} exist in both -\code{idata$annotations} and the \code{annotations} data frame. +By default, the function stops if a new annotation column has the same name +as a column already present in \code{idata}. This prevents accidental replacement. -The \code{annotations} data frame is converted to a duckdb tibble internally for -efficient joining, especially with large datasets. +Use \code{conflicts = "replace"} to replace existing annotation columns. Columns +that define receptors, repertoires, strata, or other \code{ImmunData} state are +protected and cannot be replaced. The old column is removed before matching, +so rows without a new match receive \code{NA}. } -\section{Warning}{ -By default (\code{remove_limit = FALSE}), joining an \code{annotations} data frame with 100 or -more columns will trigger a warning. This is a safeguard to prevent accidental -joining of very wide data (e.g., gene expression data) that could lead to -performance degradation or crashes. If you understand the risks and intend to join -a wide data frame, set \code{remove_limit = TRUE}. +\section{Repertoire and strata summaries}{ + + +With the default \code{keep_repertoires = TRUE}, existing repertoire and strata +summaries are copied to the new object without recalculation. Use this option +when you are only adding information and the matching identifiers in +\code{annotations} are unique. + +Set \code{keep_repertoires = FALSE} when you plan to filter rows or define new +repertoires using the added information. This removes existing repertoire and +strata summaries and their derived columns. After annotation and filtering, +use \code{\link[=agg_repertoires]{agg_repertoires()}} to define the new repertoires. +} + +\section{Very wide annotation tables}{ + + +By default, the function stops when \code{annotations} contains 100 or more +columns. Adding a very wide table, such as a complete gene-expression matrix, +can be slow and require a large amount of memory. If you understand this cost, +set \code{remove_limit = TRUE} to allow the operation. } \examples{ -\dontrun{ -# Assuming 'my_immun_data' is an ImmunData object and 'sample_info' is a data frame -# with a column 'sample_id' matching 'sample' in my_immun_data$annotations -# and additional columns like 'treatment' and 'disease_status'. - -sample_info <- data.frame( - sample_id = c("sample1", "sample2", "sample3", "sample4"), - treatment = c("Treatment A", "Treatment B", "Treatment A", "Treatment C"), - disease_status = c("Healthy", "Disease", "Healthy", "Disease"), - stringsAsFactors = FALSE # Important to keep characters as characters +library(immundata) +library(dplyr) + +options(immundata.verbose = FALSE) + +# Load data included with immundata +idata <- get_test_idata() + +# Add cell types by matching barcode identifiers +cell_labels <- tibble( + barcode = c("S1_1", "S1_2"), + cell_type = c("CD8 T cell", "CD4 T cell") ) -# Join sample information using the 'sample' column -my_immun_data_annotated <- annotate( - idata = my_immun_data, - annotations = sample_info, - by = c("sample" = "sample_id") +idata_with_cells <- idata |> + annotate_barcodes( + annotations = cell_labels, + annot_col = "barcode" + ) + +idata_with_cells |> + collect() |> + filter(imd_barcode \%in\% c("S1_1", "S1_2", "S1_3")) |> + select(imd_barcode, cell_type) |> + arrange(imd_barcode) +# Expected result: +# imd_barcode cell_type +# S1_1 CD8 T cell +# S1_2 CD4 T cell +# S1_3 NA + +# Add antigen labels to selected receptors +receptor_labels <- tibble( + receptor_id = c(738L, 1567L), + antigen = c("CMV", "CMV") ) -# New sample_info +idata_with_antigens <- idata |> + annotate_receptors( + annotations = receptor_labels, + annot_col = "receptor_id" + ) -# Join data by multiple columns, e.g., 'sample' and 'barcode' -# Assuming 'cell_annotations' is a data frame with 'sample_barcode' and 'cell_type' -my_immun_data_cell_annotated <- annotate( - idata = my_immun_data, - annotations = cell_annotations, - by = c("sample" = "sample", "barcode" = "sample_barcode") +idata_with_antigens |> + collect() |> + filter(!is.na(antigen)) |> + distinct(imd_receptor_id, cdr3_aa, antigen) |> + arrange(imd_receptor_id) +# Expected result: +# imd_receptor_id cdr3_aa antigen +# 738 ASRAGAGTGELF CMV +# 1567 ASFPVLSPYNEQF CMV + +# Match columns with different names +response_info <- tibble( + response_code = c("FR", "PR"), + response_label = c("Full response", "Partial response") ) -# Join a wide dataframe, suppressing the column limit warning -# Assuming 'gene_expression' is a data frame with 'barcode' and many gene columns -my_immun_data_gene_expression <- annotate( - idata = my_immun_data, - annotations = gene_expression, - by = c("barcode" = "barcode"), - remove_limit = TRUE +idata_with_response <- idata |> + annotate( + annotations = response_info, + by = c(Response = "response_code") + ) + +idata_with_response |> + collect() |> + distinct(Response, response_label) |> + arrange(Response) +# Expected result: +# Response response_label +# FR Full response +# PR Partial response + +# Replace an annotation column intentionally +revised_cell_labels <- tibble( + barcode = c("S1_1", "S1_2"), + cell_type = c("Cytotoxic T cell", "Helper T cell") ) -} +idata_with_revised_cells <- idata_with_cells |> + annotate_barcodes( + annotations = revised_cell_labels, + annot_col = "barcode", + conflicts = "replace" + ) + +# Remove old repertoire summaries before defining repertoires by cell type +cell_repertoires <- idata |> + annotate_barcodes( + annotations = cell_labels, + annot_col = "barcode", + keep_repertoires = FALSE + ) |> + filter(!is.na(cell_type)) |> + agg_repertoires(schema = "cell_type") + +cell_repertoires$repertoires |> + arrange(cell_type) +# Expected result: +# imd_repertoire_id cell_type n_barcodes n_receptors +# 1 CD4 T cell 1 1 +# 2 CD8 T cell 1 1 + +} +\seealso{ +\code{\link[dplyr:left_join]{dplyr::left_join()}}, \code{\link[=agg_repertoires]{agg_repertoires()}}, \code{\link[=filter_immundata]{filter_immundata()}}, +\code{\link[=mutate_immundata]{mutate_immundata()}}, \link{ImmunData} } \concept{Annotation} \concept{annotation} diff --git a/man/collect.ImmunData.Rd b/man/collect.ImmunData.Rd new file mode 100644 index 0000000..bb5f116 --- /dev/null +++ b/man/collect.ImmunData.Rd @@ -0,0 +1,22 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/operations_compute_collect.R +\name{collect.ImmunData} +\alias{collect.ImmunData} +\title{Collect ImmunData annotations} +\usage{ +\method{collect}{ImmunData}(x, ...) +} +\arguments{ +\item{x}{ImmunData object.} + +\item{...}{Additional arguments passed to \code{\link[dplyr:collect]{dplyr::collect()}} for +\code{x$annotations}.} +} +\value{ +A tibble with collected annotations. +} +\description{ +Collects annotations from an \code{ImmunData} object and returns them as a tibble. +Factor columns are converted to character. +} +\concept{operations} diff --git a/man/compute.ImmunData.Rd b/man/compute.ImmunData.Rd new file mode 100644 index 0000000..8de0bad --- /dev/null +++ b/man/compute.ImmunData.Rd @@ -0,0 +1,23 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/operations_compute_collect.R +\name{compute.ImmunData} +\alias{compute.ImmunData} +\title{Compute ImmunData annotations} +\usage{ +\method{compute}{ImmunData}(x, ...) +} +\arguments{ +\item{x}{ImmunData object.} + +\item{...}{Additional arguments passed to \code{\link[dplyr:compute]{dplyr::compute()}} for +\code{x$annotations}.} +} +\value{ +A new \code{ImmunData} object with computed annotations and the input +repertoire, strata, schema, and provenance state preserved. +} +\description{ +Materializes the annotation table of an \code{ImmunData} object via +\code{\link[dplyr:compute]{dplyr::compute()}} and returns a new \code{ImmunData}. +} +\concept{operations} diff --git a/man/count.ImmunData.Rd b/man/count.ImmunData.Rd index fbe1ae6..5796076 100644 --- a/man/count.ImmunData.Rd +++ b/man/count.ImmunData.Rd @@ -2,22 +2,70 @@ % Please edit documentation in R/operations_count.R \name{count.ImmunData} \alias{count.ImmunData} -\title{Count the number of chains in ImmunData} +\title{Count chain rows in ImmunData} \usage{ \method{count}{ImmunData}(x, ..., wt = NULL, sort = FALSE, name = NULL) } \arguments{ -\item{x}{ImmunData object.} +\item{x}{An \link{ImmunData} object.} -\item{...}{Not used.} +\item{...}{Additional arguments. Accepted for compatibility with +\code{\link[dplyr:count]{dplyr::count()}}, but currently ignored.} -\item{wt}{Not used.} +\item{wt}{Any value or \code{NULL}. Accepted for compatibility with +\code{\link[dplyr:count]{dplyr::count()}}, but currently ignored.} -\item{sort}{Not used.} +\item{sort}{A logical value. Accepted for compatibility with +\code{\link[dplyr:count]{dplyr::count()}}, but currently ignored.} -\item{name}{Not used.} +\item{name}{A character string or \code{NULL}. Accepted for compatibility with +\code{\link[dplyr:count]{dplyr::count()}}, but currently ignored. The result column is always named +\code{n}.} +} +\value{ +A one-row duckplyr table with an integer column named \code{n}. This value +is the number of rows in the chain-level annotation table. } \description{ -Count the number of chains in ImmunData +Use \code{count()} to find how many chain rows are stored in an \link{ImmunData} +object. + +Use this method for a quick check of dataset size. The unit counted is one +retained chain row. Each retained cell with a paired receptor usually +contributes two rows, one for each chain. The same receptor can therefore +contribute two rows for every cell carrying it. For bulk data with an +abundance column, this method counts table rows rather than the summed +sequence abundance. + +The function returns a one-row duckplyr table. The original object is not +changed. +} +\details{ +This method currently provides only the total row count. The grouping, +weighting, sorting, and result-name arguments of \code{\link[dplyr:count]{dplyr::count()}} are +accepted for method compatibility but are not applied. + +The calculation runs on the duckplyr annotation table and can remain in +DuckDB. Use \code{\link[dplyr:pull]{dplyr::pull()}} or \code{\link[dplyr:collect]{dplyr::collect()}} to bring the small result +into R. +} +\examples{ +library(immundata) +library(dplyr) + +options(immundata.verbose = FALSE) +idata <- get_test_idata() + +idata |> count() +# Expected result: +# n +# 1902 + +# The result means that the object contains 1,902 retained chain rows. +# It does not mean that it contains 1,902 unique receptors. + +} +\seealso{ +\code{\link[dplyr:count]{dplyr::count()}}, \code{\link[dplyr:collect]{dplyr::collect()}}, \link{ImmunData} } \concept{operations} diff --git a/man/dimnames-set-.ImmunData.Rd b/man/dimnames-set-.ImmunData.Rd new file mode 100644 index 0000000..a820c56 --- /dev/null +++ b/man/dimnames-set-.ImmunData.Rd @@ -0,0 +1,17 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/operations_dimnames.R +\name{dimnames<-.ImmunData} +\alias{dimnames<-.ImmunData} +\title{Prevent Renaming ImmunData via dimnames} +\usage{ +\method{dimnames}{ImmunData}(x) <- value +} +\arguments{ +\item{x}{ImmunData object.} + +\item{value}{Not used.} +} +\description{ +Disallows replacing dimension names on \code{ImmunData}. +} +\concept{operations} diff --git a/man/dimnames.ImmunData.Rd b/man/dimnames.ImmunData.Rd new file mode 100644 index 0000000..3823df8 --- /dev/null +++ b/man/dimnames.ImmunData.Rd @@ -0,0 +1,19 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/operations_dimnames.R +\name{dimnames.ImmunData} +\alias{dimnames.ImmunData} +\title{Get Annotation Dimnames from ImmunData} +\usage{ +\method{dimnames}{ImmunData}(x) +} +\arguments{ +\item{x}{ImmunData object.} +} +\value{ +A list with \code{NULL} row names and annotation column names. +} +\description{ +Returns dimension names for an \code{ImmunData} object so that +\code{colnames(idata)} maps to annotation column names. +} +\concept{operations} diff --git a/man/downsample_immundata.Rd b/man/downsample_immundata.Rd new file mode 100644 index 0000000..0f04c90 --- /dev/null +++ b/man/downsample_immundata.Rd @@ -0,0 +1,142 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/operations_downsample.R +\name{downsample_immundata} +\alias{downsample_immundata} +\title{Reduce repertoires to a common sampling depth} +\usage{ +downsample_immundata(idata, n, seed = NULL) +} +\arguments{ +\item{idata}{An \link{ImmunData} object. For comparisons between repertoires, its +repertoires should already be defined with \code{\link[=read_repertoires]{read_repertoires()}} or +\code{\link[=agg_repertoires]{agg_repertoires()}}.} + +\item{n}{A number. Sampling depth. Use a value strictly between 0 and 1 for a +fraction, or a whole number greater than or equal to 1 for an absolute +number of cells or bulk sequence counts.} + +\item{seed}{A non-negative integer or \code{NULL}. Used to reproduce the same +random sample. The default is \code{NULL}.} +} +\value{ +A new \link{ImmunData} object containing the sampled chain observations. +If the input has repertoires or strata, their summaries are recalculated +for the sampled data. +} +\description{ +Use \code{downsample_immundata()} to reduce every repertoire to the same number or +fraction of observed cells or bulk sequence counts before comparing +repertoires. So, it is just a downsampling. + +Use this function when different sequencing depths could affect a comparison +of repertoire diversity or composition. In single-cell data, the sampling +unit is a cell barcode and all selected chains from that cell stay together. +In bulk data with abundance values, the sampling unit is one sequence count. + +The function returns a new \link{ImmunData} object. The original object is not +changed. +} +\section{Meaning of \code{n} for single-cell data}{ + +\itemize{ +\item \verb{0 < n < 1} keeps \verb{floor(n * number of cells)} cells from each repertoire. +\item \code{n >= 1} keeps \code{n} cells from each repertoire. +} + +Cell barcodes are sampled without replacement. For paired receptors, all +retained chains belonging to a selected cell stay together. +} + +\section{Meaning of \code{n} for bulk data}{ + +\itemize{ +\item \verb{0 < n < 1} keeps \verb{floor(n * total abundance)} sequence counts from each +repertoire. +\item \code{n >= 1} keeps a total abundance of \code{n} from each repertoire. +} + +Counts are sampled without replacement according to their observed +abundance. A retained receptor can therefore have a smaller abundance than +it had before downsampling. For example, \code{n = 1000} makes the total retained +abundance equal to 1000 in every repertoire that originally contained at +least 1000 counts. + +If a requested whole-number \code{n} is larger than a repertoire, that repertoire +is returned unchanged and the function gives a warning. If a fraction is so +small that it selects zero units in any repertoire, the function stops and +asks for a larger value. +} + +\section{Repertoire and strata summaries}{ + + +When repertoires are defined, the function recalculates receptor counts, +proportions, repertoire sizes, and the number of repertoires containing each +receptor. Existing strata are also rebuilt, and their labels are retained. +When repertoires are not defined, the complete dataset is treated as one +sampling group and no repertoire summary is added. +} + +\section{Backend and storage}{ + + +Chain-level selection and reconstruction use the duckplyr annotation table. +The small table of sampling units is collected into R for random sampling. +The function does not overwrite the stored input object. Use +\code{\link[=write_immundata]{write_immundata()}} to save the returned object. +} + +\examples{ +library(immundata) +library(dplyr) + +options(immundata.verbose = FALSE) + +# Create two small bulk T-cell repertoires with different total abundances. +bulk_data <- tibble( + Sample = c("Tumor", "Tumor", "Blood", "Blood"), + cdr3_aa = c("CASSA", "CASSB", "CASSA", "CASSC"), + v_call = c("TRBV1", "TRBV2", "TRBV1", "TRBV3"), + abundance = c(20L, 5L, 4L, 6L) +) +bulk_file <- tempfile(fileext = ".tsv") +readr::write_tsv(bulk_data, bulk_file) + +idata <- read_repertoires( + path = bulk_file, + schema = c("cdr3_aa", "v_call"), + count_col = "abundance", + repertoire_schema = "Sample", + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL, + output_folder = tempfile("immundata-downsample-") +) + +before <- idata$repertoires |> + select(Sample, n_barcodes) |> + rename(before = n_barcodes) + +sampled <- downsample_immundata(idata, n = 5, seed = 42) + +before |> + left_join( + sampled$repertoires |> + select(Sample, n_barcodes) |> + rename(after = n_barcodes), + by = "Sample" + ) |> + arrange(Sample) +# Expected result: +# Sample before after +# Blood 10 5 +# Tumor 25 5 + +# Each returned repertoire has five sequence counts. `idata` still has its +# original repertoire sizes of 10 and 25. + +} +\seealso{ +\code{\link[=agg_repertoires]{agg_repertoires()}}, \code{\link[=filter_immundata]{filter_immundata()}}, \code{\link[=write_immundata]{write_immundata()}} +} +\concept{filtering} diff --git a/man/filter_immundata.Rd b/man/filter_immundata.Rd index 1bb844b..2d41dcd 100644 --- a/man/filter_immundata.Rd +++ b/man/filter_immundata.Rd @@ -5,7 +5,7 @@ \alias{filter.ImmunData} \alias{filter_barcodes} \alias{filter_receptors} -\title{Filter ImmunData by receptor features, barcodes or any annotations} +\title{Keep selected rows or receptors in ImmunData} \usage{ filter_immundata(idata, ..., seq_options = NULL, keep_repertoires = TRUE) @@ -23,162 +23,178 @@ filter_barcodes(idata, barcodes, keep_repertoires = TRUE) filter_receptors(idata, receptors, keep_repertoires = TRUE) } \arguments{ -\item{idata, .data}{An \code{ImmunData} object.} +\item{idata, .data}{An \link{ImmunData} object.} -\item{...}{For \code{filter}, these are regular \code{dplyr}-style filtering -expressions (e.g., \code{V_gene == "IGHV1-1"}, \code{chain == "IGH"}) applied to the -\verb{$annotations} table \emph{before} sequence filtering. Ignored by \code{filter_barcodes} -and \code{filter_receptors}.} +\item{...}{One or more conditions used to keep rows. Refer to annotation +columns directly by name. Multiple conditions are combined with \code{&}. +Conditions are applied before sequence matching.} -\item{seq_options}{For \code{filter}, an optional named list specifying sequence-based -filtering options. Use \code{\link[=make_seq_options]{make_seq_options()}} for convenient creation. -The list can contain: -\itemize{ -\item \code{query_col} (Character scalar): The name of the column in \verb{$annotations} -containing sequences to compare (e.g., \code{"CDR3_aa"}, \code{"FR1_nt"}). -\item \code{patterns} (Character vector): A vector of sequences or regular expressions -to match against \code{query_col}. -\item \code{method} (Character scalar): The matching method. One of \code{"exact"}, -\code{"regex"}, \code{"lev"} (Levenshtein distance), or \code{"hamm"} (Hamming distance). -Defaults typically handled by \code{make_seq_options}. -\item \code{max_dist} (Numeric scalar): For fuzzy methods (\code{"lev"}, \code{"hamm"}), the -maximum allowed distance. Rows with distance <= \code{max_dist} are kept. -Defaults typically handled by \code{make_seq_options}. -\item \code{name_type} (Character scalar): Determines column names in intermediate distance -calculations if applicable (\code{"index"} or \code{"pattern"}). Passed through to -internal annotation functions. Defaults typically handled by \code{make_seq_options}. -If \code{seq_options} is \code{NULL} (the default), no sequence-based filtering is performed. -}} - -\item{keep_repertoires}{Logical scalar. If \code{TRUE} (the default) and the input -\code{idata} has repertoire information (\code{idata$schema_repertoire} is not \code{NULL}), -the repertoire summaries will be recalculated based on the filtered data using -\code{\link[=agg_repertoires]{agg_repertoires()}}. If \code{FALSE}, or if no repertoire schema exists, the -returned \code{ImmunData} object will not contain repertoire summaries (\verb{$repertoires} -will be \code{NULL}).} - -\item{.by}{Not used.} - -\item{.preserve}{Not used.} - -\item{barcodes}{For \code{filter_barcodes}, a vector of cell identifiers (barcodes) -to keep. Can be character, integer, or numeric.} - -\item{receptors}{For \code{filter_receptors}, a vector of receptor identifiers -to keep. Can be character, integer, or numeric.} +\item{seq_options}{Options for matching sequences with reference sequences or +patterns. Create these options with \code{\link[=make_seq_options]{make_seq_options()}}. If \code{NULL}, the +default, no sequence matching is performed.} + +\item{keep_repertoires}{If \code{TRUE}, the default, existing repertoire and strata +summaries are recalculated from the filtered data. If \code{FALSE}, the returned +object does not contain these summaries.} + +\item{.by, .preserve}{Accepted for compatibility with \code{\link[dplyr:filter]{dplyr::filter()}}, but +currently not used for \link{ImmunData} objects.} + +\item{barcodes}{A character, integer, or numeric vector of cell barcodes to +keep with \code{filter_barcodes()}.} + +\item{receptors}{A character, integer, or numeric vector of receptor +identifiers to keep with \code{filter_receptors()}.} } \value{ -A new \code{ImmunData} object containing only the filtered annotations -(and potentially recalculated repertoire summaries). The schema remains the same. +A new \link{ImmunData} object containing the selected rows and receptors. +If requested, repertoire and strata summaries are recalculated for the +selected data. } \description{ -Provides flexible filtering options for an \code{ImmunData} object. +Use \code{filter()} to keep selected rows in an \link{ImmunData} object. For example, +you can keep rows from one response group, rows using a selected V gene, or +receptors containing a CDR3 sequence similar to a reference sequence. -\code{filter()} is the main function, allowing filtering based on receptor features -(e.g., CDR3 sequence) using various matching methods (exact, regex, fuzzy) and/or -standard \code{dplyr}-style filtering on annotation columns. +The function returns a new \link{ImmunData} object. The original object is not +changed. -\code{filter_barcodes()} is a convenience function to filter by specific cell barcodes. +This function is a direct implementation of \link[dplyr:filter]{dplyr::filter}. Alternative +function name is \code{filter_immundata}. -\code{filter_receptors()} is a convenience function to filter by specific receptor identifiers. +Use \code{filter_barcodes()} to keep selected cell barcodes and +\code{filter_receptors()} to keep selected receptor identifiers. } \details{ -For \code{filter}: +You can filter an \link{ImmunData} object in three ways: \itemize{ -\item User-provided \code{dplyr}-style filters (\code{...}) are applied \emph{before} any sequence-based -filtering defined in \code{seq_options}. -\item Sequence filtering compares values in the \code{query_col} of the annotations table -against the provided \code{patterns}. -\item Supported sequence matching methods are: -\itemize{ -\item \code{"exact"}: Keeps rows where \code{query_col} exactly matches any of the \code{patterns}. -\item \code{"regex"}: Keeps rows where \code{query_col} matches any of the regular expressions -in \code{patterns}. -\item \code{"lev"} (Levenshtein distance): Keeps rows where the edit distance between -\code{query_col} and any pattern is less than or equal to \code{max_dist}. -\item \code{"hamm"} (Hamming distance): Keeps rows where the Hamming distance (for -equal length strings) between \code{query_col} and any pattern is less than -or equal to \code{max_dist}. -} -\item The filtering operations act on the \verb{$annotations} table. A new \code{ImmunData} -object is created containing only the rows (and corresponding receptors) -that pass the filter(s). -\item If \code{keep_repertoires = TRUE} (and repertoire data exists in the input), -the repertoire-level summaries (\verb{$repertoires} table) are recalculated based -on the filtered annotations. Otherwise, the \verb{$repertoires} table in the -output will be \code{NULL}. +\item Supply conditions in \code{...} to filter using annotation columns. Refer to +columns directly by name. For example, \code{Response == "FR"} keeps rows from +the \code{FR} response group. +\item Supply \code{seq_options}, created with \code{\link[=make_seq_options]{make_seq_options()}}, to find receptors +containing a sequence that matches one or more reference sequences or +patterns. +\item Use \code{filter_barcodes()} or \code{filter_receptors()} when you already have the +identifiers that you want to keep. } -For \code{filter_barcodes} and \code{filter_receptors}: -\itemize{ -\item These functions provide a simpler interface for common filtering tasks based on -cell barcodes or receptor IDs, respectively. They use efficient \code{semi_join} -operations internally. -} -} -\examples{ -# Basic setup (assuming idata_test is a valid ImmunData object) -# print(idata_test) - -# --- filter examples --- -\dontrun{ -# Example 1: dplyr-style filtering on annotations -filtered_heavy <- filter(idata_test, chain == "IGH") -print(filtered_heavy) - -# Example 2: Exact sequence matching on CDR3 amino acid sequence -cdr3_patterns <- c("CARGLGLVFYGMDVW", "CARDNRGAVAGVFGEAFYW") -seq_opts_exact <- make_seq_options(query_col = "CDR3_aa", patterns = cdr3_patterns) -filtered_exact_cdr3 <- filter(idata_test, seq_options = seq_opts_exact) -print(filtered_exact_cdr3) - -# Example 3: Combining dplyr-style and fuzzy sequence matching (Levenshtein) -seq_opts_lev <- make_seq_options( - query_col = "CDR3_aa", - patterns = "CARGLGLVFYGMDVW", - method = "lev", - max_dist = 1 -) -filtered_combined <- filter(idata_test, - chain == "IGH", - C_gene == "IGHG1", - seq_options = seq_opts_lev -) -print(filtered_combined) - -# Example 4: Regex matching on V gene -v_gene_pattern <- "^IGHV[13]-" # Keep only IGHV1 or IGHV3 families -seq_opts_regex <- make_seq_options( - query_col = "V_gene", - patterns = v_gene_pattern, - method = "regex" -) -filtered_regex_v <- filter(idata_test, seq_options = seq_opts_regex) -print(filtered_regex_v) +Conditions in \code{...} are applied before sequence matching. Sequence matching +then identifies receptors from the remaining rows. When one chain matches, +all remaining chains belonging to the same receptor are kept. A chain removed +by a condition in \code{...} is not added back by sequence matching. -# Example 5: Filtering without recalculating repertoires -filtered_no_rep <- filter(idata_test, chain == "IGK", keep_repertoires = FALSE) -print(filtered_no_rep) # $repertoires should be NULL +Sequence matching methods are: +\itemize{ +\item \code{"exact"}: the sequence must be identical to one of the references. +\item \code{"regex"}: the sequence must match a regular-expression pattern. This is +an advanced option for matching text patterns. +\item \code{"lev"}: the Levenshtein distance counts the substitutions, insertions, or +deletions needed to change one sequence into the other. +\item \code{"hamm"}: the Hamming distance counts different positions between +sequences of the same length. Sequences of different lengths do not match. } -# --- filter_barcodes example --- -\dontrun{ -# Assuming 'cell1_barcode' and 'cell5_barcode' exist in idata_test$annotations$cell_id -specific_barcodes <- c("cell1_barcode", "cell5_barcode") -filtered_cells <- filter_barcodes(idata_test, barcodes = specific_barcodes) -print(filtered_cells) -} +For \code{"lev"} and \code{"hamm"}, provide \code{max_dist}. A sequence is accepted when +its distance from at least one reference is less than or equal to this value. +A distance of \code{0} means an exact match, and smaller values mean more similar +sequences. -# --- filter_receptors example --- -\dontrun{ -# Assuming receptor IDs 101 and 205 exist in idata_test$annotations$receptor_id -specific_receptors <- c(101, 205) # Or character IDs if applicable -filtered_recs <- filter_receptors(idata_test, receptors = specific_receptors) -print(filtered_recs) +By default, existing repertoire summaries are recalculated from the filtered +data. Existing strata are also rebuilt, and their labels are retained. Set +\code{keep_repertoires = FALSE} to return an object without repertoire or strata +summaries. } +\examples{ +library(immundata) +library(dplyr) + +options(immundata.verbose = FALSE) + +# Load data included with immundata +idata <- get_test_idata() + +# Keep rows from one response group +fr_response <- idata |> + filter(Response == "FR") + +fr_response |> + collect() |> + summarise( + n_rows = n(), + n_receptors = n_distinct(imd_receptor_id) + ) +# Expected result: +# n_rows n_receptors +# 955 871 + +# Keep receptors containing one reference CDR3 sequence +reference_cdr3 <- "ASFPVLSPYNEQF" + +exact_match <- idata |> + filter( + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = reference_cdr3, + method = "exact" + ) + ) + +exact_match |> + collect() |> + select(cdr3_aa, v_call, Response) +# Expected result: +# cdr3_aa v_call Response +# ASFPVLSPYNEQF TRBV28*01 FR + +# Keep receptors within four sequence changes of the reference +similar_sequences <- idata |> + filter( + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = reference_cdr3, + method = "lev", + max_dist = 4 + ) + ) + +similar_sequences |> + collect() |> + distinct(cdr3_aa) |> + arrange(cdr3_aa) +# Expected result: +# cdr3_aa +# ASFPVLSPYNEQF +# ASSPDSPSYNEQF +# ASSPGLAAYNEQF +# ASSPTLYNEQF + +# Keep two selected cell barcodes +selected_barcodes <- c("S1_1", "S1_2") + +selected_cells <- idata |> + filter_barcodes(selected_barcodes) + +selected_cells |> + collect() |> + distinct(imd_barcode) +# Expected result: +# imd_barcode +# S1_1 +# S1_2 + +# The same approach can keep selected receptor identifiers +selected_receptors <- idata |> + collect() |> + distinct(imd_receptor_id) |> + slice_head(n = 2) |> + pull(imd_receptor_id) + +selected_receptors_data <- idata |> + filter_receptors(selected_receptors) } \seealso{ -\code{\link[=make_seq_options]{make_seq_options()}}, \code{\link[dplyr:filter]{dplyr::filter()}}, \code{\link[=agg_repertoires]{agg_repertoires()}}, \link{ImmunData} +\code{\link[dplyr:filter]{dplyr::filter()}}, \code{\link[=make_seq_options]{make_seq_options()}}, \code{\link[=mutate_immundata]{mutate_immundata()}}, +\code{\link[=agg_repertoires]{agg_repertoires()}}, \link{ImmunData} } \concept{filtering} diff --git a/man/from_immunarch.Rd b/man/from_immunarch.Rd index 37dc273..d2492cc 100644 --- a/man/from_immunarch.Rd +++ b/man/from_immunarch.Rd @@ -35,7 +35,7 @@ immunarch object, with data saved under \code{output_folder}. \description{ The \code{from_immunarch()} function takes an \strong{immunarch} object (as returned by \code{immunarch::repLoad()}), writes each repertoire to a TSV file with an added -\code{filename} column in a specified folder, and then imports those files into +internal filename column in a specified folder, and then imports those files into an \strong{ImmunData} object via \code{read_repertoires()}. } \examples{ diff --git a/man/get_test_idata.Rd b/man/get_test_idata.Rd index 30a872e..6e7a036 100644 --- a/man/get_test_idata.Rd +++ b/man/get_test_idata.Rd @@ -1,5 +1,5 @@ % Generated by roxygen2: do not edit by hand -% Please edit documentation in R/test_utils.R +% Please edit documentation in R/utils_test.R \name{get_test_idata} \alias{get_test_idata} \title{Get test datasets from \code{immundata}} diff --git a/man/get_test_immundata.Rd b/man/get_test_immundata.Rd deleted file mode 100644 index 85df0fc..0000000 --- a/man/get_test_immundata.Rd +++ /dev/null @@ -1,12 +0,0 @@ -% Generated by roxygen2: do not edit by hand -% Please edit documentation in R/test_utils.R -\name{get_test_immundata} -\alias{get_test_immundata} -\title{Get test datasets from \code{immundata}} -\usage{ -get_test_immundata() -} -\description{ -Get test datasets from \code{immundata} -} -\keyword{internal} diff --git a/man/imd_input_columns.Rd b/man/imd_input_columns.Rd new file mode 100644 index 0000000..6c59d7c --- /dev/null +++ b/man/imd_input_columns.Rd @@ -0,0 +1,53 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/globals.R +\name{imd_rename_cols} +\alias{imd_rename_cols} +\alias{imd_drop_cols} +\title{Get input-column presets} +\usage{ +imd_rename_cols(format = "default") + +imd_drop_cols(format = "airr") +} +\arguments{ +\item{format}{A character string. The input format preset. For +\code{imd_rename_cols()}, use \code{"default"} or \code{"10x"}; the default is +\code{"default"}. For \code{imd_drop_cols()}, use \code{"universal"}, \code{"airr"}, or +\code{"10x"}; the default is \code{"airr"}.} +} +\value{ +\code{imd_rename_cols()} returns a named character vector in the form +\code{c(new_name = "source_name")}. \code{imd_drop_cols()} returns a character +vector of source columns to remove. +} +\description{ +Use these helpers to inspect or customize the column renaming and removal +presets used by \code{\link[=read_repertoires]{read_repertoires()}}. + +\code{imd_rename_cols()} returns mappings from standard output names to source +names. \code{imd_drop_cols()} returns technical columns that can usually be +removed before receptors are defined. These functions return definitions +only; they do not change input files or an \link{ImmunData} object. +} +\examples{ +imd_rename_cols("10x") +# Includes c(v_call = "v_gene", locus = "chain"). + +head(imd_drop_cols("10x"), 3) +# Expected result: +# "full_length" "is_cell" "contig_id" + +# Keep the 10x `contig_id` column while dropping the other default columns. +columns_to_drop <- setdiff(imd_drop_cols("10x"), "contig_id") +custom_preprocessing <- list( + exclude_columns = make_exclude_columns(columns_to_drop), + filter_nonproductive = make_productive_filter( + truthy = c("TRUE", "true", "1") + ) +) + +} +\seealso{ +\code{\link[=read_repertoires]{read_repertoires()}}, \code{\link[=make_default_preprocessing]{make_default_preprocessing()}}, \code{\link[=imd_schema]{imd_schema()}} +} +\concept{ingestion} diff --git a/man/imd_schema.Rd b/man/imd_schema.Rd index 2981e18..cb8145f 100644 --- a/man/imd_schema.Rd +++ b/man/imd_schema.Rd @@ -2,44 +2,46 @@ % Please edit documentation in R/globals.R \name{imd_schema} \alias{imd_schema} -\alias{imd_schema_sym} -\alias{imd_meta_schema} -\alias{imd_files} -\alias{imd_rename_cols} -\alias{imd_drop_cols} -\alias{imd_repertoire_schema} -\alias{imd_receptor_features} -\alias{imd_receptor_chains} -\title{Get Immundata internal schema field names} +\title{Get a standard ImmunData column name} \usage{ imd_schema(key = NULL) - -imd_schema_sym(key = NULL) - -imd_meta_schema() - -imd_files() - -imd_rename_cols(format = "default") - -imd_drop_cols(format = "airr") - -imd_repertoire_schema(format = "airr") - -imd_receptor_features(schema) - -imd_receptor_chains(schema) } \arguments{ -\item{key}{Character which field to return.} +\item{key}{A character string or \code{NULL}. One schema key, for example +\code{"barcode"}, \code{"receptor"}, \code{"repertoire"}, \code{"count"}, or +\code{"proportion"}. Use \code{NULL}, the default, to return all available keys and +column names.} +} +\value{ +If \code{key} is supplied, one character string containing the standard +column name. If \code{key = NULL}, a named list of all schema keys and column +names. +} +\description{ +Use \code{imd_schema()} when code needs the standard column name for an +\code{ImmunData} identifier or calculated value, such as the cell barcode, +receptor identifier, repertoire identifier, count, or proportion. -\item{format}{Character what format to load - "airr" or "10x".} +Use this helper in reusable analysis code or package extensions instead of +writing an internal name such as \code{"imd_barcode"} directly. It only returns +names; it does not inspect or change an \link{ImmunData} object. +} +\examples{ +imd_schema("barcode") +# Expected result: "imd_barcode" + +imd_schema("receptor") +# Expected result: "imd_receptor_id" + +# Use a returned name for programmatic selection. +barcode_column <- imd_schema("barcode") +get_test_idata() |> + dplyr::collect() |> + dplyr::select(dplyr::all_of(barcode_column)) |> + head(2) -\item{schema}{Receptor schema from \code{\link[=make_receptor_schema]{make_receptor_schema()}}.} } -\description{ -Returns the standardized field names used across Immundata objects and processing functions, -as defined in \code{IMD_GLOBALS$schema}. These include column names for cell ids or barcodes, receptors, -repertoires, and related metadata. +\seealso{ +\code{\link[=make_receptor_schema]{make_receptor_schema()}}, \code{\link[=imd_rename_cols]{imd_rename_cols()}}, \link{ImmunData} } \concept{schema} diff --git a/man/imd_schema_sym.Rd b/man/imd_schema_sym.Rd new file mode 100644 index 0000000..abab6d0 --- /dev/null +++ b/man/imd_schema_sym.Rd @@ -0,0 +1,76 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/globals.R +\name{imd_schema_sym} +\alias{imd_schema_sym} +\alias{imd_meta_schema} +\alias{imd_files} +\alias{imd_repertoire_schema} +\alias{imd_receptor_features} +\alias{imd_receptor_chains} +\title{Developer helpers for ImmunData schemas and storage} +\usage{ +imd_schema_sym(key = NULL) + +imd_meta_schema() + +imd_files() + +imd_repertoire_schema(format = "airr") + +imd_receptor_features(schema) + +imd_receptor_chains(schema) +} +\arguments{ +\item{key}{A character string or \code{NULL}. For \code{imd_schema_sym()}, one schema +key accepted by \code{\link[=imd_schema]{imd_schema()}}. Use \code{NULL}, the default, to return the +complete named schema list.} + +\item{format}{A character string. For \code{imd_repertoire_schema()}, the preset +name. Currently only \code{"airr"} is accepted.} + +\item{schema}{A receptor-schema list. For \code{imd_receptor_features()} and +\code{imd_receptor_chains()}, a schema created by \code{\link[=make_receptor_schema]{make_receptor_schema()}}.} +} +\value{ +\itemize{ +\item \code{imd_schema_sym()} returns an rlang symbol for one standard column. With +\code{key = NULL}, it returns the complete named schema list. +\item \code{imd_meta_schema()} returns a named list of fields used in +\code{metadata.json}. +\item \code{imd_files()} returns a named list of standard snapshot file names. +\item \code{imd_repertoire_schema()} returns the configured preset for \code{format}, or +\code{NULL} when no preset is configured. +\item \code{imd_receptor_features()} returns the character vector in +\code{schema$features}. +\item \code{imd_receptor_chains()} returns the character vector in \code{schema$chains}, or +\code{NULL} for a chain-agnostic schema. +} +} +\description{ +These helpers expose package constants for extension developers. They are not +needed for routine biological analysis. Use \code{\link[=imd_schema]{imd_schema()}} for a standard +column name and \code{\link[=make_receptor_schema]{make_receptor_schema()}} to define biological receptors. + +The functions remain exported for compatibility with packages that extend +\code{immundata}, but their values describe implementation details and may grow as +the storage format develops. +} +\examples{ +schema <- make_receptor_schema( + features = c("junction_aa", "v_call"), + chains = c("TRA", "TRB") +) + +imd_receptor_features(schema) +# Expected result: c("junction_aa", "v_call") + +imd_receptor_chains(schema) +# Expected result: c("TRA", "TRB") + +imd_files() +# Lists the standard metadata and Parquet file names. + +} +\concept{schema} +\keyword{internal} diff --git a/man/immundata-package.Rd b/man/immundata-package.Rd index ca01045..32531c0 100644 --- a/man/immundata-package.Rd +++ b/man/immundata-package.Rd @@ -20,5 +20,10 @@ Useful links: \author{ \strong{Maintainer}: Vadim I. Nazarov \email{support@immunomind.com} (\href{https://orcid.org/0000-0003-3659-2709}{ORCID}) +Authors: +\itemize{ + \item Vadim I. Nazarov \email{support@immunomind.com} (\href{https://orcid.org/0000-0003-3659-2709}{ORCID}) +} + } \keyword{internal} diff --git a/man/make_receptor_schema.Rd b/man/make_receptor_schema.Rd index 6dd4103..0337ecb 100644 --- a/man/make_receptor_schema.Rd +++ b/man/make_receptor_schema.Rd @@ -1,10 +1,10 @@ % Generated by roxygen2: do not edit by hand -% Please edit documentation in R/operations_agg.R +% Please edit documentation in R/utils_schema.R \name{make_receptor_schema} \alias{make_receptor_schema} \alias{assert_receptor_schema} \alias{test_receptor_schema} -\title{Create or validate a receptor schema object} +\title{Define which chain observations form the same receptor} \usage{ make_receptor_schema(features, chains = NULL) @@ -13,31 +13,116 @@ assert_receptor_schema(schema) test_receptor_schema(schema) } \arguments{ -\item{features}{Character vector. Column names defining the features of a -single receptor chain (e.g., V gene, J gene, CDR3 sequence).} +\item{features}{A non-empty character vector. Column names containing the +chain fields that must match, such as +\code{c("junction_aa", "v_call", "j_call")}. Use names as they appear after any +input-column renaming.} -\item{chains}{Optional character vector (max length 2). Locus names (e.g., -\code{"TRA"}, \code{"TRB"}) to filter by or pair. If \code{NULL} or length 1, only -filtering occurs. If length 2, pairing logic is enabled in \code{\link[=agg_receptors]{agg_receptors()}}. -Default: \code{NULL}.} +\item{chains}{A character vector of length one or two, or \code{NULL}. Use one +value, such as \code{"TRB"}, to keep one chain; two values, such as +\code{c("TRA", "TRB")}, to define a strict pair; or the \code{"IGK|IGL"} syntax in +the second value to accept either alternative. The default is \code{NULL}, which +does not select loci.} -\item{schema}{An object to test or assert as a valid schema. Can be a list -created by \code{make_receptor_schema} or a character vector (for \code{test_receptor_schema}).} +\item{schema}{A non-empty character vector or receptor-schema list. An object +to check. A schema created by \code{make_receptor_schema()} is accepted. A +character vector supplies feature names for a chain-agnostic schema.} } \value{ -\code{make_receptor_schema} returns a list with elements \code{features} and \code{chains}. -\code{assert_receptor_schema} returns \code{TRUE} invisibly if valid, or stops execution. -\code{test_receptor_schema} returns \code{TRUE} or \code{FALSE}. +\code{make_receptor_schema()} returns a list with character elements +\code{features} and \code{chains}; \code{chains} is \code{NULL} when loci are not selected. +\code{assert_receptor_schema()} returns \code{TRUE} for accepted input and otherwise +stops with an error. \code{test_receptor_schema()} returns one logical value. } \description{ -Helper functions for defining and validating the \code{schema} used by -\code{\link[=agg_receptors]{agg_receptors()}} to identify unique receptors. - -\code{make_receptor_schema()} creates a schema list object. -\code{assert_receptor_schema()} checks if an object is a valid schema list and throws -an error if not. -\code{test_receptor_schema()} checks if an object is a valid schema list or a -character vector (which \code{agg_receptors} can also accept) and returns \code{TRUE} -or \code{FALSE}. +Use \code{make_receptor_schema()} to define a biological receptor from sequence +features and one or two receptor chains. + +Use this function when reading single-cell data with one selected chain, when +pairing chains such as TRA-TRB, or when accepting alternative light chains +such as IGK or IGL. The unit being defined is the receptor. Creating a schema +does not change any data or an existing \link{ImmunData} object. +} +\section{When observations are the same receptor}{ + + +\code{features} names the fields that define chain identity. Common choices are +the CDR3 amino acid sequence, V gene, and J gene. Two observations represent +the same receptor only when the relevant chain loci and every selected +feature match. +\itemize{ +\item With one chain, such as \code{chains = "TRB"}, only that locus is used. Two TRB +observations are the same receptor when all their selected feature values +match. +\item With a strict pair, such as \code{chains = c("TRA", "TRB")}, chains are first +paired within each cell barcode. Receptors from two cells are the same only +when every selected TRA feature and every selected TRB feature match. +\item With an alternative second chain, such as +\code{chains = c("IGH", "IGK|IGL")}, each receptor must contain IGH and exactly +one of IGK or IGL. Cells containing both IGK and IGL are excluded. The +light-chain locus and all selected heavy- and light-chain features must +match for two observations to be the same receptor. +} + +A barcode determines which chains belong to the same cell; it does not by +itself define receptor identity across cells. During single-cell import, +\code{\link[=read_repertoires]{read_repertoires()}} uses \code{umi_col} to choose one chain when a cell contains +several observations from the same locus. + +Use \code{chains = NULL} for chain-agnostic bulk or pre-filtered data. In that +case, only the values in \code{features} define receptor identity. +} + +\section{Validate a schema}{ + + +\code{assert_receptor_schema()} stops with an error if \code{schema} is not accepted. +Use it inside another function when invalid input must stop the calculation. +\code{test_receptor_schema()} returns one \code{TRUE} or \code{FALSE} value and is useful in +conditional code. +} + +\section{Backend and storage}{ + + +A receptor schema is a small R list containing \code{features} and \code{chains}. It +stores no sequence data. \code{\link[=read_repertoires]{read_repertoires()}} and \code{\link[=agg_receptors]{agg_receptors()}} apply the +schema to chain observations using duckplyr. +} + +\examples{ +# Single-chain TCR: compare TRB observations by CDR3, V gene, and J gene. +trb_schema <- make_receptor_schema( + features = c("junction_aa", "v_call", "j_call"), + chains = "TRB" +) +trb_schema +# Expected result: +# $features: "junction_aa" "v_call" "j_call" +# $chains: "TRB" + +# Paired alpha-beta TCR: all selected fields must match on both TRA and TRB. +ab_tcr_schema <- make_receptor_schema( + features = c("junction_aa", "v_call", "j_call"), + chains = c("TRA", "TRB") +) +ab_tcr_schema +# The result defines one receptor as a matched TRA-TRB pair from one cell. + +# BCR: require IGH and accept either an IGK or IGL light chain. +bcr_schema <- make_receptor_schema( + features = c("junction_aa", "v_call", "j_call"), + chains = c("IGH", "IGK|IGL") +) +bcr_schema +# The result accepts IGH-IGK and IGH-IGL receptors, while keeping the two +# light-chain loci biologically distinct. + +test_receptor_schema(bcr_schema) +# Expected result: TRUE + +} +\seealso{ +\code{\link[=read_repertoires]{read_repertoires()}}, \code{\link[=agg_receptors]{agg_receptors()}}, \code{\link[=imd_schema]{imd_schema()}} } \concept{utils} diff --git a/man/make_seq_options.Rd b/man/make_seq_options.Rd index c2d0950..7e5b4b6 100644 --- a/man/make_seq_options.Rd +++ b/man/make_seq_options.Rd @@ -1,8 +1,8 @@ % Generated by roxygen2: do not edit by hand -% Please edit documentation in R/operations_utils.R +% Please edit documentation in R/utils_seq.R \name{make_seq_options} \alias{make_seq_options} -\title{Build a \code{seq_options} list for sequence‑based receptor filtering} +\title{Create options for comparing receptor sequences} \usage{ make_seq_options( query_col, @@ -13,30 +13,35 @@ make_seq_options( ) } \arguments{ -\item{query_col}{Character(1). Name of the receptor column to compare -(e.g. \code{"cdr3_aa"}).} +\item{query_col}{Name of the sequence column to compare, such as \code{"cdr3_aa"}.} -\item{patterns}{Character vector of sequences or regular expressions to -search for.} +\item{patterns}{One or more reference sequences or regular-expression +patterns.} -\item{method}{One of \code{"exact"}, \code{"regex"}, \code{"lev"} (Levenshtein), or -\code{"hamm"} (Hamming). Defaults to \code{"exact"}.} +\item{method}{Comparison method: \code{"exact"}, \code{"regex"}, \code{"lev"} +(Levenshtein distance), or \code{"hamm"} (Hamming distance). The default is +\code{"exact"}.} -\item{max_dist}{Numeric distance threshold for \code{"lev"} / \code{"hamm"} -filtering. Use \code{NA} (default) to keep all rows after annotation.} +\item{max_dist}{Maximum distance accepted by \code{\link[=filter_immundata]{filter_immundata()}} when +\code{method = "lev"} or \code{method = "hamm"}. A value is required when filtering +with either distance method. This argument has no effect on +\code{\link[=mutate_immundata]{mutate_immundata()}}, which reports every calculated distance.} -\item{name_type}{Passed straight to \code{annotate_tbl_distance()}; either -\code{"index"} (default) or \code{"pattern"}.} +\item{name_type}{How result columns created by \code{\link[=mutate_immundata]{mutate_immundata()}} are named. +\code{"index"}, the default, creates short numbered names. \code{"pattern"} includes +the reference pattern in each name. This argument does not change which +receptors are kept by \code{\link[=filter_immundata]{filter_immundata()}}.} } \value{ -A named list suitable for the \code{seq_options} argument of -\code{\link[=filter_receptors]{filter_receptors()}}. +A named list for the \code{seq_options} argument of \code{\link[=filter_immundata]{filter_immundata()}} or +\code{\link[=mutate_immundata]{mutate_immundata()}}. } \description{ -A convenience wrapper that validates the common arguments for -\strong{\code{filter_receptors()}} and returns them in the required list form. +Create sequence comparison options for the \code{seq_options} argument of +\code{\link[=filter_immundata]{filter_immundata()}} or \code{\link[=mutate_immundata]{mutate_immundata()}}. Use these options to compare a +sequence column with one or more reference sequences or patterns. } \seealso{ -\code{\link[=filter_receptors]{filter_receptors()}}, \code{\link[=annotate_receptors]{annotate_receptors()}} +\code{\link[=filter_immundata]{filter_immundata()}}, \code{\link[=mutate_immundata]{mutate_immundata()}}, \code{\link[=annotate_receptors]{annotate_receptors()}} } \concept{utils} diff --git a/man/mutate_immundata.Rd b/man/mutate_immundata.Rd index 1cd29b0..e5d6d29 100644 --- a/man/mutate_immundata.Rd +++ b/man/mutate_immundata.Rd @@ -3,132 +3,261 @@ \name{mutate_immundata} \alias{mutate_immundata} \alias{mutate.ImmunData} -\title{Modify or Add Columns to ImmunData Annotations} +\title{Add or change annotation columns in ImmunData} \usage{ -mutate_immundata(idata, ..., seq_options = NULL) +mutate_immundata(idata, ..., .by = NULL, seq_options = NULL) -\method{mutate}{ImmunData}(.data, ..., seq_options = NULL) +\method{mutate}{ImmunData}(.data, ..., .by = NULL, seq_options = NULL) } \arguments{ -\item{idata, .data}{An \code{ImmunData} object.} - -\item{...}{\code{dplyr::mutate}-style named expressions (e.g., \code{new_col = existing_col * 2}, -\code{category = ifelse(value > 10, "high", "low")}). These are applied first. -\strong{Important}: You cannot use names for new or modified columns that conflict -with the core \code{ImmunData} schema columns (retrieved via \code{imd_schema()}).} - -\item{seq_options}{Optional named list specifying sequence-based annotation options. -Use \code{\link[=make_seq_options]{make_seq_options()}} for convenient creation. See \code{filter_immundata} -documentation (\code{?filter_immundata}) or the details section here for the list -structure (\code{query_col}, \code{patterns}, \code{method}, \code{name_type}). \code{max_dist} is -ignored for mutation. If \code{NULL} (the default), no sequence-based columns are added.} +\item{idata, .data}{An \link{ImmunData} object.} + +\item{...}{One or more named calculations in the form +\code{new_column = calculation}. Refer to existing columns directly by name. You +can add new annotation columns or change columns that are not protected.} + +\item{.by}{Optional columns used to form temporary groups for this operation. +For example, \code{.by = Response} calculates separately for each response, and +\code{.by = c(Response, imd_group_id)} uses each response and receptor-cluster +combination. The grouping applies only to this \code{mutate()} call.} + +\item{seq_options}{Options for comparing sequences with reference sequences or +patterns. Create these options with \code{\link[=make_seq_options]{make_seq_options()}}. If \code{NULL}, the +default, no sequence comparisons are performed.} } \value{ -A \emph{new} \code{ImmunData} object with the \verb{$annotations} table modified according -to the provided expressions and \code{seq_options}. The \verb{$repertoires} table (if present) -is carried over unchanged from the input \code{idata}. +A new \link{ImmunData} object containing the added or changed annotation +columns. Existing repertoire and strata summaries are preserved. } \description{ -Applies transformations to the \verb{$annotations} table within an \code{ImmunData} -object, similar to \code{dplyr::mutate}. It allows adding new columns or modifying -existing non-schema columns using standard \code{dplyr} expressions. Additionally, -it can add new columns based on sequence comparisons (exact match, regular -expression matching, or distance calculation) against specified patterns. +Use \code{mutate()} to add information to each row of an \link{ImmunData} object. For +example, you can calculate CDR3 length, mark sequences of interest, or compare +receptor sequences with reference sequences. + +The function returns a new \link{ImmunData} object. The original object is not +changed. + +This function is a direct implementation of \link[dplyr:mutate]{dplyr::mutate}. Alternative +function name is \code{mutate_immundata}. } \details{ -The function operates in two main steps: -\enumerate{ -\item \strong{Standard Mutations (\code{...})}: Applies the standard \code{dplyr::mutate}-style -expressions provided in \code{...} to the \verb{$annotations} table. You can create -new columns or modify existing ones, but you \emph{cannot} modify columns -defined in the core \code{ImmunData} schema (e.g., \code{receptor_id}, \code{cell_id}). -An error will occur if you attempt to do so. -\item \strong{Sequence-based Annotations (\code{seq_options})}: If \code{seq_options} is provided, -the function calculates sequence similarities or distances and adds corresponding -new columns to the \verb{$annotations} table. +You can use \code{mutate()} in three ways: \itemize{ -\item \code{method = "exact"}: Adds boolean columns (TRUE/FALSE) indicating whether the -\code{query_col} value exactly matches each \code{pattern}. Column names are generated -using a prefix (e.g., \code{sim_exact_}) and the pattern or its index. -\item \code{method = "regex"}: Uses \code{annotate_tbl_regex} to add columns indicating -matches for each regular expression pattern against the \code{query_col}. The -exact nature of the added columns depends on \code{annotate_tbl_regex} (e.g., -boolean flags or captured groups). -\item \code{method = "lev"} or \code{method = "hamm"}: Uses \code{annotate_tbl_distance} to -calculate Levenshtein or Hamming distances between the \code{query_col} and -each \code{pattern}, adding columns containing these numeric distances. -\code{max_dist} is ignored in this context (internally treated as \code{NA}) as -all distances are calculated and added, not used for filtering. -\item The naming of the new sequence-based columns depends on the \code{name_type} -option within \code{seq_options} and internal helper functions like -\code{make_pattern_columns}. Prefixes like \code{sim_exact_}, \code{sim_regex_}, -\code{dist_lev_}, \code{dist_hamm_} are typically used based on the schema. +\item Supply named calculations in \code{...} to create annotation columns from +existing data. For example, \code{cmv_specific = cdr3_aa \%in\% cmv_cdr3s} adds a +column containing \code{TRUE} or \code{FALSE} for each row. +\item Supply \code{.by} to perform calculations separately for temporary groups. The +number of rows does not change. A group statistic is repeated for all rows +in that group. +\item Supply \code{seq_options}, created with \code{\link[=make_seq_options]{make_seq_options()}}, to compare a +sequence column with one or more reference sequences or patterns. One result +column is added for each reference. } + +Named calculations in \code{...} are performed before sequence comparisons. + +Most grouped calculations are translated directly to DuckDB. Some group +statistics, such as \code{n_distinct()}, are not available as DuckDB window +calculations when a large dataset must stay on disk. In that case, \code{mutate()} +automatically calculates one summary row per group and joins the values back +to the annotation rows. This remains lazy and does not load the full dataset +into R memory. + +The automatic fallback works when every calculation in the call produces one +value per group. If a call combines a row-level calculation with a group +statistic that needs the fallback, use two \code{mutate()} calls. Also use a second +call when a later calculation refers to a group statistic created by the +fallback. See the examples below. + +Columns used to identify receptors or repertoires, and identifiers managed by +\code{ImmunData}, are protected. This prevents accidental changes that would make +the object inconsistent. You can add new columns and change other annotation +columns. + +Sequence comparison methods are: +\itemize{ +\item \code{"exact"}: \code{TRUE} when the sequence is identical to the reference. +\item \code{"regex"}: \code{TRUE} when the sequence matches a regular-expression pattern. +This is an advanced option for matching text patterns. +\item \code{"lev"}: the number of substitutions, insertions, or deletions needed to +change one sequence into the other. +\item \code{"hamm"}: the number of different positions between sequences of the same +length. Sequences with different lengths receive \code{NA}. } -The \verb{$repertoires} table, if present in the input \code{idata}, is copied to the -output object without modification. This function only affects the \verb{$annotations} -table. +For the distance methods, \code{0} means an exact match and smaller values mean +more similar sequences. With \code{name_type = "index"}, the result columns have +short names such as \code{imd_sim_exact_1} or \code{imd_sim_lev_1}. With +\code{name_type = "pattern"}, each column name includes its reference pattern. + +\code{max_dist} is used by \code{\link[=filter_immundata]{filter_immundata()}} but has no effect here because +\code{mutate()} reports every calculated distance. + +Existing repertoire and strata summaries are carried to the new object +without modification. } \examples{ -# Basic setup (assuming idata_test is a valid ImmunData object) -# print(idata_test) - -\dontrun{ -# Example 1: Add a simple derived column -idata_mut1 <- mutate(idata_test, V_family = substr(V_gene, 1, 5)) -print(idata_mut1$annotations) - -# Example 2: Add multiple columns and modify one (if 'custom_score' exists) -# Note: Avoid modifying core schema columns like 'V_gene' itself. -idata_mut2 <- mutate(idata_test, - V_basic = gsub("-.*", "", V_gene), - J_len = nchar(J_gene), - custom_score = custom_score * 1.1 -) # Fails if custom_score doesn't exist -print(idata_mut2$annotations) - -# Example 3: Add boolean columns for exact CDR3 matches -cdr3_patterns <- c("CARGLGLVFYGMDVW", "CARDNRGAVAGVFGEAFYW") -seq_opts_exact <- make_seq_options( - query_col = "CDR3_aa", - patterns = cdr3_patterns, - method = "exact", - name_type = "pattern" -) # Name cols by pattern -idata_mut_exact <- mutate(idata_test, seq_options = seq_opts_exact) -# Look for new columns like 'sim_exact_CARGLGLVFYGMDVW' -print(idata_mut_exact$annotations) - -# Example 4: Add Levenshtein distance columns for a CDR3 pattern -seq_opts_lev <- make_seq_options( - query_col = "CDR3_aa", - patterns = "CARGLGLVFYGMDVW", - method = "lev", - name_type = "index" -) # Name col like 'dist_lev_1' -idata_mut_lev <- mutate(idata_test, seq_options = seq_opts_lev) -# Look for new column 'dist_lev_1' (or similar based on schema) -print(idata_mut_lev$annotations) - -# Example 5: Combine standard mutation and sequence annotation -seq_opts_regex <- make_seq_options( - query_col = "V_gene", - patterns = c(ighv1 = "^IGHV1-", ighv3 = "^IGHV3-"), - method = "regex", - name_type = "pattern" +library(immundata) +library(dplyr) + +options(immundata.verbose = FALSE) + +# Load data included with immundata +idata <- get_test_idata() + +# Add the length of each CDR3 amino acid sequence +idata_with_length <- idata |> + mutate(cdr3_length = dd$length(cdr3_aa)) + +idata_with_length |> + collect() |> + select(cdr3_aa, cdr3_length) |> + slice_head(n = 3) +# Expected result: +# cdr3_aa cdr3_length +# ASFPVLSPYNEQF 13 +# ASRAGAGTGELF 12 +# ASSPGQGLDTQY 12 + +# Compare CDR3 sequences with one reference sequence +reference_cdr3 <- "ASFPVLSPYNEQF" + +idata_with_matches <- idata |> + mutate( + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = reference_cdr3, + method = "exact" + ) + ) + +idata_with_matches |> + collect() |> + count(imd_sim_exact_1) +# Expected result: +# imd_sim_exact_1 n +# FALSE 1901 +# TRUE 1 + +# Calculate Levenshtein distance from the reference sequence +idata_with_distance <- idata |> + mutate( + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = reference_cdr3, + method = "lev" + ) + ) + +idata_with_distance |> + collect() |> + select(cdr3_aa, imd_sim_lev_1) |> + arrange(imd_sim_lev_1, cdr3_aa) |> + slice_head(n = 3) +# Expected result: +# cdr3_aa imd_sim_lev_1 +# ASFPVLSPYNEQF 0 +# ASSPDSPSYNEQF 4 +# ASSPGLAAYNEQF 4 + +# Mark selected sequences +cmv_cdr3s <- c( + "ASFPVLSPYNEQF", + "ASRAGAGTGELF" ) -idata_mut_combo <- mutate(idata_test, - chain_upper = toupper(chain), - seq_options = seq_opts_regex + +marked_sequences <- idata |> + mutate( + cmv_specific = cdr3_aa \%in\% cmv_cdr3s + ) + +marked_sequences |> + collect() |> + count(cmv_specific) +# Expected result: +# cmv_specific n +# FALSE 1900 +# TRUE 2 + +# Mark selected receptor identities +cmv_hits <- tibble( + imd_receptor_id = c(1L, 105L), + cmv_specific = TRUE ) -# Look for 'chain_upper' and regex match columns (e.g., 'sim_regex_ighv1') -print(idata_mut_combo) -} + +marked_receptors <- idata |> + annotate_receptors(cmv_hits) |> + mutate( + cmv_specific = coalesce(cmv_specific, FALSE) + ) + +marked_receptors |> + collect() |> + count(cmv_specific) +# Expected result: +# cmv_specific n +# FALSE 1898 +# TRUE 4 + +# Add response-level statistics to every annotation row +# `.by` means: calculate separately for each response. +response_stats <- idata |> + mutate( + response_n_rows = n(), + response_n_receptors = n_distinct(imd_receptor_id), + .by = Response + ) + +response_stats |> + collect() |> + distinct(Response, response_n_rows, response_n_receptors) |> + arrange(Response) +# Expected result: +# Response response_n_rows response_n_receptors +# FR 955 871 +# PR 947 867 + +# A grouped calculation can also produce a different value for every row. +response_centered <- idata |> + mutate( + centered_counts = counts - mean(counts, na.rm = TRUE), + .by = Response + ) + +# Do not combine that row-level calculation with a statistic that needs the +# automatic summary fallback in the same call: +# idata |> +# mutate( +# centered_counts = counts - mean(counts, na.rm = TRUE), +# response_n_receptors = n_distinct(imd_receptor_id), +# .by = Response +# ) + +# Use two mutate calls instead. The work remains lazy in DuckDB. +response_details <- idata |> + mutate( + centered_counts = counts - mean(counts, na.rm = TRUE), + .by = Response + ) |> + mutate( + response_n_receptors = n_distinct(imd_receptor_id), + .by = Response + ) + +# Also use a second call when a new calculation uses a statistic created by +# the fallback. +response_details <- idata |> + mutate( + response_n_receptors = n_distinct(imd_receptor_id), + .by = Response + ) |> + mutate( + twice_response_n_receptors = response_n_receptors * 2 + ) } \seealso{ -\code{\link[dplyr:mutate]{dplyr::mutate()}}, \code{\link[=make_seq_options]{make_seq_options()}}, \code{\link[=filter_immundata]{filter_immundata()}}, \link{ImmunData}, -\code{vignette("immundata-classes", package = "immunarch")} (replace with actual package name if different) +\code{\link[dplyr:mutate]{dplyr::mutate()}}, \code{\link[=make_seq_options]{make_seq_options()}}, \code{\link[=filter_immundata]{filter_immundata()}}, +\code{\link[=annotate_receptors]{annotate_receptors()}}, \code{\link[=agg_repertoires]{agg_repertoires()}}, \link{ImmunData} } \concept{mutation} diff --git a/man/names-set-.ImmunData.Rd b/man/names-set-.ImmunData.Rd new file mode 100644 index 0000000..8896b37 --- /dev/null +++ b/man/names-set-.ImmunData.Rd @@ -0,0 +1,17 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/operations_dimnames.R +\name{names<-.ImmunData} +\alias{names<-.ImmunData} +\title{Prevent Renaming ImmunData via names} +\usage{ +\method{names}{ImmunData}(x) <- value +} +\arguments{ +\item{x}{ImmunData object.} + +\item{value}{Not used.} +} +\description{ +Disallows replacing names on \code{ImmunData}. +} +\concept{operations} diff --git a/man/preprocess_postprocess.Rd b/man/preprocess_postprocess.Rd index 0c92a04..a50c7b7 100644 --- a/man/preprocess_postprocess.Rd +++ b/man/preprocess_postprocess.Rd @@ -6,9 +6,9 @@ \alias{make_exclude_columns} \alias{make_productive_filter} \alias{make_barcode_prefix} -\title{Preprocessing and postprocessing of input immune repertoire files} +\title{Process chain rows while reading repertoire files} \usage{ -make_default_preprocessing(format = c("airr", "10x")) +make_default_preprocessing(format = c("default", "airr", "10x")) make_default_postprocessing() @@ -19,83 +19,120 @@ make_productive_filter(col_name = c("productive"), truthy = TRUE) make_barcode_prefix(prefix_col = "Prefix") } \arguments{ -\item{format}{For \code{make_default_preprocessing()}, a character string specifying -the input data format. Currently supports \code{"airr"} (default) or \code{"10x"}. -This determines the default set of columns to exclude and the values -considered "productive".} - -\item{cols}{For \code{make_exclude_columns()}, a character vector of column names -to be removed from the dataset. Defaults to \code{imd_drop_cols("airr")}. -If empty, the returned function will not remove any columns.} - -\item{col_name}{For \code{make_productive_filter()}, a character vector of potential -column names that indicate sequence productivity (e.g., \code{"productive"}). -The first matching column found in the dataset will be used.} - -\item{truthy}{For \code{make_productive_filter()}, a value or vector of values -that signify a productive sequence in the \code{col_name} column. -Can be a logical \code{TRUE} (default for "airr" format) or a character vector -of strings (e.g., \code{c("true", "TRUE", "True", "t", "T", "1")} for "10x" format).} - -\item{prefix_col}{For \code{make_barcode_prefix()}, the name of the column in the -dataset that contains the prefix string to be added to each cell barcode. -Defaults to \code{"Prefix"}. The barcode column itself is identified internally -via \code{imd_schema("barcode")}.} +\item{format}{A character string. One input format: \code{"default"}, \code{"airr"}, +or \code{"10x"}. The default is \code{"default"}. This choice controls which +technical columns are removed. It does not rename columns.} + +\item{cols}{A character vector. Columns to remove. The default is +\code{imd_drop_cols("airr")}. Use \code{character()} to create a step that removes +no columns.} + +\item{col_name}{A character string. Column containing the productive-chain +indicator. The default is \code{"productive"}.} + +\item{truthy}{A vector. Values that mean the chain is productive. Values are +compared as text. The default is \code{TRUE}; use a character vector when the +source uses several representations, for example +\code{c("TRUE", "true", "1")}.} + +\item{prefix_col}{A character vector. One or more candidate columns +containing the text to place before each cell barcode. The first candidate +present in the data is used. The default is \code{"Prefix"}.} } \value{ -Each \verb{make_*} function returns a \emph{new function}. This returned function takes -a \code{dataset} as its first argument and \code{...} for any additional arguments, -and performs the specific processing step. -\code{make_default_preprocessing()} and \code{make_default_postprocessing()} return a -\emph{named list} of such functions. +\code{make_default_preprocessing()} and +\code{make_default_postprocessing()} return named lists of processing functions. +The other functions return one processing function. Each processing +function accepts a duckplyr table as its first argument, accepts unused +arguments through \code{...}, and returns a new duckplyr table. } \description{ -Preprocessing and postprocessing of input immune repertoire files +Use these functions to preprocess or postprocess rows of the input data before +returning the final \code{ImmunData} object to the session. A couple of example +use cases: keep productive receptor chains, remove technical +columns, or make cell barcodes unique while importing repertoire files with +\code{\link[=read_repertoires]{read_repertoires()}}. + +The defaults provide steps for common AIRR or 10x inputs. Use an individual step +when your files need only one operation or when you are building a custom +\code{preprocess} or \code{postprocess} list. + +Preprocessing changes chain rows before receptors are defined. Barcode +prefixing changes the cell identifier after receptor and manifest information +are combined. The input files and input table are not changed: every step +returns a new duckplyr table. } -\details{ -This collection of "maker" functions generates common preprocessing and -postprocessing function steps tailored for immune repertoire data. -Each \verb{make_*} function returns a new function that can then be applied -to a dataset. - -These functions are designed to be flexible components in constructing -custom data processing workflows. - -The functions generated by these factories typically expect a \code{dataset} -(e.g., a \code{duckplyr} with annotations) as their first argument -and may accept additional arguments via \code{...} (though often unused in the -predefined steps). +\section{Choose processing steps}{ + \itemize{ -\item \code{make_default_preprocessing()} and \code{make_default_postprocessing()} assemble -a list of such processing functions. -\item The individual \code{make_exclude_columns()}, \code{make_productive_filter()}, and -\code{make_barcode_prefix()} functions create specific transformation steps. +\item \code{make_default_preprocessing()} returns two steps. The first removes common +technical columns. The second keeps rows whose \code{productive} value indicates +a productive chain. If the \code{productive} column is absent, the filtering +step gives a warning and keeps all rows. +\item \code{make_default_postprocessing()} returns one step that adds a sample-specific +prefix to cell barcodes. If the prefix column is absent, the step gives a +warning and leaves barcodes unchanged. +\item \code{make_exclude_columns()} creates one step that removes the columns in +\code{cols}. Column names that are not present are ignored. +\item \code{make_productive_filter()} creates one step that keeps rows whose value in +\code{col_name} matches any value in \code{truthy}. +\item \code{make_barcode_prefix()} creates one step that joins a prefix, such as +\code{"Tumor_"}, to the start of each \code{imd_barcode} value. } -These steps are often used when reading data to standardize formats, filter -unwanted records, or enrich information like cell barcodes. They are designed -to gracefully handle cases where an operation is not applicable (e.g., a specified -column is not found) by issuing a warning and returning the dataset unmodified. +\code{read_repertoires()} applies functions in list order. You can therefore add, +remove, or reorder steps in a custom list. } -\section{Functions}{ -\itemize{ -\item \code{make_default_preprocessing()}: Creates a default list of preprocessing -functions suitable for "airr" or "10x" formatted data. This typically -includes steps to exclude unnecessary columns and filter for productive sequences. -\item \code{make_default_postprocessing()}: Creates a default list of postprocessing -functions, such as adding a prefix to cell barcodes. -\item \code{make_exclude_columns()}: Creates a function that, when applied to a -dataset, removes a specified set of columns. -\item \code{make_productive_filter()}: Creates a function that filters a dataset -to retain only rows where sequences are marked as productive, based on -a specified column and set of "truthy" values. -\item \code{make_barcode_prefix()}: Creates a function that prepends a prefix -(sourced from a specified column in the dataset) to the cell barcodes. -} +\section{Input formats}{ + + +For \code{make_default_preprocessing()}, \code{format = "default"} removes the union of +the standard AIRR and 10x technical columns. Use \code{format = "airr"} or +\code{format = "10x"} to remove only the columns expected for that format. All +three defaults recognize common text representations of a productive value, +including \code{"TRUE"}, \code{"true"}, \code{"yes"}, and \code{"1"}. } +\examples{ +library(immundata) +library(dplyr) + +# Three 10x chain rows from two samples. One chain is non-productive. +chains <- duckplyr::duckdb_tibble( + imd_barcode = c("AAAC-1", "AAAG-1", "AATT-1"), + cdr3_aa = c("CASSA", "CASSB", "CASSC"), + productive = c("TRUE", "FALSE", "TRUE"), + full_length = c(TRUE, TRUE, TRUE), + Prefix = c("Tumor_", "Tumor_", "Blood_") +) + +# read_repertoires() performs these calls for you. They are shown here to +# make the effect of each list clear. +prepared <- Reduce( + function(data, step) step(data), + make_default_preprocessing("10x"), + init = chains +) +prepared <- Reduce( + function(data, step) step(data), + make_default_postprocessing(), + init = prepared +) + +prepared |> + collect() |> + select(imd_barcode, cdr3_aa, productive) +# Expected result: +# imd_barcode cdr3_aa productive +# Tumor_AAAC-1 CASSA TRUE +# Blood_AATT-1 CASSC TRUE + +# The non-productive chain was removed, `full_length` was dropped, and the +# sample prefixes made the retained cell barcodes unique. + +} \seealso{ -\code{\link[=read_repertoires]{read_repertoires()}} +\code{\link[=read_repertoires]{read_repertoires()}}, \code{\link[=imd_drop_cols]{imd_drop_cols()}}, \code{\link[=imd_rename_cols]{imd_rename_cols()}} } \concept{processing} diff --git a/man/print.ImmunData.Rd b/man/print.ImmunData.Rd new file mode 100644 index 0000000..076935d --- /dev/null +++ b/man/print.ImmunData.Rd @@ -0,0 +1,60 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/operations_print.R +\name{print.ImmunData} +\alias{print.ImmunData} +\title{Display the contents and biological definitions of ImmunData} +\usage{ +\method{print}{ImmunData}(x, ...) +} +\arguments{ +\item{x}{An \link{ImmunData} object to display.} + +\item{...}{Additional arguments. Currently not used.} +} +\value{ +\code{x}, invisibly. The displayed output is a human-readable overview; +no data are modified. +} +\description{ +Use \code{print()} to inspect the receptor table, chain annotations, and biological +schemas stored in an \link{ImmunData} object. + +Use this method for a quick overview after reading, filtering, or aggregating +repertoire data. It displays the units available in the object: receptors, +chain rows, repertoires, and strata. It also shows the feature and chain +definitions used to construct receptors. + +Printing is read-only. It does not collect the complete dataset into R and +does not change the original object. The object is returned invisibly so it +can still be assigned or used in a pipeline. +} +\details{ +A section is shown only when that information is available. An object without +repertoire definitions, for example, has no repertoire schema or repertoire +summary section. Duckplyr prints a preview of large tables rather than every +row. +} +\examples{ +library(immundata) + +options(immundata.verbose = FALSE) +idata <- get_test_idata() + +print(idata) +# Expected output contains these sections: +# ImmunData +# Receptors +# Annotations +# Receptor schema +# Repertoire schema +# List of repertoires + +# `Receptors` previews distinct biological receptor definitions. +# `Annotations` previews the retained chain rows and sample information. +# The schema sections explain how receptors and repertoires were defined. + +} +\seealso{ +\link{ImmunData}, \code{\link[dplyr:collect]{dplyr::collect()}}, \code{\link[dplyr:count]{dplyr::count()}} +} +\concept{operations} diff --git a/man/read_immundata.Rd b/man/read_immundata.Rd index 976492c..87d25af 100644 --- a/man/read_immundata.Rd +++ b/man/read_immundata.Rd @@ -2,83 +2,132 @@ % Please edit documentation in R/io_immundata_read.R \name{read_immundata} \alias{read_immundata} -\title{Load a saved ImmunData from disk} +\title{Load an ImmunData object from disk} \usage{ -read_immundata(path, prudence = "stingy", verbose = TRUE) +read_immundata( + path, + tag = NULL, + version = NULL, + prudence = "stingy", + verbose = getOption("immundata.verbose", TRUE) +) } \arguments{ -\item{path}{Character(1). Path to the \strong{directory} containing the saved -\code{ImmunData} files (\code{annotations.parquet} and \code{metadata.json}).} +\item{path}{A character string. Path to a saved dataset directory. The +directory must contain \code{annotations.parquet} and \code{metadata.json}. When +\code{tag} is supplied, use the project home directory that contains the +\code{snapshots} directory. Read more about snapshots on the website.} -\item{prudence}{Character(1). Controls strictness of type inference when -reading the Parquet file, passed to \code{duckplyr::read_parquet_duckdb()}. -Default \code{"stingy"} likely implies stricter type checking or safer inference.} +\item{tag}{A character string or \code{NULL}. Snapshot tag to read from +\verb{path/snapshots//vNNN}. If \code{NULL}, the default, \code{path} itself is read.} -\item{verbose}{Logical(1). If \code{TRUE} (default), prints informative messages -using \code{cli} during loading. Set to \code{FALSE} for quiet operation.} +\item{version}{A non-negative integer or \code{NULL}. Snapshot version within +\code{tag}. For example, \code{1} reads \code{v001}. If \code{NULL}, the default, the latest +available version for the tag is read. \code{version} can only be used with +\code{tag}.} + +\item{prudence}{A character string. Memory protection used while reading the +Parquet data. This controls whether duckplyr may convert an intermediate +result from DuckDB-managed memory to an R data frame: \code{"stingy"}, the +default here, never permits conversion; \code{"thrifty"} permits up to 1 million +table cells (rows multiplied by columns); and \code{"lavish"} permits conversion +regardless of size. Here, "table cells" does not mean biological cells. +Passed to \code{\link[duckplyr:read_parquet_duckdb]{duckplyr::read_parquet_duckdb()}}.} + +\item{verbose}{A logical value. Whether to print progress and summary +messages. Defaults to \code{getOption("immundata.verbose", TRUE)}.} } \value{ -A new \code{ImmunData} object reconstructed from the saved files. If -repertoire information was saved, it will be recalculated and included. +A new, disk-backed \link{ImmunData} object representing the selected saved +state. Its provenance records the directory that was read. } \description{ -Reconstructs an \code{ImmunData} object from files previously saved to a directory -by \code{\link[=write_immundata]{write_immundata()}} or the internal saving step of \code{\link[=read_repertoires]{read_repertoires()}}. -It reads the \code{annotations.parquet} file for the main data and \code{metadata.json} -to retrieve the necessary receptor and repertoire schemas. +Continue an analysis later by reopening an \link{ImmunData} dataset saved on disk. +Use \code{read_immundata()} after restarting R, in another script, or when another +person gives you a dataset created by \code{\link[=write_immundata]{write_immundata()}} or +\code{\link[=read_repertoires]{read_repertoires()}}. It is that simple, just don't forget to save the +\code{ImmunData} object first! + +The unit restored retains all information: chain rows, +cell and receptor identifiers, repertoire and stratum definitions, and +provenance. The function does not change these biological units or the saved +files. It returns a new \link{ImmunData} object. } \details{ -This function expects a directory structure created by \code{\link[=write_immundata]{write_immundata()}}, -containing at least: -\itemize{ -\item \code{annotations.parquet}: The main annotation data table. -\item \code{metadata.json}: Contains package version, receptor schema, and optionally -repertoire schema. +Read either a dataset directory directly or a versioned snapshot within its +project home. } +\section{Choose the saved state}{ + -The loading process involves: -\enumerate{ -\item Checking that the specified \code{path} is a directory and contains the -required \code{annotations.parquet} and \code{metadata.json} files. -\item Reading \code{metadata.json} using \code{jsonlite::read_json()}. -\item Reading \code{annotations.parquet} using \code{duckplyr::read_parquet_duckdb()} with -the specified \code{prudence} level. -\item Extracting the \code{receptor_schema} and \code{repertoire_schema} from the loaded -metadata. -\item Instantiating a new \code{ImmunData} object using the loaded \code{annotations} data -and the \code{receptor_schema}. -\item If a non-empty \code{repertoire_schema} was found in the metadata, it calls -\code{\link[=agg_repertoires]{agg_repertoires()}} on the newly created object to recalculate and -attach repertoire-level information based on that schema. +To reopen a dataset saved directly in a folder, supply that folder as \code{path} +and leave \code{tag} and \code{version} as \code{NULL}. + +To reopen a managed snapshot, supply the project home as \code{path} and its tag. +By default, the latest version for that tag is read. Supply \code{version} when +you need an exact earlier state. } + +\section{Backend and serialized data}{ + + +\code{annotations.parquet} stores the retained chain-level annotation table. +It is reopened as a lazy duckplyr table, so the complete table does not need +to be loaded into R memory. \code{metadata.json} stores the format and package +versions, receptor, repertoire, and stratum schemas, the repertoire +table, the snapshot identifier, lineage events, and provenance paths. + +Receptor and stratum views are reconstructed from this serialized state; they +are not stored as separate files. Please also mind, that the saved files +is an ImmunData-specific serialization, not an RDS file. } + \examples{ -\dontrun{ -# Assume 'my_idata' is an ImmunData object created previously -# my_idata <- read_repertoires(...) +library(immundata) +library(dplyr) -# Define a temporary directory for saving -save_dir <- tempfile("saved_immundata_") +options(immundata.verbose = FALSE) -# Save the ImmunData object -write_immundata(my_idata, save_dir) +# Create a project home and save a filtered biological state as a snapshot +idata <- get_test_idata() +project_dir <- tempfile("immundata-project-") -# --- Later, in a new session or script --- +project_idata <- write_immundata( + idata, + output_folder = project_dir, + rehome = TRUE +) -# Load the ImmunData object back from the directory -loaded_idata <- read_immundata(save_dir) +fr_response <- project_idata |> + filter(Response == "FR") -# Verify the loaded object -print(loaded_idata) -# compare_methods(my_idata$annotations, loaded_idata$annotations) # If available +write_immundata(fr_response, tag = "fr-response") -# Clean up -unlink(save_dir, recursive = TRUE) -} +# Read the exact first version of this snapshot +continued_fr <- read_immundata( + project_dir, + tag = "fr-response", + version = 1 +) + +continued_fr |> + collect() |> + summarise( + n_chains = n(), + n_receptors = n_distinct(imd_receptor_id) + ) +# Expected result: the snapshot contains the 955 chain rows and 871 +# receptors from the FR response group. +# n_chains n_receptors +# 955 871 + +list.files(file.path(project_dir, "snapshots", "fr-response")) +# Expected result: "v001" + +unlink(project_dir, recursive = TRUE) } \seealso{ -\code{\link[=write_immundata]{write_immundata()}} for saving \code{ImmunData} objects, -\code{\link[=read_repertoires]{read_repertoires()}} for the primary data loading pipeline, \link{ImmunData} class, -\code{\link[=agg_repertoires]{agg_repertoires()}} for repertoire definition. +\code{\link[=write_immundata]{write_immundata()}} for saving an analysis, \code{\link[=read_repertoires]{read_repertoires()}} for +importing AIRR-seq files, \link{ImmunData} } \concept{ingestion} diff --git a/man/read_manifest.Rd b/man/read_manifest.Rd new file mode 100644 index 0000000..05aa24f --- /dev/null +++ b/man/read_manifest.Rd @@ -0,0 +1,53 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/io_manifest_read.R +\name{read_manifest} +\alias{read_manifest} +\title{Load and Validate a Manifest for Immune Repertoire Files} +\usage{ +read_manifest( + manifest, + file_col = "file", + delim = NULL, + ..., + verbose = getOption("immundata.verbose", TRUE) +) +} +\arguments{ +\item{manifest}{A manifest table. Can be either: +\itemize{ +\item a data frame with per-file annotations, +\item or a path to a CSV/TSV/TXT manifest file. +}} + +\item{file_col}{A string specifying the name of the column in the manifest +that contains paths to repertoire files. Defaults to \code{"file"}.} + +\item{delim}{Delimiter used to read the manifest file. If \code{NULL}, it is +inferred from the extension: comma for \code{.csv}, tab for \code{.tsv} and \code{.txt}.} + +\item{...}{Additional arguments passed to \code{readr::read_delim()} when reading +a manifest from a file.} + +\item{verbose}{Logical(1). Whether to print informative messages. Defaults to +\code{getOption("immundata.verbose", TRUE)}.} +} +\value{ +A validated and updated manifest data frame with absolute file paths +and an additional internal column named \code{imd_filename}. +} +\description{ +This function loads a manifest from either a file path or a data frame, +validates the presence of a column with repertoire file paths, and converts +all file paths to absolute paths. It is used to support flexible pipelines +for loading bulk or single-cell immune repertoire data across samples. + +If the input is a file path, the function reads it with \code{readr::read_delim}. +If the input is a data frame, it checks whether file paths are absolute; +relative paths are only allowed when the manifest is loaded from a file. + +It warns the user if many of the files listed in the manifest are missing, +and stops execution if none of the files exist. + +The column with file paths is normalized into the internal filename schema. +} +\concept{ingestion} diff --git a/man/read_metadata.Rd b/man/read_metadata.Rd deleted file mode 100644 index 22b18ca..0000000 --- a/man/read_metadata.Rd +++ /dev/null @@ -1,42 +0,0 @@ -% Generated by roxygen2: do not edit by hand -% Please edit documentation in R/io_metadata_read.R -\name{read_metadata} -\alias{read_metadata} -\title{Load and Validate Metadata Table for Immune Repertoire Files} -\usage{ -read_metadata(metadata, filename_col = "File", delim = "\\t", ...) -} -\arguments{ -\item{metadata}{A metadata table. Can be either: -\itemize{ -\item a data frame with metadata, -\item or a path to a text/TSV/CSV file that can be read with \code{readr::read_delim}. -}} - -\item{filename_col}{A string specifying the name of the column in the metadata table -that contains paths to repertoire files. Defaults to \code{"File"}.} - -\item{delim}{Delimiter used to read the metadata file (if a path is provided). Defaults to \code{"\\t"}.} - -\item{...}{Additional arguments passed to \code{readr::read_delim()} when reading metadata from a file.} -} -\value{ -A validated and updated metadata data frame with absolute file paths, -and an additional column renamed according to \code{IMD_GLOBALS$schema$filename}. -} -\description{ -This function loads a metadata table from either a file path or a data frame, -validates the presence of a column with repertoire file paths, and converts all -file paths to absolute paths. It is used to support flexible pipelines for -loading bulk or single-cell immune repertoire data across samples. - -If the input is a file path, the function attempts to read it with \code{readr::read_delim}. -If the input is a data frame, it checks whether file paths are absolute; -relative paths are only allowed when metadata is loaded from a file. - -It warns the user if many of the files listed in the metadata table are missing, -and stops execution if none of the files exist. - -The column with file paths is normalized and renamed to match the internal filename schema. -} -\concept{ingestion} diff --git a/man/read_repertoires.Rd b/man/read_repertoires.Rd index 8c63f9f..6c32bb5 100644 --- a/man/read_repertoires.Rd +++ b/man/read_repertoires.Rd @@ -2,12 +2,12 @@ % Please edit documentation in R/io_repertoires_read.R \name{read_repertoires} \alias{read_repertoires} -\title{Read and process immune repertoire files to immundata} +\title{Read immune repertoire files into ImmunData} \usage{ read_repertoires( path, schema, - metadata = NULL, + manifest = NULL, barcode_col = NULL, count_col = NULL, locus_col = NULL, @@ -16,199 +16,312 @@ read_repertoires( postprocess = make_default_postprocessing(), rename_columns = imd_rename_cols("10x"), enforce_schema = TRUE, - metadata_file_col = "File", + manifest_file_col = "file", output_folder = NULL, - repertoire_schema = NULL + repertoire_schema = "", + verbose = getOption("immundata.verbose", TRUE), + prematerialize = TRUE, + prematerialize_folder = NULL ) } \arguments{ -\item{path}{Character vector. Path(s) to input repertoire files (e.g., -\code{"/path/to/data/*.tsv.gz"}). Supports glob patterns via \code{\link[=Sys.glob]{Sys.glob()}}. -Files can be Parquet, CSV, TSV, or gzipped versions thereof. All files -must be of the same type. -Alternatively, pass the special string \code{""} to read file paths -from the \code{metadata} table (see \code{metadata} and \code{metadata_file_col} params).} - -\item{schema}{Defines how unique receptors are identified. Can be: +\item{path}{One or more repertoire file paths, or a glob pattern such as +\code{"/path/to/data/*.tsv.gz"}. Supported formats are Parquet, CSV, TSV, and +gzipped CSV or TSV. All input files must have the same file type. + +Use \code{""} to take file paths from \code{manifest} instead. In that +case, \code{manifest} is required.} + +\item{schema}{Definition of receptor identity. Supply either: \itemize{ -\item A character vector of column names (e.g., \code{c("v_call", "j_call", "junction_aa")}). -\item A schema object created by \code{\link[=make_receptor_schema]{make_receptor_schema()}}, allowing specification -of chains for pairing (e.g., \code{make_receptor_schema(features = c("v_call", "junction_aa"), chains = c("TRA", "TRB"))}). +\item A character vector naming the features that must match, such as +\code{c("v_call", "j_call", "junction_aa")}. +\item An object created by \code{\link[=make_receptor_schema]{make_receptor_schema()}} to select one locus or pair +two loci from the same cell. +} + +Use column names as they appear \emph{after} \code{rename_columns} is applied. For +example, if the input columns are \code{CDR3.aa} and \code{V.name}, use +\code{rename_columns = c(cdr3_aa = "CDR3.aa", v_call = "V.name")} together with +\code{schema = c("cdr3_aa", "v_call")}.} + +\item{manifest}{An optional data frame with one row per repertoire file and +columns containing sample, donor, tissue, treatment, or other information. +Use \code{\link[=read_manifest]{read_manifest()}} to read and validate a manifest file. Manifest paths +must be unique. When \code{path = ""}, the column named by +\code{manifest_file_col} supplies the repertoire file paths. The default is +\code{NULL}.} + +\item{barcode_col}{Name of the column containing cell barcodes. Supplying it +selects single-cell processing, requires \code{umi_col}, and prevents use of +\code{count_col}. Use the column name after renaming. The default is \code{NULL}.} + +\item{count_col}{Name of the column containing non-negative abundance values +for bulk repertoire data. It cannot be used with \code{barcode_col}. Use the +column name after renaming. The default is \code{NULL}.} + +\item{locus_col}{Name of the column containing receptor loci such as \code{"TRA"}, +\code{"TRB"}, \code{"IGH"}, \code{"IGK"}, or \code{"IGL"}. It is required when \code{schema} +selects or pairs chains. Use the column name after renaming. The default is +\code{NULL}.} + +\item{umi_col}{Name of the column containing per-chain UMI or read counts. +It is required whenever \code{barcode_col} is supplied and is used to choose one +chain when a cell contains several chains from the same locus. Use the +column name after renaming. The default is \code{NULL}.} + +\item{preprocess}{A named list of functions applied in order before receptors +are defined. Each function must accept a duckplyr table as its first +argument and return a duckplyr table. By default, +\code{\link[=make_default_preprocessing]{make_default_preprocessing()}} removes selected technical columns and keeps +productive sequences when a \code{productive} column is available. Use \code{NULL} +or \code{list()} to disable preprocessing.} + +\item{postprocess}{A named list of functions applied in order after receptors +are defined and manifest information is added. Each function must accept +and return a duckplyr table. By default, \code{\link[=make_default_postprocessing]{make_default_postprocessing()}} +prefixes cell barcodes when the manifest contains a \code{Prefix} column. Use +\code{NULL} or \code{list()} to disable postprocessing.} + +\item{rename_columns}{An optional named character vector in the form +\code{c(new_name = "old_name")}. Renaming occurs before preprocessing and +receptor definition. The default, \code{imd_rename_cols("10x")}, standardizes +common 10x names such as \code{v_gene} to \code{v_call} and \code{chain} to \code{locus} when +those source columns are present. Use \code{NULL} to preserve all input names.} + +\item{enforce_schema}{Whether multiple input files must have the same columns +and column types. The default is \code{TRUE}. If \code{FALSE}, columns are combined +by name and missing values are added where necessary. This is slower and +can require more memory.} + +\item{manifest_file_col}{Name of the manifest column containing repertoire +file paths when \code{path = ""}. The default is \code{"file"}. Use the +same name passed as \code{file_col} to \code{\link[=read_manifest]{read_manifest()}} when it is not \code{"file"}.} + +\item{output_folder}{Directory in which to write \code{annotations.parquet} and +\code{metadata.json}. These files are the persistent backing storage for the +returned object. If \code{NULL}, a folder beginning with \verb{immundata-} is created +beside the first input file. Supplying an existing folder replaces its +\code{annotations.parquet} and \code{metadata.json}. The default is \code{NULL}.} + +\item{repertoire_schema}{Definition of repertoires. Supply one of: +\itemize{ +\item A character vector naming columns that define one repertoire, such as +\code{c("donor", "timepoint")}. +\item \code{""}, the default. This creates one repertoire per input file, or +one per manifest row when \code{path = ""}. +\item \code{""}, which uses all manifest columns when a manifest is +available, or the input filename otherwise. +\item \code{NULL} to leave repertoires undefined. }} -\item{metadata}{Optional. A data frame containing -metadata to be joined with the repertoire data, read by -\code{\link[=read_metadata]{read_metadata()}} function. If \code{path = ""}, this table \emph{must} -be provided and contain the file paths column specified by \code{metadata_file_col}. -Default: \code{NULL}.} - -\item{barcode_col}{Character(1). Name of the column containing cell barcodes -or other unique cell/clone identifiers for single-cell data. Triggers -single-cell processing logic in \code{\link[=agg_receptors]{agg_receptors()}}. Default: \code{NULL}.} - -\item{count_col}{Character(1). Name of the column containing UMI counts or -frequency counts for bulk sequencing data. Triggers bulk processing logic -in \code{\link[=agg_receptors]{agg_receptors()}}. Default: \code{NULL}. Cannot be specified if \code{barcode_col} is also -specified.} - -\item{locus_col}{Character(1). Name of the column specifying the receptor chain -locus (e.g., "TRA", "TRB", "IGH", "IGK", "IGL"). Required if \code{schema} -specifies chains for pairing. Default: \code{NULL}.} - -\item{umi_col}{Character(1). Name of the column containing UMI counts for -single-cell data. Used during paired-chain processing to select the most -abundant chain per barcode per locus. Default: \code{NULL}.} - -\item{preprocess}{List. A named list of functions to apply sequentially to the -raw data \emph{before} receptor aggregation. Each function should accept a -data frame (or duckplyr_df) as its first argument. See -\code{\link[=make_default_preprocessing]{make_default_preprocessing()}} for examples. -Default: \code{make_default_preprocessing()}. Set to \code{NULL} or \code{list()} to disable.} - -\item{postprocess}{List. A named list of functions to apply sequentially to the -annotation data \emph{after} receptor aggregation and metadata joining. Each -function should accept a data frame (or duckplyr_df) as its first argument. -See \code{\link[=make_default_postprocessing]{make_default_postprocessing()}} for examples. -Default: \code{make_default_postprocessing()}. Set to \code{NULL} or \code{list()} to disable.} - -\item{rename_columns}{Named character vector. Optional mapping to rename columns -in the input files using \code{dplyr::rename()} syntax (e.g., -\code{c(new_name = "old_name", barcode = "cell_id")}). Renaming happens \emph{before} -preprocessing and schema application. See \code{\link[=imd_rename_cols]{imd_rename_cols()}} for presets. -Default: \code{imd_rename_cols("10x")}.} - -\item{enforce_schema}{Logical(1). If \code{TRUE} (default), reading multiple files -requires them to have the exact same columns and types. If \code{FALSE}, columns -are unioned across files (potentially slower, requires more memory). -Default: \code{TRUE}.} - -\item{metadata_file_col}{Character(1). The name of the column in the \code{metadata} -table that contains the full paths to the repertoire files. Only used when -\code{path = ""}. Default: \code{"File"}.} - -\item{output_folder}{Character(1). Path to a directory where intermediate -processed annotation data will be saved as \code{annotations.parquet} and -\code{metadata.json}. If \code{NULL} (default), a folder named -\verb{immundata-} is created in the same directory as the -first input file specified in \code{path}. The final \code{ImmunData} object reads -from these saved files. Default: \code{NULL}.} - -\item{repertoire_schema}{Character vector or Function. Defines columns used to -group annotations into distinct repertoires (e.g., by sample or donor). -If provided, \code{\link[=agg_repertoires]{agg_repertoires()}} is called after loading to add repertoire-level -summaries and metrics. Default: \code{NULL}.} +\item{verbose}{Whether to print progress and summary messages. Defaults to +\code{getOption("immundata.verbose", TRUE)}.} + +\item{prematerialize}{Whether CSV, TSV, and compressed text inputs should be +combined into a temporary Parquet file before receptor processing. This +avoids repeatedly scanning text input during downstream lazy queries. +Existing Parquet input is used directly. The default is \code{TRUE}.} + +\item{prematerialize_folder}{Directory in which to create the temporary +combined Parquet file. If \code{NULL}, the default, \code{\link[=tempdir]{tempdir()}} is used. The +directory is created when necessary. The temporary file is deleted when +\code{read_repertoires()} exits, including after an error.} } \value{ -An \code{ImmunData} object containing the processed receptor annotations. -If \code{repertoire_schema} was provided, the object will also contain repertoire -definitions and summaries calculated by \code{\link[=agg_repertoires]{agg_repertoires()}}. +A disk-backed \link{ImmunData} object containing the retained chain rows, +receptor definitions, manifest annotations, and ingestion provenance. If +\code{repertoire_schema} is not \code{NULL}, it also contains repertoire definitions +and summary statistics calculated by \code{\link[=agg_repertoires]{agg_repertoires()}}. } \description{ -This is the main function for reading immune repertoire data into the -\code{immundata} framework. It reads one or more repertoire files (AIRR TSV, -10X CSV, Parquet), performs optional preprocessing and column renaming, -aggregates sequences into receptors based on a provided schema, optionally -joins external metadata, performs optional postprocessing, and returns -an \code{ImmunData} object. - -The function handles different data types (bulk, single-cell) based on -the presence of \code{barcode_col} and \code{count_col}. For efficiency with large -datasets, it processes the data and saves intermediate results (annotations) -as a Parquet file before loading them back into the final \code{ImmunData} object. +\code{read_repertoires()} is the main function for importing AIRR-seq data. It +reads one or more repertoire files, defines biological receptors, adds +sample information from an optional manifest, and returns an \link{ImmunData} +object. + +The function saves the processed data in \code{output_folder}. This lets you work +with large datasets without loading everything into memory and reopen the +result later with \code{\link[=read_immundata]{read_immundata()}}. } \details{ -The function executes the following steps: +The required arguments depend on how receptor observations are represented in +the input files. +} +\section{Choose arguments for your data}{ + +\itemize{ +\item \strong{Uncounted repertoire table:} Supply \code{schema}. Leave \code{barcode_col} and +\code{count_col} as \code{NULL}. Each retained row represents one observed chain. +\item \strong{Bulk repertoire with abundance:} Supply \code{schema} and \code{count_col}. The +abundance values are preserved for later repertoire statistics. +\item \strong{Single-cell, one selected chain:} Use \code{\link[=make_receptor_schema]{make_receptor_schema()}} with one +chain and supply \code{barcode_col}, \code{locus_col}, and \code{umi_col}. +\item \strong{Single-cell, paired chains:} Use \code{\link[=make_receptor_schema]{make_receptor_schema()}} with two chains +and supply \code{barcode_col}, \code{locus_col}, and \code{umi_col}. Only cells containing +both requested chains are retained. +\item \strong{Single-cell, relaxed paired chains:} Use a schema such as +\code{chains = c("IGH", "IGL|IGK")} with \code{barcode_col}, \code{locus_col}, and +\code{umi_col}. This accepts either an IGH-IGL or IGH-IGK receptor. +} + +In single-cell data, the chain with the highest \code{umi_col} value is retained +when a cell contains several chains from the same locus. +} + +\section{What happens by default}{ + + +Unless you override the relevant arguments, \code{read_repertoires()}: +\itemize{ +\item temporarily combines text input into Parquet before processing; +\item standardizes common 10x column names; +\item removes selected technical columns; +\item keeps productive sequences when productivity information is present; +\item prefixes barcodes when a manifest \code{Prefix} column is present; +\item creates repertoires automatically; and +\item writes the completed dataset to disk. +} + +Set \code{rename_columns}, \code{preprocess}, \code{postprocess}, or \code{repertoire_schema} to +\code{NULL} to disable the corresponding behavior. +} + +\section{Processing order}{ + + +The function: \enumerate{ -\item Validates inputs. -\item Determines the list of input files based on \code{path} and \code{metadata}. Checks file extensions. -\item Reads data using \code{duckplyr} (\code{read_parquet_duckdb} or \code{read_csv_duckdb}). Handles \code{.gz}. -\item Applies column renaming if \code{rename_columns} is provided. -\item Applies preprocessing steps sequentially if \code{preprocess} is provided. -\item Aggregates sequences into receptors using \code{\link[=agg_receptors]{agg_receptors()}}, based on \code{schema}, \code{barcode_col}, \code{count_col}, \code{locus_col}, and \code{umi_col}. This creates the core annotation table. -\item Joins the \code{metadata} table if provided. -\item Applies postprocessing steps sequentially if \code{postprocess} is provided. -\item Creates a temporary \code{ImmunData} object in memory. -\item Determines the \code{output_folder} path. -\item Saves the processed annotation table and metadata using \code{\link[=write_immundata]{write_immundata()}} to the \code{output_folder}. -\item Loads the data back from the saved Parquet files using \code{\link[=read_immundata]{read_immundata()}} to create the final \code{ImmunData} object. This ensures the returned object is backed by efficient storage. -\item If \code{repertoire_schema} is provided, calls \code{\link[=agg_repertoires]{agg_repertoires()}} on the loaded object to define and summarize repertoires. -\item Returns the final \code{ImmunData} object. +\item finds and reads the input files as one duckplyr table; +\item temporarily combines non-Parquet input into one Parquet file when +\code{prematerialize = TRUE}; +\item renames columns; +\item applies preprocessing; +\item defines receptors using \code{schema}; +\item adds manifest information; +\item applies postprocessing; +\item defines repertoires when requested; and +\item writes and reopens the completed \link{ImmunData} dataset. } } + +\section{Manifests and repertoires}{ + + +A manifest \emph{annotates} each input file with biological information. The +\code{repertoire_schema} argument chooses which annotation columns \emph{define a +repertoire} and therefore determine receptor counts and proportions. + +With \code{path = ""} and the default \code{repertoire_schema = ""}, +all manifest columns are used and each manifest row becomes one repertoire. +With an explicit file path or vector of paths, \code{""} creates one +repertoire per input file. +} + +\section{Output storage}{ + + +The output folder is not a temporary cache. The returned object reads its +receptor annotations from \code{annotations.parquet}, while \code{metadata.json} stores +its schemas, repertoire summaries, and provenance. Keep this folder for as +long as you need the object, or reopen it later with \code{\link[=read_immundata]{read_immundata()}}. + +\strong{Important:} Reusing the same \code{output_folder} replaces the existing +\code{annotations.parquet} and \code{metadata.json} without creating a new version. +} + \examples{ -\dontrun{ -# -# Example 1: single-chain, one file -# -# Read a single AIRR TSV file, defining receptors by V/J/CDR3_aa -# Assume "my_sample.tsv" exists and follows AIRR format - -# Create a dummy file for illustration -airr_data <- data.frame( - sequence_id = paste0("seq", 1:5), - v_call = c("TRBV1", "TRBV1", "TRBV2", "TRBV1", "TRBV3"), - j_call = c("TRBJ1", "TRBJ1", "TRBJ2", "TRBJ1", "TRBJ1"), - junction_aa = c("CASSL...", "CASSL...", "CASSD...", "CASSL...", "CASSF..."), - productive = c(TRUE, TRUE, TRUE, FALSE, TRUE), - locus = c("TRB", "TRB", "TRB", "TRB", "TRB") +library(immundata) +library(dplyr) + +options(immundata.verbose = FALSE) + +# Read one bulk AIRR file and preserve its abundance column +bulk_file <- system.file( + "extdata/tsv", + "sample_0_1k.tsv", + package = "immundata" +) + +bulk_idata <- read_repertoires( + path = bulk_file, + schema = c("cdr3_aa", "v_call"), + count_col = "counts", + output_folder = tempfile("immundata-bulk-") ) -readr::write_tsv(airr_data, "my_sample.tsv") -# Define receptor schema -receptor_def <- c("v_call", "j_call", "junction_aa") +tibble( + n_records = bulk_idata |> count() |> pull(n), + n_receptors = bulk_idata$receptors |> count() |> collect() |> pull(n), + n_repertoires = nrow(bulk_idata$repertoires) +) +# Expected result: +# n_records n_receptors n_repertoires +# 955 871 1 -# Specify output folder -out_dir <- tempfile("immundata_output_") +# Read multiple files and their sample information from a manifest +manifest_path <- system.file( + "extdata/tsv", + "manifest.csv", + package = "immundata" +) +manifest <- read_manifest(manifest_path) -# Read the data (disabling default preprocessing for this simple example) -idata <- read_repertoires( - path = "my_sample.tsv", - schema = receptor_def, - output_folder = out_dir, - preprocess = NULL, # Disable default productive filter for demo - postprocess = NULL # Disable default barcode prefixing +manifest_idata <- read_repertoires( + path = "", + manifest = manifest, + schema = c("cdr3_aa", "v_call"), + count_col = "counts", + output_folder = tempfile("immundata-manifest-") ) -print(idata) -print(idata$annotations) - -# -# Example 2: single-chain, multiple files -# -# Read multiple files using metadata -# Create dummy files and metadata -readr::write_tsv(airr_data[1:2, ], "sample1.tsv") -readr::write_tsv(airr_data[3:5, ], "sample2.tsv") -meta <- data.frame( - SampleID = c("S1", "S2"), - Tissue = c("PBMC", "Tumor"), - FilePath = c(normalizePath("sample1.tsv"), normalizePath("sample2.tsv")) +manifest_idata$repertoires |> + select(Therapy, Response, n_barcodes, n_receptors) |> + arrange(Response) +# Expected result: +# Therapy Response n_barcodes n_receptors +# ICI FR 4725 871 +# CAR-T PR 4758 867 + +# Read paired TRA-TRB receptors from a small single-cell table +paired_input <- tibble( + cell_id = c("cell1", "cell1", "cell2", "cell2", "cell3"), + locus = c("TRA", "TRB", "TRA", "TRB", "TRA"), + v_call = c("TRAV1", "TRBV1", "TRAV1", "TRBV1", "TRAV2"), + j_call = c("TRAJ1", "TRBJ1", "TRAJ1", "TRBJ1", "TRAJ2"), + junction_aa = c("CAVA", "CASSB", "CAVA", "CASSB", "CAVC"), + umi_count = c(10L, 8L, 12L, 9L, 7L) ) -readr::write_tsv(meta, "metadata.tsv") - -idata_multi <- read_repertoires( - path = "", - metadata = meta, - metadata_file_col = "FilePath", - schema = receptor_def, - repertoire_schema = "SampleID", # Aggregate by SampleID - output_folder = tempfile("immundata_multi_"), - preprocess = make_default_preprocessing("airr"), # Use default AIRR filters - postprocess = NULL +paired_file <- tempfile(fileext = ".tsv") +readr::write_tsv(paired_input, paired_file) + +paired_idata <- read_repertoires( + path = paired_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("TRA", "TRB") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + repertoire_schema = NULL, + output_folder = tempfile("immundata-paired-") ) -print(idata_multi) -print(idata_multi$repertoires) # Check repertoire summary +tibble( + n_chains = paired_idata |> count() |> pull(n), + n_cells = paired_idata |> collect() |> distinct(imd_barcode) |> nrow(), + n_receptors = paired_idata$receptors |> count() |> collect() |> pull(n) +) +# Expected result: +# n_chains n_cells n_receptors +# 4 2 1 -# Clean up dummy files -file.remove("my_sample.tsv", "sample1.tsv", "sample2.tsv", "metadata.tsv") -unlink(out_dir, recursive = TRUE) -unlink(attr(idata_multi, "output_folder"), recursive = TRUE) # Get path used by function -} } \seealso{ -\link{ImmunData}, \code{\link[=read_immundata]{read_immundata()}}, \code{\link[=write_immundata]{write_immundata()}}, \code{\link[=read_metadata]{read_metadata()}}, -\code{\link[=agg_receptors]{agg_receptors()}}, \code{\link[=agg_repertoires]{agg_repertoires()}}, \code{\link[=make_receptor_schema]{make_receptor_schema()}}, -\code{\link[=make_default_preprocessing]{make_default_preprocessing()}}, \code{\link[=make_default_postprocessing]{make_default_postprocessing()}} +\code{\link[=read_manifest]{read_manifest()}}, \code{\link[=make_receptor_schema]{make_receptor_schema()}}, \code{\link[=agg_receptors]{agg_receptors()}}, +\code{\link[=agg_repertoires]{agg_repertoires()}}, \code{\link[=make_default_preprocessing]{make_default_preprocessing()}}, +\code{\link[=make_default_postprocessing]{make_default_postprocessing()}}, \code{\link[=read_immundata]{read_immundata()}}, \code{\link[=write_immundata]{write_immundata()}}, +\link{ImmunData} } \concept{ingestion} diff --git a/man/rename_strata.Rd b/man/rename_strata.Rd new file mode 100644 index 0000000..e0741f6 --- /dev/null +++ b/man/rename_strata.Rd @@ -0,0 +1,100 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/operations_agg_strata.R +\name{rename_strata} +\alias{rename_strata} +\title{Give biological strata readable labels} +\usage{ +rename_strata( + idata, + names, + unnamed = c("error", "auto", "keep"), + auto_prefix = "Strata" +) +} +\arguments{ +\item{idata}{An \link{ImmunData} object with strata already created by +\code{\link[=agg_strata]{agg_strata()}}.} + +\item{names}{A named character vector or a data frame. New labels matched to +\code{imd_strata_id}. Supply either: +\itemize{ +\item a named character vector, such as +\code{c("1" = "Control", "2" = "Treated")}; or +\item a data frame with columns \code{imd_strata_id} and \code{strata_name}. +} + +Every new label must be non-empty and unique.} + +\item{unnamed}{A character string. What to do when \code{names} does not include +every stratum. The default, \code{"error"}, asks for a complete mapping. Use +\code{"auto"} to generate labels for missing strata or \code{"keep"} to preserve +their current labels.} + +\item{auto_prefix}{A non-empty character string. Prefix used to generate +labels when \code{unnamed = "auto"}. The default is \code{"Strata"}.} +} +\value{ +A new \link{ImmunData} object with the requested labels in its \verb{$strata} +and \verb{$repertoires} tables. All biological group assignments and repertoire +summaries are preserved. +} +\description{ +Use \code{rename_strata()} to replace automatic stratum labels with names that are +clear in figures and result tables, such as \code{"Control"}, \code{"Treated"}, or +\code{"Tumour tissue"}. + +Use this function after \code{\link[=agg_strata]{agg_strata()}} when labels such as \code{"Strata1"} do not +describe the biological groups. The unit being changed is the stratum label. +Stratum membership and the repertoires, receptors, cells, and chains remain +unchanged. + +The function returns a new \link{ImmunData} object. The original object is not +changed. +} +\details{ +The names of a named character vector are the stratum IDs, not the current +labels. Inspect \code{idata$strata} to find the ID for each biological group. + +The mapping cannot contain unknown or repeated IDs, and the resulting labels +must be unique across strata. +} +\section{Storage details}{ + + +The readable \code{strata_name} is stored in the repertoire and strata tables. The +underlying chain annotations keep only \code{imd_strata_id}, so renaming a stratum +does not rewrite or regroup chain-level data. +} + +\examples{ +library(immundata) +library(dplyr) + +options(immundata.verbose = FALSE) + +# Create treatment strata for the sample repertoires in the test data +treatment_groups <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) |> + agg_strata(schema = "Therapy") + +# Replace automatic labels with names suitable for a figure +labeled_groups <- treatment_groups |> + rename_strata( + names = c("1" = "CAR-T arm", "2" = "ICI arm") + ) + +labeled_groups$strata |> + select(Therapy, imd_strata_id, strata_name) |> + arrange(imd_strata_id) +# Expected result: +# Therapy imd_strata_id strata_name +# CAR-T 1 CAR-T arm +# ICI 2 ICI arm + +# Only the labels changed. Each sample repertoire remains in the same +# treatment stratum. +} +\seealso{ +\code{\link[=agg_strata]{agg_strata()}}, \code{\link[=agg_repertoires]{agg_repertoires()}} +} +\concept{aggregation} diff --git a/man/write_immundata.Rd b/man/write_immundata.Rd index dbf7b08..c0939af 100644 --- a/man/write_immundata.Rd +++ b/man/write_immundata.Rd @@ -2,74 +2,130 @@ % Please edit documentation in R/io_immundata_write.R \name{write_immundata} \alias{write_immundata} -\title{Save ImmunData to disk} +\title{Save an ImmunData object to disk} \usage{ -write_immundata(idata, output_folder) +write_immundata( + idata, + output_folder = NULL, + tag = NULL, + rehome = FALSE, + compression = "zstd", + compression_level = 9, + verbose = getOption("immundata.verbose", TRUE) +) } \arguments{ -\item{idata}{The \code{ImmunData} object to save. Must be an R6 object of class -\code{ImmunData} containing at least the \verb{$annotations} table and schema information -(\verb{$schema_receptor}, optionally \verb{$schema_repertoire}).} +\item{idata}{An \link{ImmunData} object you want to save.} -\item{output_folder}{Character(1). Path to the directory where the output files -will be written. If the directory does not exist, it will be created -recursively.} +\item{output_folder}{A character string or \code{NULL}. Directory in which to +write \code{annotations.parquet} and \code{metadata.json}. If \code{NULL}, the default, a +managed snapshot is created at \verb{home_path/snapshots//vNNN}. The home +path comes from the object's provenance.} + +\item{tag}{A character string or \code{NULL}. Snapshot tag. With +\code{output_folder = NULL}, it names the managed snapshot series; if \code{tag} is +also \code{NULL}, \code{"default"} is used. With an explicit \code{output_folder}, a +supplied tag is recorded in the lineage but does not change the output +path.} + +\item{rehome}{A logical value. Whether an explicit \code{output_folder} becomes +the home for future managed snapshots. The default is \code{FALSE}, which +preserves an existing home. If the object has no home yet, its first +explicit output folder becomes the home with either value. \code{TRUE} requires +an explicit \code{output_folder}.} + +\item{compression}{A character string or \code{NULL}. Parquet compression codec +passed to DuckDB. The default is \code{"zstd"}. Use \code{NULL} to let DuckDB choose.} + +\item{compression_level}{A number or \code{NULL}. Compression level for codecs +that support it. The default is \code{9}. Use \code{NULL} to let DuckDB choose.} + +\item{verbose}{A logical value. Whether to print progress and summary +messages. Defaults to \code{getOption("immundata.verbose", TRUE)}.} } \value{ -Invisibly returns the input \code{idata} object, saved to disk. -In other words, this allows you to create snapshots of the data in the -\code{output_folder}. Mind that by saving the object, you execute all the -stored computations, so this operations can take longer than expected. -Read more about snapshots on our website in the \href{https://immunomind.github.io/docs/concepts/basics/immutability/}{"Concept" section}. +Invisibly returns a newly reopened, disk-backed \link{ImmunData} object +with provenance for the new save. The input \code{idata} remains unchanged. } \description{ -Serializes the essential components of an \code{ImmunData} object to disk for -efficient storage and later retrieval. It saves the core annotation data -(\code{idata$annotations}) as a compressed Parquet file and accompanying metadata -(including receptor/repertoire schemas and package version) as a JSON file -within a specified directory. +Save \code{ImmunData} to disk so you can close R and continue the work later (I cannot +believe it, but it works, I tried it). Use +\code{write_immundata()} after importing or transforming repertoire data, or when +you want a named snapshot before the next analysis step. + +The unit saved is the complete \link{ImmunData} object. This includes retained +chain rows, cell and receptor identifiers, repertoire and stratum definitions, +and provenance. Saving does not add, remove, or change any biological unit. } \details{ -The function performs the following actions: -\enumerate{ -\item Validates the input \code{idata} object and \code{output_folder} path. -\item Creates the \code{output_folder} if it doesn't exist. -\item Constructs a list containing metadata: \code{immundata} package version, -receptor schema (\code{idata$schema_receptor}), and repertoire schema -(\code{idata$schema_repertoire}). -\item Writes the metadata list to \code{metadata.json} within \code{output_folder}. -\item Writes the \code{idata$annotations} table (a \code{duckplyr_df} or similar) to -\code{annotations.parquet} within \code{output_folder}. Uses Zstandard compression -(\code{compression = "zstd"}, \code{compression_level = 9}) for a good balance -between file size and read/write speed. -\item Uses internal helper \code{imd_files()} to determine the standard filenames -(\code{metadata.json}, \code{annotations.parquet}). +Save to an explicit folder for a direct saved state, or use the object's home +to create a versioned managed snapshot. } +\section{Choose how to save}{ + + +Supply \code{output_folder} to save a standalone state in a specific directory. +This is useful when sharing a dataset or choosing its first project home. +If the directory already contains an ImmunData dataset, its +\code{annotations.parquet} and \code{metadata.json} are replaced. -The receptor data itself (if stored separately in future versions) is not -saved by this function; only the annotations linking to receptors are saved, -along with the schema needed to reconstruct/interpret them. +Leave \code{output_folder = NULL} to create a managed snapshot. The function uses +the object's home path and writes the next version under +\verb{snapshots//vNNN}, for example \code{snapshots/baseline/v001}. Later writes +with the same tag create \code{v002}, \code{v003}, and so on; earlier versions remain +available. Use \code{\link[=read_immundata]{read_immundata()}} with \code{tag} and \code{version} to reopen one. + +Every save receives a new snapshot identifier and appends a provenance event. +The returned object records the new saved directory as its current path. } + +\section{Backend and serialization}{ + + +The retained chain-level annotation table is materialized as compressed +\code{annotations.parquet}. Materialization executes any pending lazy duckplyr +calculations. \code{metadata.json} serializes format and package versions, +receptor, repertoire, and stratum schemas, the small repertoire table, the +snapshot identifier, lineage events, and provenance paths. + +Receptor and stratum views are not written as separate files; they can be +reconstructed from the annotation table and metadata. This Parquet and JSON +pair is an ImmunData-specific serialization, not an RDS file. +} + \examples{ -\dontrun{ -# Assume 'my_idata' is an ImmunData object created previously -# my_idata <- read_repertoires(...) +library(immundata) +library(dplyr) -# Define an output directory -save_dir <- tempfile("saved_immundata_") +options(immundata.verbose = FALSE) -# Save the ImmunData object -write_immundata(my_idata, save_dir) +# Save a small immune-repertoire analysis +idata <- get_test_idata() +save_dir <- tempfile("saved-immundata-") -# Check the created files -list.files(save_dir) # Should show "annotations.parquet" and "metadata.json" +saved_idata <- write_immundata(idata, save_dir) + +list.files(save_dir) +# Expected result: the analysis is serialized as two files. +# [1] "annotations.parquet" "metadata.json" + +# Continue the analysis from the saved files +continued_idata <- read_immundata(save_dir) + +continued_idata |> + collect() |> + summarise( + n_chains = n(), + n_receptors = n_distinct(imd_receptor_id) + ) +# Expected result: all 1,902 chain rows and 1,668 receptors are restored. +# n_chains n_receptors +# 1902 1668 -# Clean up unlink(save_dir, recursive = TRUE) } -} \seealso{ -\code{\link[=read_immundata]{read_immundata()}} for loading the saved data, \code{\link[=read_repertoires]{read_repertoires()}} -which uses this function internally, \link{ImmunData} class definition. +\code{\link[=read_immundata]{read_immundata()}} for continuing a saved analysis, +\code{\link[=read_repertoires]{read_repertoires()}} for importing AIRR-seq files, \link{ImmunData} } \concept{ingestion} diff --git a/tests/testthat.R b/tests/testthat.R index c2210d3..f98664c 100644 --- a/tests/testthat.R +++ b/tests/testthat.R @@ -7,6 +7,18 @@ # * https://testthat.r-lib.org/articles/special-files.html library(testthat) + +Sys.setenv( + DUCKPLYR_FALLBACK_INFO = "FALSE", + DUCKPLYR_FALLBACK_COLLECT = "0", + DUCKPLYR_FALLBACK_AUTOUPLOAD = "0", + DUCKPLYR_FALLBACK_VERBOSE = "FALSE" +) +options( + immundata.verbose = FALSE, + rlib_message_verbosity = "quiet" +) + library(immundata) test_check("immundata") diff --git a/tests/testthat/helper-data.R b/tests/testthat/helper-data.R new file mode 100644 index 0000000..fdaf343 --- /dev/null +++ b/tests/testthat/helper-data.R @@ -0,0 +1,133 @@ +make_bulk_count_test_data <- function() { + tibble::tibble( + sample_id = c("S1", "S1", "S2", "S2"), + v_call = c("TRBV1", "TRBV2", "TRBV3", "TRBV4"), + j_call = c("TRBJ1", "TRBJ2", "TRBJ1", "TRBJ2"), + junction_aa = c("AAAA", "BBBB", "CCCC", "DDDD"), + clone_count = c(8L, 7L, 6L, 9L) + ) +} + +make_single_cell_downsample_test_data <- function() { + tibble::tibble( + cell_id = paste0("c", seq_len(8L)), + sample_id = rep(c("S1", "S2"), each = 4L), + v_call = rep(paste0("IGHV", seq_len(4L)), 2L), + j_call = rep(paste0("IGHJ", seq_len(4L)), 2L), + junction_aa = paste0("CAR", LETTERS[seq_len(8L)]), + locus = "IGH", + umi_count = 10:17 + ) +} + +make_duplicate_chain_test_data <- function(tied = FALSE) { + umi_count <- if (tied) { + c(100L, 100L, 80L, 80L) + } else { + c(100L, 150L, 80L, 60L) + } + + tibble::tibble( + cell_id = rep("cell1", 4L), + v_call = c("IGHV1", "IGHV2", "IGLV1", "IGLV2"), + j_call = c("IGHJ1", "IGHJ2", "IGLJ1", "IGLJ2"), + junction_aa = c("CARW", "CBRW", "CASL", "CBSL"), + locus = c("IGH", "IGH", "IGL", "IGL"), + umi_count = umi_count + ) +} + +make_paired_filter_test_idata <- function() { + annotations <- tibble::tibble( + imd_receptor_id = rep(1L, 4L), + imd_barcode = c("bc1", "bc1", "bc2", "bc2"), + imd_chain_id = seq_len(4L), + imd_n_chains = rep(1L, 4L), + locus = rep(c("IGH", "IGL"), 2L), + cdr3_aa = rep(c("AAA", "CCC"), 2L), + sample_id = rep(c("S1", "S2"), each = 2L) + ) |> + duckplyr::as_duckdb_tibble() + + ImmunData$new( + schema = make_receptor_schema( + features = "cdr3_aa", + chains = c("IGH", "IGL") + ), + annotations = annotations + ) +} + +make_basic_test_annotations <- function() { + tibble::tibble( + imd_receptor_id = seq_len(4L), + imd_barcode = paste0("bc", seq_len(4L)), + imd_chain_id = seq_len(4L), + imd_n_chains = 1L, + cdr3_aa = c("AAA", "AAT", "AAAA", "BBB"), + v_call = c("V1", "V1", "V2", "V3"), + sample_id = c("S1", "S1", "S2", "S2") + ) |> + duckplyr::as_duckdb_tibble() +} + +make_single_chain_shared_receptor_test_data <- function() { + tibble::tibble( + cell_id = paste0("cell", seq_len(5L)), + sample_id = c("Sample1", "Sample1", "Sample1", "Sample2", "Sample2"), + v_call = c("IGHV1", "IGHV1", "IGHV2", "IGHV3", "IGHV4"), + j_call = c("IGHJ1", "IGHJ1", "IGHJ2", "IGHJ3", "IGHJ4"), + junction_aa = c("CARW", "CARW", "CBRW", "CCRW", "CDRW"), + locus = "IGH", + umi_count = c(100L, 150L, 200L, 250L, 300L) + ) +} + +make_relaxed_pairing_test_data <- function() { + tibble::tibble( + cell_id = c( + "normal_igl", "normal_igl", + "normal_igk", "normal_igk", + "artifact", "artifact", "artifact", + "heavy_only", + "light_only", + "two_lights", "two_lights" + ), + v_call = c( + "IGHV1", "IGLV1", + "IGHV2", "IGKV2", + "IGHV3", "IGLV3", "IGKV3", + "IGHV4", + "IGLV5", + "IGLV6", "IGKV6" + ), + j_call = c( + "IGHJ1", "IGLJ1", + "IGHJ2", "IGKJ2", + "IGHJ3", "IGLJ3", "IGKJ3", + "IGHJ4", + "IGLJ5", + "IGLJ6", "IGKJ6" + ), + junction_aa = c( + "CARW", "CASL", + "CBRW", "CBSK", + "CCRW", "CCSL", "CCSK", + "CDRW", + "CESL", + "CFSL", "CFSK" + ), + locus = c( + "IGH", "IGL", + "IGH", "IGK", + "IGH", "IGL", "IGK", + "IGH", + "IGL", + "IGL", "IGK" + ), + umi_count = c( + 100L, 80L, 120L, 90L, 110L, 85L, + 75L, 130L, 70L, 60L, 65L + ) + ) +} diff --git a/tests/testthat/helper-io.R b/tests/testthat/helper-io.R new file mode 100644 index 0000000..938cf46 --- /dev/null +++ b/tests/testthat/helper-io.R @@ -0,0 +1,428 @@ +create_test_output_dir <- function(name = "test_immundata_") { + tempfile(name) +} + +cleanup_output_dir <- function(dir) { + if (dir.exists(dir)) { + unlink(dir, recursive = TRUE) + } +} + +snapshot_test_root <- function() { + normalizePath(file.path(tempdir(), "imd_snap_tests"), mustWork = FALSE) +} + +create_snapshot_test_layout <- function() { + root <- snapshot_test_root() + if (dir.exists(root)) { + unlink(root, recursive = TRUE) + } + + dir.create(root, recursive = TRUE, showWarnings = FALSE) + project_a <- file.path(root, "projectA") + project_b <- file.path(root, "projectB") + dir.create(project_a, recursive = TRUE, showWarnings = FALSE) + dir.create(project_b, recursive = TRUE, showWarnings = FALSE) + + list( + root = root, + projectA = project_a, + projectB = project_b + ) +} + +cleanup_snapshot_test_root <- function() { + keep_snap_tests <- identical(Sys.getenv("IMD_KEEP_SNAP_TESTS"), "1") + if (keep_snap_tests) { + return(invisible(NULL)) + } + + root <- snapshot_test_root() + if (dir.exists(root)) { + unlink(root, recursive = TRUE) + } +} + +test_ig_data <- function() { + system.file("extdata/ig", "multiple_ig_loci.tsv.gz", package = "immundata") +} + +get_test_idata_tsv_no_manifest <- function( + schema = c("cdr3_aa", "v_call"), + repertoire_schema = "" +) { + sample_files <- c( + system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata"), + system.file("extdata/tsv", "sample_1k_2k.tsv", package = "immundata") + ) + read_repertoires( + path = sample_files, + schema = schema, + repertoire_schema = repertoire_schema, + output_folder = create_test_output_dir() + ) +} + +format_integrity_df_dump <- function(df) { + if (is.null(df)) { + return("") + } + + old_width <- getOption("width") + old_max_print <- getOption("max.print") + on.exit(options(width = old_width, max.print = old_max_print), add = TRUE) + options(width = 10000, max.print = 1e6) + + data_dump <- as.data.frame(df, stringsAsFactors = FALSE) + dump <- utils::capture.output(print(data_dump, row.names = FALSE, right = FALSE)) + if (length(dump) == 0) { + return("") + } + paste(dump, collapse = "\n") +} + +expect_agg_repertoires_integrity <- function( + idata, + context = NULL, + before_annotations = NULL, + before_repertoires = NULL, + schema = NULL +) { + checkmate::assert_r6(idata, "ImmunData") + checkmate::assert_data_frame(before_annotations, null.ok = TRUE) + checkmate::assert_data_frame(before_repertoires, null.ok = TRUE) + checkmate::assert_character(schema, null.ok = TRUE) + + if (is.null(context)) { + context <- "agg_repertoires integrity" + } + + ann <- idata$annotations |> collect() + reps <- idata$repertoires |> collect() + + ann_required <- c("imd_receptor_id", "imd_repertoire_id", "imd_count", "imd_proportion", "n_repertoires") + reps_required <- c("imd_repertoire_id", "n_barcodes", "n_receptors") + + ann_missing <- setdiff(ann_required, names(ann)) + reps_missing <- setdiff(reps_required, names(reps)) + + ann_na_cols <- intersect(ann_required, names(ann)) + reps_na_cols <- intersect(reps_required, names(reps)) + + ann_na_counts <- if (length(ann_na_cols) > 0) { + vapply(ann[ann_na_cols], function(x) sum(is.na(x)), integer(1)) + } else { + integer() + } + + reps_na_counts <- if (length(reps_na_cols) > 0) { + vapply(reps[reps_na_cols], function(x) sum(is.na(x)), integer(1)) + } else { + integer() + } + + unmatched_repertoire_ids <- NULL + unmatched_repertoire_ids_display <- "" + if ("imd_repertoire_id" %in% names(ann) && "imd_repertoire_id" %in% names(reps)) { + unmatched_repertoire_ids <- ann |> + dplyr::distinct(imd_repertoire_id) |> + dplyr::anti_join( + reps |> dplyr::distinct(imd_repertoire_id), + by = "imd_repertoire_id" + ) + + unmatched_repertoire_ids_display <- if (nrow(unmatched_repertoire_ids) > 0) { + paste(unmatched_repertoire_ids$imd_repertoire_id, collapse = ", ") + } else { + "" + } + } + + mapping_mismatches <- NULL + mapping_mismatches_display <- "" + count_check_missing <- character() + count_mismatches <- NULL + count_mismatches_display <- "" + if (!is.null(schema)) { + mapping_cols <- unique(c("imd_repertoire_id", schema)) + + if (all(mapping_cols %in% names(ann)) && all(mapping_cols %in% names(reps))) { + annotation_map <- ann |> + dplyr::distinct(dplyr::across(dplyr::all_of(mapping_cols))) + + repertoire_map <- reps |> + dplyr::distinct(dplyr::across(dplyr::all_of(mapping_cols))) + + mapping_mismatches <- dplyr::bind_rows( + annotation_map |> + dplyr::anti_join(repertoire_map, by = mapping_cols) |> + dplyr::mutate(.mapping_source = "annotations"), + repertoire_map |> + dplyr::anti_join(annotation_map, by = mapping_cols) |> + dplyr::mutate(.mapping_source = "repertoires") + ) + + mapping_mismatches_display <- as.character(nrow(mapping_mismatches)) + } + + count_annotation_cols <- unique(c( + schema, + "imd_receptor_id", + "imd_barcode", + "imd_n_chains" + )) + count_repertoire_cols <- unique(c( + schema, + "n_barcodes", + "n_receptors" + )) + missing_annotation_count_cols <- setdiff( + count_annotation_cols, + names(ann) + ) + missing_repertoire_count_cols <- setdiff( + count_repertoire_cols, + names(reps) + ) + count_check_missing <- c( + if (length(missing_annotation_count_cols) > 0) { + paste0("annotations.", missing_annotation_count_cols) + }, + if (length(missing_repertoire_count_cols) > 0) { + paste0("repertoires.", missing_repertoire_count_cols) + } + ) + + if (length(count_check_missing) == 0) { + expected_counts <- ann |> + dplyr::summarise( + imd_n_chains = dplyr::first(imd_n_chains), + .by = dplyr::all_of(c( + schema, + "imd_receptor_id", + "imd_barcode" + )) + ) |> + dplyr::summarise( + n_barcodes = sum(imd_n_chains), + n_receptors = dplyr::n_distinct(imd_receptor_id), + .by = dplyr::all_of(schema) + ) + + actual_counts <- reps |> + dplyr::select(dplyr::all_of(count_repertoire_cols)) + + count_mismatches <- dplyr::bind_rows( + expected_counts |> + dplyr::anti_join( + actual_counts, + by = count_repertoire_cols, + na_matches = "na" + ) |> + dplyr::mutate(.count_source = "recomputed from annotations"), + actual_counts |> + dplyr::anti_join( + expected_counts, + by = count_repertoire_cols, + na_matches = "na" + ) |> + dplyr::mutate(.count_source = "repertoire table") + ) + + count_mismatches_display <- as.character(nrow(count_mismatches)) + } + } + + diag_lines <- c( + paste0("context: ", context), + if (!is.null(schema)) { + paste0("agg schema: ", paste(schema, collapse = ", ")) + } else { + "agg schema: " + }, + if (!is.null(before_annotations)) { + paste0("input annotation shape: ", nrow(before_annotations), "x", ncol(before_annotations)) + } else { + "input annotation shape: " + }, + if (!is.null(before_repertoires)) { + paste0("input repertoire shape: ", nrow(before_repertoires), "x", ncol(before_repertoires)) + } else { + "input repertoire shape: " + }, + paste0("annotation shape: ", nrow(ann), "x", ncol(ann)), + paste0("repertoire shape: ", nrow(reps), "x", ncol(reps)), + if (length(ann_missing) > 0) { + paste0("missing annotation columns: ", paste(ann_missing, collapse = ", ")) + } else { + "missing annotation columns: " + }, + if (length(reps_missing) > 0) { + paste0("missing repertoire columns: ", paste(reps_missing, collapse = ", ")) + } else { + "missing repertoire columns: " + }, + if (length(ann_na_counts) > 0) { + paste0("annotation NA counts: ", paste(names(ann_na_counts), ann_na_counts, sep = "=", collapse = ", ")) + } else { + "annotation NA counts: " + }, + if (length(reps_na_counts) > 0) { + paste0("repertoire NA counts: ", paste(names(reps_na_counts), reps_na_counts, sep = "=", collapse = ", ")) + } else { + "repertoire NA counts: " + }, + paste0("unmatched repertoire ids: ", unmatched_repertoire_ids_display), + paste0("repertoire mapping mismatch rows: ", mapping_mismatches_display), + if (length(count_check_missing) > 0) { + paste0( + "missing count-check columns: ", + paste(count_check_missing, collapse = ", ") + ) + } else { + "missing count-check columns: " + }, + paste0("repertoire count mismatch rows: ", count_mismatches_display) + ) + diag <- paste(diag_lines, collapse = "\n") + + has_ann_na_mismatch <- length(ann_na_counts) > 0 && any(ann_na_counts != 0) + has_reps_na_mismatch <- length(reps_na_counts) > 0 && any(reps_na_counts != 0) + has_unmatched_repertoire_ids <- !is.null(unmatched_repertoire_ids) && nrow(unmatched_repertoire_ids) > 0 + has_count_mismatches <- !is.null(count_mismatches) && nrow(count_mismatches) > 0 + has_mismatch <- nrow(reps) <= 0 || + length(ann_missing) > 0 || + length(reps_missing) > 0 || + has_ann_na_mismatch || + has_reps_na_mismatch || + has_unmatched_repertoire_ids || + length(count_check_missing) > 0 || + has_count_mismatches + + if (has_mismatch) { + before_input_dump <- if (!is.null(before_annotations) || !is.null(before_repertoires)) { + paste0( + "\n\ninput annotations dump:\n", + format_integrity_df_dump(before_annotations), + "\n\ninput repertoires dump:\n", + format_integrity_df_dump(before_repertoires) + ) + } else { + "" + } + + diag <- paste0( + diag, + before_input_dump, + "\n\nannotations dump:\n", + format_integrity_df_dump(ann), + "\n\nrepertoires dump:\n", + format_integrity_df_dump(reps) + ) + } + + testthat::expect_true(nrow(reps) > 0, info = diag) + testthat::expect_equal(length(ann_missing), 0, info = diag) + testthat::expect_equal(length(reps_missing), 0, info = diag) + + if (length(ann_na_counts) > 0) { + testthat::expect_true(all(ann_na_counts == 0), info = diag) + } + if (length(reps_na_counts) > 0) { + testthat::expect_true(all(reps_na_counts == 0), info = diag) + } + + if (!is.null(unmatched_repertoire_ids)) { + testthat::expect_equal( + nrow(unmatched_repertoire_ids), + 0, + info = diag + ) + } + + if (!is.null(mapping_mismatches)) { + mapping_mismatches_preview <- utils::head(mapping_mismatches, 20L) + mapping_diag <- paste0( + diag, + "\n\nrepertoire mapping mismatches:\n", + format_integrity_df_dump(mapping_mismatches_preview), + if (nrow(mapping_mismatches) > nrow(mapping_mismatches_preview)) { + paste0( + "\n... ", + nrow(mapping_mismatches) - nrow(mapping_mismatches_preview), + " additional mismatch rows omitted" + ) + } else { + "" + } + ) + + testthat::expect_equal( + nrow(mapping_mismatches), + 0L, + info = mapping_diag + ) + } + + if (!is.null(schema)) { + testthat::expect_equal( + length(count_check_missing), + 0L, + info = diag + ) + } + + if (!is.null(count_mismatches)) { + count_mismatches_preview <- utils::head(count_mismatches, 20L) + count_diag <- paste0( + diag, + "\n\nrepertoire count mismatches:\n", + format_integrity_df_dump(count_mismatches_preview), + if (nrow(count_mismatches) > nrow(count_mismatches_preview)) { + paste0( + "\n... ", + nrow(count_mismatches) - nrow(count_mismatches_preview), + " additional mismatch rows omitted" + ) + } else { + "" + } + ) + + testthat::expect_equal( + nrow(count_mismatches), + 0L, + info = count_diag + ) + } + + invisible(list( + annotations = ann, + repertoires = reps + )) +} + +agg_repertoires_with_integrity <- function(idata, schema, context = NULL) { + checkmate::assert_r6(idata, "ImmunData") + checkmate::assert_character(schema, min.len = 1) + checkmate::assert_character(context, null.ok = TRUE) + + input_annotations <- idata$annotations |> collect() + input_repertoires <- if (is.null(idata$repertoires)) { + NULL + } else { + idata$repertoires |> collect() + } + + idata_agg <- agg_repertoires(idata, schema = schema) + + expect_agg_repertoires_integrity( + idata = idata_agg, + context = context, + before_annotations = input_annotations, + before_repertoires = input_repertoires, + schema = schema + ) + + idata_agg +} diff --git a/tests/testthat/setup.R b/tests/testthat/setup.R new file mode 100644 index 0000000..eb442b0 --- /dev/null +++ b/tests/testthat/setup.R @@ -0,0 +1,4 @@ +# Avoid nested parallelism: testthat already runs test files in parallel. +duckplyr_db_exec <- utils::getFromNamespace("db_exec", "duckplyr") +duckplyr_db_exec("SET threads TO 1") +rm(duckplyr_db_exec) diff --git a/tests/testthat/test-agg-strata.R b/tests/testthat/test-agg-strata.R new file mode 100644 index 0000000..49c3b32 --- /dev/null +++ b/tests/testthat/test-agg-strata.R @@ -0,0 +1,350 @@ +test_that("agg_strata adds id and name to repertoires, id only to annotations", { + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) + + stratified <- agg_strata(idata, schema = "Response") + rep_tbl <- stratified$repertoires + + expect_s3_class(rep_tbl, "tbl_df") + expect_true(strata_col %in% names(rep_tbl)) + expect_true(strata_name_col %in% names(rep_tbl)) + expect_true(strata_col %in% names(stratified$annotations)) + expect_false(strata_name_col %in% names(stratified$annotations)) + expect_equal(stratified$schema_strata, "Response") + expect_equal( + names(stratified$strata), + c(strata_col, strata_name_col, "Response") + ) + + expected_n <- length(unique(rep_tbl$Response)) + observed_n <- length(unique(rep_tbl[[strata_col]])) + expect_equal(observed_n, expected_n) + + rep_names <- unique(rep_tbl[c(strata_col, strata_name_col)]) + expect_equal(anyDuplicated(rep_names[[strata_col]]), 0) + expect_equal(anyDuplicated(rep_names[[strata_name_col]]), 0) +}) + +test_that("agg_strata validates grouping columns", { + idata <- get_test_idata() |> + agg_repertoires("Response") + + expect_error( + agg_strata(idata, schema = "not_a_metadata_column"), + "not found in idata\\$repertoires" + ) +}) + +test_that("agg_strata keeps repertoire-strata mapping consistent", { + repertoire_col <- imd_schema("repertoire") + strata_col <- imd_schema("strata") + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) + + stratified <- agg_strata(idata, schema = c("Response", "Therapy")) + + rep_map <- unique(stratified$repertoires[c(repertoire_col, strata_col)]) + rep_map <- rep_map[order(rep_map[[repertoire_col]]), , drop = FALSE] + + ann_map <- stratified$annotations |> + dplyr::select(all_of(c(repertoire_col, strata_col))) |> + dplyr::collect() + ann_map <- unique(ann_map[c(repertoire_col, strata_col)]) + ann_map <- ann_map[order(ann_map[[repertoire_col]]), , drop = FALSE] + + expect_equal(nrow(rep_map), nrow(ann_map)) + expect_equal(anyDuplicated(rep_map[[repertoire_col]]), 0) + expect_equal(anyDuplicated(ann_map[[repertoire_col]]), 0) + expect_equal( + as.data.frame(rep_map, stringsAsFactors = FALSE), + as.data.frame(ann_map, stringsAsFactors = FALSE), + ignore_attr = TRUE + ) +}) + +test_that("agg_strata handles NA group values", { + repertoire_col <- imd_schema("repertoire") + strata_col <- imd_schema("strata") + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) + + rep_tbl <- idata$repertoires + rep_tbl$Response[1] <- NA_character_ + + idata_with_na <- ImmunData$new( + schema = idata$schema_receptor, + annotations = idata$annotations, + repertoires = rep_tbl + ) + + stratified <- agg_strata(idata_with_na, schema = "Response") + strat_rep <- stratified$repertoires + + expect_equal( + length(unique(strat_rep[[strata_col]])), + length(unique(strat_rep$Response)) + ) + + na_rep <- unique(strat_rep[is.na(strat_rep$Response), c(repertoire_col, strata_col), drop = FALSE]) + expect_equal(nrow(na_rep), 1) + + ann_na_strata <- stratified$annotations |> + dplyr::select(all_of(c(repertoire_col, strata_col))) |> + dplyr::collect() + ann_na_strata <- unique( + ann_na_strata[ann_na_strata[[repertoire_col]] == na_rep[[repertoire_col]], strata_col, drop = FALSE] + ) + + expect_equal(nrow(ann_na_strata), 1) + expect_equal(ann_na_strata[[strata_col]], na_rep[[strata_col]]) +}) + +test_that("agg_strata requires repertoire aggregation", { + idata <- ImmunData$new( + schema = "cdr3_aa", + annotations = make_basic_test_annotations() + ) + + expect_error( + agg_strata(idata, schema = "Response"), + "agg_repertoires" + ) +}) + +test_that("agg_strata keeps repertoire schema and strata metadata are dropped on re-aggregation", { + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) + + stratified <- agg_strata(idata, schema = "Response") + + expect_equal(stratified$schema_repertoire, idata$schema_repertoire) + expect_equal(stratified$schema_strata, "Response") + expect_true(strata_col %in% names(stratified$annotations)) + expect_true(strata_col %in% names(stratified$repertoires)) + expect_true(strata_name_col %in% names(stratified$repertoires)) + + reaggregated <- agg_repertoires(stratified, schema = stratified$schema_repertoire) + + expect_false(strata_col %in% names(reaggregated$annotations)) + expect_false(strata_name_col %in% names(reaggregated$annotations)) + expect_false(strata_col %in% names(reaggregated$repertoires)) + expect_false(strata_name_col %in% names(reaggregated$repertoires)) + expect_null(reaggregated$schema_strata) + expect_null(reaggregated$strata) +}) + +test_that("re-aggregation rebuilds repertoire ids and removes strata metadata", { + repertoire_col <- imd_schema("repertoire") + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) |> + agg_strata(schema = "Response") + + poisoned_annotations <- idata$annotations |> + dplyr::mutate(!!repertoire_col := 999999L, !!strata_col := 999999L) + + poisoned_repertoires <- idata$repertoires + poisoned_repertoires[[repertoire_col]] <- 999999L + poisoned_repertoires[[strata_col]] <- 999999L + poisoned_repertoires[[strata_name_col]] <- "BrokenName" + + poisoned_idata <- ImmunData$new( + schema = idata$schema_receptor, + annotations = poisoned_annotations, + repertoires = poisoned_repertoires + ) + + reaggregated <- agg_repertoires(poisoned_idata, schema = poisoned_idata$schema_repertoire) + + expect_false(strata_col %in% names(reaggregated$annotations)) + expect_false(strata_name_col %in% names(reaggregated$annotations)) + expect_false(strata_col %in% names(reaggregated$repertoires)) + expect_false(strata_name_col %in% names(reaggregated$repertoires)) + + expect_true(repertoire_col %in% names(reaggregated$annotations)) + expect_true(repertoire_col %in% names(reaggregated$repertoires)) + + ann_tbl <- reaggregated$annotations |> + dplyr::select(all_of(repertoire_col)) |> + dplyr::collect() + rep_tbl <- reaggregated$repertoires + + expect_false(any(ann_tbl[[repertoire_col]] == 999999L)) + expect_false(any(rep_tbl[[repertoire_col]] == 999999L)) + expect_equal(length(unique(rep_tbl[[repertoire_col]])), nrow(rep_tbl)) + expect_true(all(ann_tbl[[repertoire_col]] %in% rep_tbl[[repertoire_col]])) +}) + +test_that("agg_strata can be re-run and overwrites previous strata assignment", { + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) + + rep_tbl <- idata$repertoires + rep_tbl$Response <- "mixed" + + mutated <- ImmunData$new( + schema = idata$schema_receptor, + annotations = idata$annotations, + repertoires = rep_tbl + ) + + stratified_once <- agg_strata(mutated, schema = "Response") + expect_equal(length(unique(stratified_once$repertoires[[strata_col]])), 1) + + stratified_twice <- agg_strata(stratified_once, schema = "Therapy") + expect_equal(length(unique(stratified_twice$repertoires[[strata_col]])), 2) + expect_true(strata_name_col %in% names(stratified_twice$repertoires)) + expect_false(strata_name_col %in% names(stratified_twice$annotations)) +}) + +test_that("agg_strata validates schema argument contract", { + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) + + expect_error(agg_strata(idata, schema = character()), "Assertion on 'schema' failed") + expect_error(agg_strata(idata, schema = NA_character_), "Assertion on 'schema' failed") + expect_error(agg_strata(idata, schema = c("Response", "Response")), "Assertion on 'schema' failed") + expect_error(agg_strata(idata, schema = 1), "Assertion on 'schema' failed") +}) + +test_that("agg_strata errors clearly when repertoire id column is missing", { + repertoire_col <- imd_schema("repertoire") + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) + + bad_annotations <- idata$annotations |> + dplyr::select(-all_of(repertoire_col)) + + bad_ann_idata <- ImmunData$new( + schema = idata$schema_receptor, + annotations = bad_annotations, + repertoires = idata$repertoires + ) + + expect_error( + agg_strata(bad_ann_idata, schema = "Response"), + "missing in .*idata\\$annotations" + ) + + bad_repertoires <- idata$repertoires + bad_repertoires[[repertoire_col]] <- NULL + + bad_rep_idata <- ImmunData$new( + schema = idata$schema_receptor, + annotations = idata$annotations, + repertoires = bad_repertoires + ) + + expect_error( + agg_strata(bad_rep_idata, schema = "Response"), + "missing in .*idata\\$repertoires" + ) +}) + +test_that("rename_strata renames using a full named vector mapping", { + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) |> + agg_strata(schema = "Response") + + strata_ids <- sort(unique(idata$repertoires[[strata_col]])) + new_labels <- paste0("Group_", strata_ids) + names(new_labels) <- as.character(strata_ids) + + renamed <- rename_strata(idata, names = new_labels) + rep_tbl <- renamed$repertoires + + expect_equal(renamed$schema_strata, idata$schema_strata) + expect_equal( + sort(unique(rep_tbl[[strata_name_col]])), + sort(unname(new_labels)) + ) + expect_false(strata_name_col %in% names(renamed$annotations)) +}) + +test_that("print.ImmunData shows strata schema and strata table", { + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) |> + agg_strata(schema = "Response") + + output <- testthat::capture_messages(print(idata)) + + expect_true(any(grepl("Strata schema", output, fixed = TRUE))) + expect_true(any(grepl("List of strata", output, fixed = TRUE))) +}) + +test_that("rename_strata supports partial renaming with unnamed policy", { + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) |> + agg_strata(schema = "Response") + + strata_ids <- sort(unique(idata$repertoires[[strata_col]])) + partial_map <- c("CustomA") + names(partial_map) <- as.character(strata_ids[1]) + + expect_error( + rename_strata(idata, names = partial_map), + "Missing names for strata ID" + ) + + auto_named <- rename_strata(idata, names = partial_map, unnamed = "auto", auto_prefix = "Auto") + rep_tbl <- auto_named$repertoires + + expect_true("CustomA" %in% rep_tbl[[strata_name_col]]) + expect_true(any(grepl("^Auto", rep_tbl[[strata_name_col]]))) + + keep_named <- rename_strata(idata, names = partial_map, unnamed = "keep") + keep_tbl <- keep_named$repertoires + expect_true("CustomA" %in% keep_tbl[[strata_name_col]]) +}) + +test_that("rename_strata validates mapping integrity", { + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) |> + agg_strata(schema = "Response") + + strata_ids <- sort(unique(idata$repertoires[[strata_col]])) + + bad_dup_label <- data.frame( + imd_strata_id = strata_ids, + strata_name = rep("dup", length(strata_ids)) + ) + expect_error(rename_strata(idata, names = bad_dup_label), "duplicated strata labels") + + bad_unknown_id <- data.frame( + imd_strata_id = 999999L, + strata_name = "Unknown" + ) + expect_error(rename_strata(idata, names = bad_unknown_id), "Unknown strata ID") + + bad_empty <- data.frame( + imd_strata_id = strata_ids[1], + strata_name = "" + ) + expect_error(rename_strata(idata, names = bad_empty), "non-empty strings") + + rep_tbl <- rename_strata(idata, names = setNames("OkName", as.character(strata_ids[1])), unnamed = "auto")$repertoires + expect_true(strata_name_col %in% names(rep_tbl)) +}) diff --git a/tests/testthat/test-annotate-barcodes.R b/tests/testthat/test-annotate-barcodes.R index d066165..bc9124d 100644 --- a/tests/testthat/test-annotate-barcodes.R +++ b/tests/testthat/test-annotate-barcodes.R @@ -1,5 +1,5 @@ test_that("annotate_barcodes adds cell‑level annotations", { - idata <- get_test_idata_tsv_no_metadata() + idata <- get_test_idata_tsv_no_manifest() cell_id_col <- imd_schema()$cell idata <- ImmunData$new(schema = idata$schema_receptor, diff --git a/tests/testthat/test-annotate-external.R b/tests/testthat/test-annotate-external.R new file mode 100644 index 0000000..62312d1 --- /dev/null +++ b/tests/testthat/test-annotate-external.R @@ -0,0 +1,199 @@ +warn_if_pkg_missing <- function(pkg, test_context) { + if (!requireNamespace(pkg, quietly = TRUE)) { + warning( + sprintf( + "Package '%s' is not installed; %s checks were not executed.", + pkg, + test_context + ), + call. = FALSE + ) + return(TRUE) + } + + FALSE +} + +make_external_idata <- function() { + idata <- get_test_idata_tsv_no_manifest() + barcode_col <- imd_schema("barcode") + + barcode_tbl <- idata$annotations |> + dplyr::select(barcode = !!rlang::sym(barcode_col)) |> + dplyr::distinct(.data$barcode, .keep_all = TRUE) |> + dplyr::collect() |> + dplyr::slice_head(n = 2) + + ann <- data.frame( + barcode = c(barcode_tbl$barcode[1], barcode_tbl$barcode[1], barcode_tbl$barcode[2]), + external_score = c(1L, 999999L, 2L), + stringsAsFactors = FALSE + ) + + idata_annotated <- annotate_barcodes(idata, annotations = ann, annot_col = "barcode") + + first_scores <- idata_annotated$annotations |> + dplyr::select(barcode = !!rlang::sym(barcode_col), external_score) |> + dplyr::collect() |> + as.data.frame() + + first_score <- first_scores$external_score[ + match(barcode_tbl$barcode[1], first_scores$barcode) + ] + + list( + idata = idata_annotated, + barcode_col = barcode_col, + first_barcode = barcode_tbl$barcode[1], + first_score = first_score + ) +} + +make_mock_adata <- function(obs_names) { + structure( + list( + obs_names = obs_names, + obs = data.frame( + placeholder = seq_along(obs_names), + row.names = paste0("row_", seq_along(obs_names)) + ) + ), + class = "AbstractAnnData" + ) +} + +testthat::test_that("annotate_seurat transfers selected columns by barcode", { + if (warn_if_pkg_missing("Seurat", "annotate_seurat")) { + testthat::succeed() + return(invisible(NULL)) + } + + idata <- get_test_idata_tsv_no_manifest() + barcode_col <- imd_schema("barcode") + + ann <- idata$annotations |> + dplyr::select(barcode = !!rlang::sym(barcode_col), cdr3_aa) |> + dplyr::distinct(.data$barcode, .keep_all = TRUE) |> + dplyr::collect() + + cells <- c(ann$barcode[1:3], "barcode_not_found") + counts <- matrix( + c(1, 0, 2, 3, 4, 5, 6, 7), + nrow = 2, + dimnames = list(c("geneA", "geneB"), cells) + ) + sdata <- suppressWarnings(Seurat::CreateSeuratObject(counts = counts)) + + out <- annotate_seurat(idata, sdata, cols = "cdr3_aa") + + expected <- ann$cdr3_aa[match(cells, ann$barcode)] + actual <- out@meta.data[cells, "cdr3_aa", drop = TRUE] + + testthat::expect_equal(unname(actual), expected) +}) + +testthat::test_that("annotate_seurat keeps first record per duplicated barcode", { + if (warn_if_pkg_missing("Seurat", "annotate_seurat")) { + testthat::succeed() + return(invisible(NULL)) + } + + ext <- make_external_idata() + + counts <- matrix( + c(1, 0, 2, 3, 4, 5), + nrow = 2, + dimnames = list( + c("geneA", "geneB"), + c(ext$first_barcode, "barcode_not_found_1", "barcode_not_found_2") + ) + ) + sdata <- suppressWarnings(Seurat::CreateSeuratObject(counts = counts)) + + out <- annotate_seurat(ext$idata, sdata, cols = "external_score") + + testthat::expect_equal( + as.integer(out@meta.data[ext$first_barcode, "external_score", drop = TRUE]), + as.integer(ext$first_score) + ) +}) + +testthat::test_that("annotate_seurat errors for missing annotation columns", { + if (warn_if_pkg_missing("Seurat", "annotate_seurat")) { + testthat::succeed() + return(invisible(NULL)) + } + + idata <- get_test_idata_tsv_no_manifest() + + counts <- matrix( + c(1, 0, 2, 3, 4, 5), + nrow = 2, + dimnames = list(c("geneA", "geneB"), c("barcode_1", "barcode_2", "barcode_3")) + ) + sdata <- suppressWarnings(Seurat::CreateSeuratObject(counts = counts)) + + testthat::expect_error( + annotate_seurat(idata, sdata, cols = "column_does_not_exist"), + "not found in idata\\$annotations" + ) +}) + +testthat::test_that("annotate_anndata transfers selected columns by obs_names", { + ext <- make_external_idata() + annotate_anndata <- immundata:::annotate_anndata + + adata <- make_mock_adata(c(ext$first_barcode, "barcode_not_found")) + + out <- annotate_anndata(ext$idata, adata, cols = "external_score") + + testthat::expect_equal( + as.integer(out$obs$external_score[1]), + as.integer(ext$first_score) + ) + testthat::expect_true(is.na(out$obs$external_score[2])) +}) + +testthat::test_that("annotate_anndata validates obs_names and requested columns", { + idata <- get_test_idata_tsv_no_manifest() + annotate_anndata <- immundata:::annotate_anndata + + adata_empty <- structure( + list(obs_names = character(0), obs = data.frame(placeholder = integer())), + class = "AbstractAnnData" + ) + testthat::expect_error( + annotate_anndata(idata, adata_empty, cols = "cdr3_aa"), + "missing or empty" + ) + + adata_bad_names <- structure( + list( + obs_names = c("cell_1", ""), + obs = data.frame(placeholder = 1:2, row.names = c("r1", "r2")) + ), + class = "AbstractAnnData" + ) + testthat::expect_error( + annotate_anndata(idata, adata_bad_names, cols = "cdr3_aa"), + "contains NA/empty values" + ) + + adata_dup <- structure( + list( + obs_names = c("cell_1", "cell_1"), + obs = data.frame(placeholder = 1:2, row.names = c("r1", "r2")) + ), + class = "AbstractAnnData" + ) + testthat::expect_error( + annotate_anndata(idata, adata_dup, cols = "cdr3_aa"), + "must be unique" + ) + + adata_ok <- make_mock_adata(c("cell_1", "cell_2")) + testthat::expect_error( + annotate_anndata(idata, adata_ok, cols = "column_does_not_exist"), + "not found in idata\\$annotations" + ) +}) diff --git a/tests/testthat/test-annotate-immundata.R b/tests/testthat/test-annotate-immundata.R index e69de29..ff48da0 100644 --- a/tests/testthat/test-annotate-immundata.R +++ b/tests/testthat/test-annotate-immundata.R @@ -0,0 +1,401 @@ +make_annotate_test_idata <- function() { + ImmunData$new( + schema = c("cdr3_aa", "v_call"), + annotations = make_basic_test_annotations(), + provenance = list( + home_path = tempdir(), + current_path = tempdir(), + snapshot_id = "annotate-test-snapshot", + lineage = list(list(event = "fixture")) + ) + ) +} + +test_that("annotate_immundata left-joins annotations with renamed keys", { + idata <- make_annotate_test_idata() + ann <- tibble::tibble( + external_v = c("V1", "V2"), + receptor_family = c("alpha", "beta") + ) + + out <- annotate_immundata( + idata, + annotations = ann, + by = c("v_call" = "external_v"), + keep_repertoires = FALSE + ) + + expected <- idata$annotations |> + collect() |> + left_join(ann, by = join_by(v_call == external_v)) |> + arrange(imd_receptor_id) + + actual <- out$annotations |> + collect() |> + arrange(imd_receptor_id) + + expect_equal(actual, expected) + expect_true(is.na(actual$receptor_family[actual$v_call == "V3"])) +}) + +test_that("annotate alias and annotate_immundata produce equivalent annotations", { + idata <- make_annotate_test_idata() + ann <- tibble::tibble(v_call = c("V1", "V2"), receptor_family = c("alpha", "beta")) + + direct <- annotate_immundata( + idata, + annotations = ann, + by = c("v_call" = "v_call"), + keep_repertoires = FALSE + ) + alias <- annotate( + idata, + annotations = ann, + by = c("v_call" = "v_call"), + keep_repertoires = FALSE + ) + + expect_equal( + direct$annotations |> collect() |> arrange(imd_receptor_id), + alias$annotations |> collect() |> arrange(imd_receptor_id) + ) +}) + +test_that("annotate_immundata supports same-name and multi-column keys", { + idata <- make_annotate_test_idata() + + same_name <- tibble::tibble(v_call = "V1", v_label = "same-name") + same_name_out <- annotate_immundata( + idata, + annotations = same_name, + by = c("v_call" = "v_call"), + keep_repertoires = FALSE + )$annotations |> + collect() |> + arrange(imd_receptor_id) + + expect_equal(same_name_out$v_label, c("same-name", "same-name", NA, NA)) + + pair_ann <- tibble::tibble( + external_v = c("V1", "V1", "V2"), + external_cdr3 = c("AAA", "AAT", "AAA"), + pair_label = c("hit-1", "hit-2", "miss") + ) + pair_out <- annotate_immundata( + idata, + annotations = pair_ann, + by = c("v_call" = "external_v", "cdr3_aa" = "external_cdr3"), + keep_repertoires = FALSE + )$annotations |> + collect() |> + arrange(imd_receptor_id) + + expect_equal(pair_out$pair_label, c("hit-1", "hit-2", NA, NA)) +}) + +test_that("non-unique annotation keys violate the contract and expand rows", { + idata <- make_annotate_test_idata() + ann <- tibble::tibble( + v_call = c("V1", "V1"), + label = c("first", "second") + ) + + out <- annotate_immundata( + idata, + annotations = ann, + by = c("v_call" = "v_call"), + keep_repertoires = FALSE + ) + + expected <- idata$annotations |> + collect() |> + left_join(ann, by = "v_call") |> + arrange(imd_receptor_id, label) + + actual <- out$annotations |> + collect() |> + arrange(imd_receptor_id, label) + + expect_equal(nrow(actual), nrow(expected)) + expect_equal(actual, expected) +}) + +test_that("annotate_immundata preserves repertoire and strata state", { + idata <- make_annotate_test_idata() |> + agg_repertoires("sample_id") |> + agg_strata("sample_id") + + ann <- tibble::tibble( + v_call = c("V1", "V2", "V3"), + receptor_family = c("alpha", "beta", "gamma") + ) + + reps_before <- idata$repertoires + strata_before <- idata$strata + annotation_state_cols <- c( + "imd_receptor_id", + "imd_repertoire_id", + "imd_strata_id", + "imd_count", + "imd_proportion", + "n_repertoires" + ) + annotation_state_before <- idata$annotations |> + select(all_of(annotation_state_cols)) |> + collect() |> + arrange(across(everything())) + + out <- annotate_immundata( + idata, + annotations = ann, + by = c("v_call" = "v_call"), + keep_repertoires = TRUE + ) + + expect_true("receptor_family" %in% names(out$annotations)) + expect_equal(out$repertoires, reps_before) + expect_equal(out$strata, strata_before) + expect_equal(out$schema_repertoire, idata$schema_repertoire) + expect_equal(out$schema_strata, idata$schema_strata) + expect_equal( + out$annotations |> + select(all_of(annotation_state_cols)) |> + collect() |> + arrange(across(everything())), + annotation_state_before + ) +}) + +test_that("annotate_immundata drops all repertoire state when keep_repertoires is FALSE", { + idata <- make_annotate_test_idata() |> + agg_repertoires("sample_id") |> + agg_strata("sample_id") + + out <- annotate_immundata( + idata, + annotations = tibble::tibble(v_call = "V1", receptor_family = "alpha"), + by = c("v_call" = "v_call"), + keep_repertoires = FALSE + ) + + expect_null(out$repertoires) + expect_null(out$strata) + expect_null(out$schema_repertoire) + expect_null(out$schema_strata) + expect_length( + intersect( + colnames(out$annotations), + c( + "imd_repertoire_id", + "imd_strata_id", + "strata_name", + "imd_count", + "imd_proportion", + "n_receptors", + "n_barcodes", + "n_repertoires" + ) + ), + 0 + ) +}) + +test_that("annotate_immundata preserves provenance", { + idata <- make_annotate_test_idata() + prov_before <- get_provenance(idata) + + out <- annotate_immundata( + idata, + annotations = tibble::tibble(v_call = "V1", receptor_family = "alpha"), + by = c("v_call" = "v_call"), + keep_repertoires = FALSE + ) + + expect_equal( + get_provenance(out)[sort(names(get_provenance(out)))], + prov_before[sort(names(prov_before))] + ) +}) + +test_that("annotate_immundata validates join inputs", { + idata <- make_annotate_test_idata() + + expect_error( + annotate_immundata( + idata, + annotations = tibble::tibble(external_v = "V1", receptor_family = "alpha"), + by = c("v_call" = "missing_col") + ), + "not found in annotations" + ) + + expect_error( + annotate_immundata( + idata, + annotations = tibble::tibble(external_v = "V1", receptor_family = "alpha"), + by = c("missing_col" = "external_v") + ), + "not found in ImmunData" + ) + + expect_error( + annotate_immundata( + idata, + annotations = tibble::tibble(external_v = "V1", receptor_family = "alpha"), + by = "external_v" + ), + "Assertion on 'by' failed" + ) +}) + +test_that("annotate_immundata allows absent schema-named annotation columns", { + idata <- make_annotate_test_idata() + + out <- annotate_immundata( + idata, + annotations = tibble::tibble( + v_call = c("V1", "V2", "V3"), + imd_group_id = c("group-1", "group-2", "group-3"), + j_gene = c("J1", "J2", "J3") + ), + by = c("v_call" = "v_call"), + keep_repertoires = FALSE + )$annotations |> + collect() |> + arrange(imd_receptor_id) + + expect_equal( + out$imd_group_id, + c("group-1", "group-1", "group-2", "group-3") + ) + expect_equal(out$j_gene, c("J1", "J1", "J2", "J3")) +}) + +test_that("annotate_immundata blocks annotation columns that already exist", { + idata <- make_annotate_test_idata() + + expect_error( + annotate_immundata( + idata, + annotations = tibble::tibble(external_v = "V1", cdr3_aa = "collision"), + by = c("v_call" = "external_v") + ), + "collide" + ) +}) + +test_that("annotate_immundata can replace annotation columns that already exist", { + idata <- annotate_immundata( + make_annotate_test_idata(), + annotations = tibble::tibble( + v_call = c("V1", "V2", "V3"), + imd_group_id = c("old-1", "old-2", "old-3") + ), + by = c("v_call" = "v_call"), + keep_repertoires = FALSE + ) + + out <- annotate_immundata( + idata, + annotations = tibble::tibble( + v_call = c("V1", "V2", "V3"), + imd_group_id = c("new-1", "new-2", "new-3") + ), + by = c("v_call" = "v_call"), + keep_repertoires = FALSE, + conflicts = "replace" + )$annotations |> + collect() |> + arrange(imd_receptor_id) + + expect_equal( + out$imd_group_id, + c("new-1", "new-1", "new-2", "new-3") + ) +}) + +test_that("annotate_immundata cannot replace protected state or schema columns", { + idata <- make_annotate_test_idata() |> + agg_repertoires("sample_id") |> + agg_strata("sample_id") + + expect_error( + annotate_immundata( + idata, + annotations = tibble::tibble( + v_call = c("V1", "V2", "V3"), + imd_count = c(10L, 20L, 30L) + ), + by = c("v_call" = "v_call"), + conflicts = "replace" + ), + "protected ImmunData columns" + ) + + expect_error( + annotate_immundata( + idata, + annotations = tibble::tibble( + v_call = c("V1", "V2", "V3"), + sample_id = c("new-1", "new-2", "new-3") + ), + by = c("v_call" = "v_call"), + conflicts = "replace" + ), + "protected ImmunData columns" + ) +}) + +test_that("annotate_immundata enforces and can bypass the wide annotation guard", { + idata <- make_annotate_test_idata() + wide <- as.data.frame( + as.list(stats::setNames(rep("x", 100), paste0("feature_", seq_len(100)))), + stringsAsFactors = FALSE + ) + wide$v_call <- "V1" + + expect_error( + annotate_immundata( + idata, + annotations = wide, + by = c("v_call" = "v_call") + ), + "you have been warned" + ) + + out <- annotate_immundata( + idata, + annotations = wide, + by = c("v_call" = "v_call"), + remove_limit = TRUE, + keep_repertoires = FALSE + ) + + expect_true("feature_1" %in% names(out$annotations)) +}) + +test_that("annotate_chains annotates by explicit chain identifier column", { + idata <- make_annotate_test_idata() + ann <- tibble::tibble( + chain_id = c(1L, 3L), + chain_label = c("chain-a", "chain-c") + ) + + out <- annotate_chains( + idata, + annotations = ann, + annot_col = "chain_id", + keep_repertoires = FALSE + ) + + expected <- idata$annotations |> + collect() |> + left_join(ann, by = join_by(imd_chain_id == chain_id)) |> + arrange(imd_chain_id) + + actual <- out$annotations |> + collect() |> + arrange(imd_chain_id) + + expect_equal(actual, expected) +}) diff --git a/tests/testthat/test-annotate-receptors.R b/tests/testthat/test-annotate-receptors.R index 67e4e8b..7701d09 100644 --- a/tests/testthat/test-annotate-receptors.R +++ b/tests/testthat/test-annotate-receptors.R @@ -1,5 +1,5 @@ testthat::test_that("annotate_receptors adds receptor‑level annotations", { - idata <- get_test_idata_tsv_no_metadata() + idata <- get_test_idata_tsv_no_manifest() receptor_id_col <- imd_schema()$receptor recs <- idata$receptors %>% @@ -18,13 +18,10 @@ testthat::test_that("annotate_receptors adds receptor‑level annotations", { annotations = ann ) - actual_annot <- out$annotations |> - collect() |> - arrange(across(everything())) + actual_annot <- out$annotations |> collect() expected_annot <- idata$annotations |> collect() |> - left_join(ann, by = join_by(imd_receptor_id == imd_receptor_id)) |> - arrange(across(everything())) + left_join(ann, by = join_by(imd_receptor_id == imd_receptor_id)) expect_equal( actual_annot |> count(), @@ -37,7 +34,34 @@ testthat::test_that("annotate_receptors adds receptor‑level annotations", { ) expect_equal( - actual_annot, - expected_annot + actual_annot |> + select(all_of(sort(colnames(actual_annot)))) |> + arrange(across(everything())), + expected_annot |> + select(all_of(sort(colnames(expected_annot)))) |> + arrange(across(everything())) ) }) + +testthat::test_that("annotate_receptors preserves join column order without repertoires", { + idata <- get_test_idata_tsv_no_manifest(repertoire_schema = NULL) + receptor_id_col <- imd_schema()$receptor + + recs <- idata$receptors |> + select(!!sym(receptor_id_col), cdr3_aa) |> + collect() |> + head(5) + ann <- tibble( + imd_receptor_id = recs[[receptor_id_col]], + receptor_seq = paste0("ANN_", recs$cdr3_aa) + ) + + actual <- annotate_receptors(idata, annotations = ann)$annotations |> + collect() + expected <- idata$annotations |> + collect() |> + left_join(ann, by = "imd_receptor_id") + + expect_null(idata$repertoires) + expect_equal(actual, expected) +}) diff --git a/tests/testthat/test-audit-edge-cases.R b/tests/testthat/test-audit-edge-cases.R new file mode 100644 index 0000000..00a96f1 --- /dev/null +++ b/tests/testthat/test-audit-edge-cases.R @@ -0,0 +1,94 @@ +test_that("agg_receptors standardizes a custom locus column before filtering", { + dataset <- duckplyr::as_duckdb_tibble( + tibble::tibble( + cell_id = c("cell_1", "cell_2", "cell_3"), + chain_locus = c("IGH", "IGL", "IGH"), + v_call = c("IGHV1", "IGLV1", "IGHV2"), + j_call = c("IGHJ1", "IGLJ1", "IGHJ2"), + junction_aa = c("CARW", "CAKW", "CARG"), + umi_count = c(10L, 10L, 10L) + ) + ) + + actual <- agg_receptors( + dataset = dataset, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = "IGH" + ), + barcode_col = "cell_id", + locus_col = "chain_locus", + umi_col = "umi_count" + ) |> + dplyr::collect() + + expect_true("locus" %in% names(actual)) + expect_false("chain_locus" %in% names(actual)) + expect_setequal(actual$locus, "IGH") + expect_setequal(actual$cell_id, c("cell_1", "cell_3")) +}) + +test_that("agg_receptors rejects custom and canonical locus columns together", { + dataset <- duckplyr::as_duckdb_tibble( + tibble::tibble( + chain_locus = "IGH", + locus = "IGL", + junction_aa = "CARW" + ) + ) + + expect_error( + agg_receptors( + dataset = dataset, + schema = make_receptor_schema( + features = "junction_aa", + chains = "IGH" + ), + locus_col = "chain_locus" + ), + "both the custom locus column.*chain_locus.*canonical locus column.*locus" + ) +}) + +test_that("distance filtering treats quoted patterns as literal values", { + annotations <- duckplyr::as_duckdb_tibble( + tibble::tibble( + imd_receptor_id = c(1L, 2L), + imd_barcode = c("bc_1", "bc_2"), + imd_chain_id = c(1L, 2L), + imd_n_chains = c(1L, 1L), + cdr3_aa = c("CA'RW", "CARRW") + ) + ) + idata <- ImmunData$new(schema = "cdr3_aa", annotations = annotations) + + for (method in c("lev", "hamm")) { + annotated <- mutate_immundata( + idata, + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = "CA'RW", + method = method + ) + )$annotations |> + dplyr::collect() |> + dplyr::arrange(imd_receptor_id) + + distance_col <- paste0("imd_sim_", method, "_1") + expect_equal(annotated[[distance_col]], c(0L, 1L), info = method) + + filtered <- filter_immundata( + idata, + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = "CA'RW", + method = method, + max_dist = 0L + ), + keep_repertoires = FALSE + )$receptors |> + dplyr::collect() + + expect_equal(filtered$cdr3_aa, "CA'RW", info = method) + } +}) diff --git a/tests/testthat/test-compute-collect-immundata.R b/tests/testthat/test-compute-collect-immundata.R new file mode 100644 index 0000000..d8281f7 --- /dev/null +++ b/tests/testthat/test-compute-collect-immundata.R @@ -0,0 +1,104 @@ +make_compute_state_test_idata <- function() { + ImmunData$new( + schema = "cdr3_aa", + annotations = make_basic_test_annotations() + ) +} + +test_that("compute() preserves repertoire, strata, and provenance state", { + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) |> + agg_strata("Response") + + reps_before <- idata$repertoires + strata_before <- idata$strata + prov_before <- get_provenance(idata) + out <- compute(idata) + + checkmate::expect_r6(out, classes = "ImmunData") + + in_annotations <- idata$annotations |> collect() + out_annotations <- out$annotations |> collect() + + expect_equal(nrow(out_annotations), nrow(in_annotations)) + expect_equal(colnames(out_annotations), colnames(in_annotations)) + expect_equal(out$repertoires, reps_before) + expect_equal(out$strata, strata_before) + expect_equal(out$schema_repertoire, idata$schema_repertoire) + expect_equal(out$schema_strata, idata$schema_strata) + expect_equal( + get_provenance(out)[sort(names(get_provenance(out)))], + prov_before[sort(names(prov_before))] + ) +}) + +test_that("compute() supports repertoire-free and strata-free state", { + annotations_only <- make_compute_state_test_idata() + annotations_only_out <- compute(annotations_only) + + expect_null(annotations_only_out$repertoires) + expect_null(annotations_only_out$strata) + expect_null(annotations_only_out$schema_repertoire) + expect_null(annotations_only_out$schema_strata) + + repertoires_only <- annotations_only |> + agg_repertoires("sample_id") + repertoires_only_out <- compute(repertoires_only) + + expect_equal(repertoires_only_out$repertoires, repertoires_only$repertoires) + expect_equal( + repertoires_only_out$schema_repertoire, + repertoires_only$schema_repertoire + ) + expect_null(repertoires_only_out$strata) + expect_null(repertoires_only_out$schema_strata) +}) + +test_that("collect() on ImmunData returns a tibble without factor columns", { + idata <- get_test_idata() + + annotations <- idata$annotations |> collect() + annotations$test_factor <- factor("x") + + idata_with_factor <- ImmunData$new( + schema = idata$schema_receptor, + annotations = annotations + ) + + out <- collect(idata_with_factor) + + expect_s3_class(out, "tbl_df") + expect_false(any(vapply(out, is.factor, logical(1)))) + expect_true(is.character(out$test_factor)) +}) + +test_that("colnames() on ImmunData returns annotation column names", { + idata <- get_test_idata() + + expect_equal(colnames(idata), colnames(idata$annotations)) +}) + +test_that("renaming via names/dimnames/colnames is blocked for ImmunData", { + idata <- get_test_idata() + + expect_error( + { + colnames(idata) <- paste0("x", seq_along(colnames(idata))) + }, + "not allowed" + ) + + expect_error( + { + dimnames(idata) <- list(NULL, paste0("y", seq_along(colnames(idata)))) + }, + "not allowed" + ) + + expect_error( + { + names(idata) <- letters[seq_along(names(idata))] + }, + "not allowed" + ) +}) diff --git a/tests/testthat/test-core-immundata.R b/tests/testthat/test-core-immundata.R new file mode 100644 index 0000000..1684e6a --- /dev/null +++ b/tests/testthat/test-core-immundata.R @@ -0,0 +1,47 @@ +test_that("ImmunData validates its annotations input", { + valid_annotations <- duckplyr::as_duckdb_tibble( + tibble::tibble( + imd_receptor_id = 1L, + imd_barcode = "bc1", + imd_chain_id = 1L, + imd_n_chains = 1L, + cdr3_aa = "AAA" + ) + ) + + expect_s3_class( + ImmunData$new(schema = "cdr3_aa", annotations = valid_annotations), + "ImmunData" + ) + + expect_error( + ImmunData$new(schema = "cdr3_aa", annotations = list(cdr3_aa = "AAA")), + "data[.]frame" + ) +}) + +test_that("receptor schema validators accept only valid schema structures", { + valid_schema <- make_receptor_schema( + features = c("junction_aa", "v_call"), + chains = c("TRA", "TRB") + ) + + expect_true(assert_receptor_schema("junction_aa")) + expect_true(assert_receptor_schema(valid_schema)) + expect_true(test_receptor_schema("junction_aa")) + expect_true(test_receptor_schema(valid_schema)) + + expect_false(test_receptor_schema(list(features = 1, chains = 2))) + expect_false(test_receptor_schema(list(features = "junction_aa"))) + expect_error(assert_receptor_schema(list(features = 1, chains = 2))) +}) + +test_that("snapshot file constants contain only files that are written", { + expect_identical( + imd_files(), + list( + metadata = "metadata.json", + annotations = "annotations.parquet" + ) + ) +}) diff --git a/tests/testthat/test-downsample.R b/tests/testthat/test-downsample.R new file mode 100644 index 0000000..d4f3c07 --- /dev/null +++ b/tests/testthat/test-downsample.R @@ -0,0 +1,517 @@ +test_that("draw_weighted_counts respects requested and available counts", { + set.seed(100) + + weights <- c(2, 3, 5) + sampled <- draw_weighted_counts(weights, size = 6) + + expect_length(sampled, length(weights)) + expect_equal(sum(sampled), 6) + expect_true(all(sampled >= 0)) + expect_true(all(sampled <= weights)) + expect_equal(draw_weighted_counts(weights, size = 0), integer(3)) + expect_equal(draw_weighted_counts(weights, size = 10), weights) +}) + +test_that("downsample_immundata downsamples single-cell repertoires and is deterministic with seed", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + test_data <- make_single_cell_downsample_test_data() + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = "IGH" + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + idata <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "downsample single-cell deterministic" + ) + + ds1 <- downsample_immundata(idata, n = 2, seed = 100) + ds2 <- downsample_immundata(idata, n = 2, seed = 100) + + expect_identical(get_provenance(ds1), get_provenance(idata)) + + reps <- ds1$repertoires + expect_true(all(reps$n_barcodes == 2)) + + sampled1 <- ds1$annotations |> + select(sample_id, imd_barcode) |> + distinct() |> + arrange(sample_id, imd_barcode) |> + collect() + + sampled_counts <- sampled1 |> + summarise(.by = sample_id, n_barcodes = n()) |> + arrange(sample_id) + + sampled2 <- ds2$annotations |> + select(sample_id, imd_barcode) |> + distinct() |> + arrange(sample_id, imd_barcode) |> + collect() + + expect_equal(sampled_counts$n_barcodes, c(2L, 2L)) + expect_true(all(sampled1$imd_barcode %in% test_data$cell_id)) + expect_equal(sampled1, sampled2) +}) + +test_that("downsample_immundata downsampled bulk repertoires by count", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + test_data <- make_bulk_count_test_data() + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = c("v_call", "j_call", "junction_aa"), + count_col = "clone_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + idata <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "downsample bulk count mode" + ) + + ds <- downsample_immundata(idata, n = 5, seed = 42) + reps <- ds$repertoires + ann <- ds$annotations |> + select(sample_id, v_call, j_call, junction_aa, imd_n_chains, imd_count, imd_proportion) |> + arrange(sample_id, v_call) |> + collect() + + count_totals <- ann |> + summarise(.by = sample_id, n_barcodes = sum(imd_n_chains)) |> + arrange(sample_id) + + original_counts <- test_data |> + rename(imd_n_chains_before = clone_count) + + ann_with_original <- ann |> + left_join(original_counts, by = c("sample_id", "v_call", "j_call", "junction_aa")) + + expect_equal(sort(reps$n_barcodes), c(5, 5)) + expect_equal(count_totals$n_barcodes, c(5L, 5L)) + expect_true(all(ann$imd_count == ann$imd_n_chains)) + expect_true(all(ann$imd_proportion > 0)) + expect_true(all(ann_with_original$imd_n_chains <= ann_with_original$imd_n_chains_before)) +}) + +test_that("downsample_immundata keeps paired chains intact", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + test_data <- data.frame( + cell_id = c( + "a1", "a1", "a2", "a2", "a3", "a3", + "b1", "b1", "b2", "b2", "b3", "b3" + ), + sample_id = c( + "S1", "S1", "S1", "S1", "S1", "S1", + "S2", "S2", "S2", "S2", "S2", "S2" + ), + v_call = c("IGHV1", "IGLV1", "IGHV2", "IGLV2", "IGHV3", "IGLV3", "IGHV1", "IGLV1", "IGHV2", "IGLV2", "IGHV3", "IGLV3"), + j_call = c("IGHJ1", "IGLJ1", "IGHJ2", "IGLJ2", "IGHJ3", "IGLJ3", "IGHJ1", "IGLJ1", "IGHJ2", "IGLJ2", "IGHJ3", "IGLJ3"), + junction_aa = c("CA1", "CL1", "CA2", "CL2", "CA3", "CL3", "CB1", "CK1", "CB2", "CK2", "CB3", "CK3"), + locus = c("IGH", "IGL", "IGH", "IGL", "IGH", "IGL", "IGH", "IGL", "IGH", "IGL", "IGH", "IGL"), + umi_count = rep(100, 12) + ) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGL") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + idata <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "downsample paired-chain integrity" + ) + + ds <- downsample_immundata(idata, n = 2, seed = 7) + reps <- ds$repertoires + expect_true(all(reps$n_barcodes == 2)) + + chain_stats <- ds$annotations |> + collect() |> + summarise( + .by = c(sample_id, imd_barcode), + n_loci = n_distinct(locus), + n_rows = n() + ) + + barcode_counts <- chain_stats |> + summarise(.by = sample_id, n_barcodes = n()) |> + arrange(sample_id) + + expect_equal(barcode_counts$n_barcodes, c(2L, 2L)) + expect_true(all(chain_stats$n_loci == 2)) + expect_true(all(chain_stats$n_rows == 2)) +}) + +test_that("downsample_immundata works on IG test data with proportion n", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + idata <- read_repertoires( + path = test_ig_data(), + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGK|IGL") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "duplicate_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) |> + mutate_immundata(cohort = "all") + idata <- agg_repertoires_with_integrity( + idata, + schema = "cohort", + context = "downsample IG proportion" + ) + + n_before <- idata$repertoires |> + pull(n_barcodes) + + ds <- downsample_immundata(idata, n = 0.5, seed = 123) + n_after <- ds$repertoires |> + pull(n_barcodes) + + expect_equal(n_after, floor(n_before * 0.5)) + + annotation_units <- ds$annotations |> + select(cohort, imd_barcode) |> + distinct() |> + collect() |> + summarise(.by = cohort, n_barcodes = n()) + + expect_equal(annotation_units$n_barcodes, n_after) + + chain_stats <- ds$annotations |> + collect() |> + summarise( + .by = imd_barcode, + n_loci = n_distinct(locus) + ) + expect_true(all(chain_stats$n_loci == 2)) +}) + +test_that("downsample_immundata retains one sampling unit per repertoire for n = 1", { + idata <- get_test_idata_tsv_no_manifest() + ds <- downsample_immundata(idata, n = 1, seed = 321) + + expect_true(all(ds$repertoires$n_barcodes == 1)) + expect_true(all(ds$repertoires$n_receptors == 1)) +}) + +test_that("downsample_immundata validates n and handles no-repertoire fallback", { + idata <- get_test_idata_tsv_no_manifest(repertoire_schema = NULL) + + expect_error( + downsample_immundata(idata, n = 2.5), + "integer count" + ) + + n_before <- idata$annotations |> + distinct(imd_barcode) |> + collect() |> + nrow() + + ds <- downsample_immundata(idata, n = 0.1, seed = 1) + + n_after <- ds$annotations |> + distinct(imd_barcode) |> + collect() |> + nrow() + + expect_equal(n_after, floor(n_before * 0.1)) + expect_null(ds$repertoires) +}) + +test_that("downsample_immundata warns and keeps repertoire unchanged when n exceeds available units", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + test_data <- make_single_cell_downsample_test_data() + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema(features = c("v_call", "j_call", "junction_aa"), chains = "IGH"), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + idata <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "downsample n exceeds units" + ) + + reps_before <- idata$repertoires |> + arrange(sample_id) + + ds <- NULL + expect_warning( + ds <- downsample_immundata(idata, n = 10, seed = 42), + "returned unchanged" + ) + + reps_after <- ds$repertoires |> + arrange(sample_id) + + annotation_counts <- ds$annotations |> + select(sample_id, imd_barcode) |> + distinct() |> + collect() |> + summarise(.by = sample_id, n_barcodes = n()) |> + arrange(sample_id) + + expect_equal(reps_after$n_barcodes, reps_before$n_barcodes) + expect_equal(annotation_counts$n_barcodes, reps_before$n_barcodes) +}) + +test_that("downsample_immundata supports count-mode proportion downsampling", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + test_data <- make_bulk_count_test_data() + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = c("v_call", "j_call", "junction_aa"), + count_col = "clone_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + idata <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "downsample count proportion" + ) + + ds <- downsample_immundata(idata, n = 0.5, seed = 101) + reps <- ds$repertoires + ann <- ds$annotations |> + select(sample_id, imd_n_chains, imd_count) |> + collect() + + count_totals <- ann |> + summarise(.by = sample_id, n_barcodes = sum(imd_n_chains)) |> + arrange(sample_id) + + expect_equal(sort(reps$n_barcodes), c(7, 7)) + expect_equal(count_totals$n_barcodes, c(7L, 7L)) + expect_true(all(ann$imd_count == ann$imd_n_chains)) +}) + +test_that("downsample_immundata is deterministic in count mode with seed", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + test_data <- make_bulk_count_test_data() + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = c("v_call", "j_call", "junction_aa"), + count_col = "clone_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + idata <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "downsample count deterministic" + ) + + ds1 <- downsample_immundata(idata, n = 5, seed = 222) + ds2 <- downsample_immundata(idata, n = 5, seed = 222) + + ann1 <- ds1$annotations |> + select(sample_id, imd_barcode, imd_n_chains) |> + arrange(sample_id, imd_barcode) |> + collect() + ann2 <- ds2$annotations |> + select(sample_id, imd_barcode, imd_n_chains) |> + arrange(sample_id, imd_barcode) |> + collect() + + expect_equal(ann1, ann2) +}) + +test_that("downsample_immundata errors when proportion results in zero target", { + idata <- get_test_idata_tsv_no_manifest() + + expect_error( + downsample_immundata(idata, n = 0.0001, seed = 1), + "Increase `n`" + ) +}) + +test_that("downsample_immundata produces consistent imd_count and imd_proportion invariants", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + test_data <- data.frame( + cell_id = c("c1", "c2", "c3", "c4", "c5", "c6"), + sample_id = c("S1", "S1", "S1", "S2", "S2", "S2"), + v_call = c("IGHV1", "IGHV1", "IGHV2", "IGHV1", "IGHV2", "IGHV2"), + j_call = c("IGHJ1", "IGHJ1", "IGHJ2", "IGHJ1", "IGHJ2", "IGHJ2"), + junction_aa = c("A1", "A1", "A2", "B1", "B2", "B2"), + locus = "IGH", + umi_count = c(10, 11, 12, 13, 14, 15) + ) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema(features = c("v_call", "j_call", "junction_aa"), chains = "IGH"), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + idata <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "downsample count/proportion invariants" + ) + + ds <- downsample_immundata(idata, n = 2, seed = 33) + + reps <- ds$repertoires + ann <- ds$annotations |> collect() + + receptor_stats <- ann |> + select(imd_repertoire_id, imd_receptor_id, imd_count, imd_proportion) |> + distinct() |> + summarise( + .by = imd_repertoire_id, + sum_count = sum(imd_count), + sum_prop = sum(imd_proportion) + ) |> + arrange(imd_repertoire_id) + + reps <- reps |> arrange(imd_repertoire_id) + + expect_equal(receptor_stats$sum_count, reps$n_barcodes) + expect_true(all(abs(receptor_stats$sum_prop - 1) < 1e-8)) +}) + +test_that("downsample_immundata uses schema_repertoire rather than stale repertoire ids", { + annotations <- duckplyr::as_duckdb_tibble(data.frame( + sample_id = rep(c("S1", "S2"), each = 3), + junction_aa = paste0("C", seq_len(6)), + imd_barcode = seq_len(6), + imd_chain_id = seq_len(6), + imd_receptor_id = seq_len(6), + imd_n_chains = rep(1L, 6) + )) + + aggregated <- ImmunData$new(schema = "junction_aa", annotations = annotations) |> + agg_repertoires("sample_id") + detached <- ImmunData$new( + schema = aggregated$schema_receptor, + annotations = aggregated$annotations + ) + + ds <- downsample_immundata(detached, n = 2, seed = 1) + sampled_barcodes <- ds$annotations |> + distinct(imd_barcode) |> + collect() + + expect_null(detached$schema_repertoire) + expect_null(ds$repertoires) + expect_equal(nrow(sampled_barcodes), 2) +}) + +test_that("downsample_immundata rebuilds strata and retains strata labels", { + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) |> + agg_strata("Response") + idata <- rename_strata( + idata, + names = stats::setNames( + paste0("Response_", idata$strata[[strata_col]]), + as.character(idata$strata[[strata_col]]) + ) + ) + + ds <- downsample_immundata(idata, n = 1, seed = 99) + + expect_equal(ds$schema_strata, "Response") + expect_true(strata_col %in% names(ds$annotations)) + expect_true(all(c(strata_col, strata_name_col) %in% names(ds$repertoires))) + expect_equal( + sort(ds$strata[[strata_name_col]]), + sort(idata$strata[[strata_name_col]]) + ) +}) diff --git a/tests/testthat/test-filter-barcodes.R b/tests/testthat/test-filter-barcodes.R index 5bf17ea..dcacffe 100644 --- a/tests/testthat/test-filter-barcodes.R +++ b/tests/testthat/test-filter-barcodes.R @@ -1,7 +1,7 @@ test_that("filter_barcodes() filters ImmunData by a set of cell barcodes", { outdir <- tempdir() - idata <- get_test_idata_tsv_no_metadata() + idata <- get_test_idata_tsv_no_manifest() barcode_col <- imd_schema_sym("barcode") diff --git a/tests/testthat/test-filter-immundata-exact.R b/tests/testthat/test-filter-immundata-exact.R index 14e2c36..3549fd2 100644 --- a/tests/testthat/test-filter-immundata-exact.R +++ b/tests/testthat/test-filter-immundata-exact.R @@ -1,5 +1,5 @@ test_that("exact matching with single and multiple patterns", { - idata <- get_test_idata_tsv_no_metadata() + idata <- get_test_idata_tsv_no_manifest() all_receptors <- idata$receptors %>% collect() # Single pattern @@ -35,3 +35,27 @@ test_that("exact matching with single and multiple patterns", { gold2 %>% arrange(cdr3_aa) ) }) + +test_that("exact matching preserves preceding annotation filters", { + idata <- make_paired_filter_test_idata() + + out <- filter_immundata( + idata, + sample_id == "S1", + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = "AAA", + method = "exact" + ), + keep_repertoires = FALSE + ) + + actual <- out$annotations |> + dplyr::collect() |> + dplyr::arrange(imd_chain_id) + + expect_equal(nrow(actual), 2L) + expect_setequal(actual$sample_id, "S1") + expect_setequal(actual$locus, c("IGH", "IGL")) + expect_setequal(actual$cdr3_aa, c("AAA", "CCC")) +}) diff --git a/tests/testthat/test-filter-immundata-hamm.R b/tests/testthat/test-filter-immundata-hamm.R index 0b1f869..87f11ac 100644 --- a/tests/testthat/test-filter-immundata-hamm.R +++ b/tests/testthat/test-filter-immundata-hamm.R @@ -1,6 +1,6 @@ # 4. Hamming fuzzy matching test_that("Hamming fuzzy matching returns correct results", { - idata <- get_test_idata_tsv_no_metadata() + idata <- get_test_idata_tsv_no_manifest() all_receptors <- idata$receptors %>% collect() orig <- all_receptors$cdr3_aa[1] @@ -29,3 +29,31 @@ test_that("Hamming fuzzy matching returns correct results", { gold %>% arrange(cdr3_aa) ) }) + +test_that("Hamming matching preserves paired chains within preceding filters", { + idata <- make_paired_filter_test_idata() + + actual <- filter_immundata( + idata, + sample_id == "S1", + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = "AAT", + method = "hamm", + max_dist = 1L + ), + keep_repertoires = FALSE + )$annotations |> + dplyr::collect() |> + dplyr::arrange(imd_chain_id) |> + dplyr::select(imd_chain_id, locus, cdr3_aa, sample_id) + + expected <- tibble::tibble( + imd_chain_id = 1:2, + locus = c("IGH", "IGL"), + cdr3_aa = c("AAA", "CCC"), + sample_id = c("S1", "S1") + ) + + expect_equal(actual, expected) +}) diff --git a/tests/testthat/test-filter-immundata-lev.R b/tests/testthat/test-filter-immundata-lev.R index ef1b67c..611edd5 100644 --- a/tests/testthat/test-filter-immundata-lev.R +++ b/tests/testthat/test-filter-immundata-lev.R @@ -1,6 +1,6 @@ # 3. Levenshtein fuzzy matching testthat::test_that("Levenshtein fuzzy matching returns correct results", { - idata <- get_test_idata_tsv_no_metadata() + idata <- get_test_idata_tsv_no_manifest() all_receptors <- idata$receptors |> collect() pat <- substr(all_receptors$cdr3_aa[1:3], 1, nchar(all_receptors$cdr3_aa[1:3]) - 1) @@ -24,9 +24,59 @@ testthat::test_that("Levenshtein fuzzy matching returns correct results", { ) }) +test_that("Levenshtein filtering keeps edits at sequence ends", { + sequences <- tibble::tibble( + cdr3_aa = c("AAAAA", "BAAAA", "AAAAB", "AAAA", "AAAAAA", "BBAAA") + ) |> + duckplyr::as_duckdb_tibble() + + actual <- annotate_tbl_distance( + sequences, + query_col = "cdr3_aa", + patterns = "AAAAA", + method = "lev", + max_dist = 1, + name_type = "index" + ) |> + collect() |> + arrange(cdr3_aa) + + expect_equal( + actual$cdr3_aa, + sort(c("AAAAA", "BAAAA", "AAAAB", "AAAA", "AAAAAA")) + ) + expect_equal(actual$imd_sim_lev_1, rep(1, 5) - (actual$cdr3_aa == "AAAAA")) +}) + +test_that("distance materialization is independent of the R random seed", { + sequences <- tibble::tibble(cdr3_aa = c("AAAAA", "AAAAB")) |> + duckplyr::as_duckdb_tibble() + + set.seed(1) + first <- annotate_tbl_distance( + sequences, + query_col = "cdr3_aa", + patterns = "AAAAA", + method = "lev" + ) + + set.seed(1) + second <- annotate_tbl_distance( + sequences, + query_col = "cdr3_aa", + patterns = "AAAAA", + method = "lev" + ) + + expect_equal( + collect(first) |> arrange(cdr3_aa), + collect(second) |> arrange(cdr3_aa) + ) +}) + # 6. Combined pre-filter and fuzzy matching test_that("combined pre-filter and fuzzy matching works correctly", { - idata <- get_test_idata_tsv_no_metadata() + idata <- get_test_idata_tsv_no_manifest() all_receptors <- idata$receptors |> collect() vc <- all_receptors$v_call[5] @@ -51,3 +101,31 @@ test_that("combined pre-filter and fuzzy matching works correctly", { gold |> arrange(cdr3_aa) ) }) + +test_that("Levenshtein matching preserves paired chains within preceding filters", { + idata <- make_paired_filter_test_idata() + + actual <- filter_immundata( + idata, + sample_id == "S1", + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = "AAT", + method = "lev", + max_dist = 1L + ), + keep_repertoires = FALSE + )$annotations |> + dplyr::collect() |> + dplyr::arrange(imd_chain_id) |> + dplyr::select(imd_chain_id, locus, cdr3_aa, sample_id) + + expected <- tibble::tibble( + imd_chain_id = 1:2, + locus = c("IGH", "IGL"), + cdr3_aa = c("AAA", "CCC"), + sample_id = c("S1", "S1") + ) + + expect_equal(actual, expected) +}) diff --git a/tests/testthat/test-filter-immundata-regex.R b/tests/testthat/test-filter-immundata-regex.R index fb29937..127de9e 100644 --- a/tests/testthat/test-filter-immundata-regex.R +++ b/tests/testthat/test-filter-immundata-regex.R @@ -1,5 +1,5 @@ testthat::test_that("Regex matching returns correct results", { - idata <- get_test_idata_tsv_no_metadata() + idata <- get_test_idata_tsv_no_manifest() all_receptors <- idata$receptors %>% collect() regex <- paste0("^", substr(all_receptors$cdr3_aa[1], 1, 4)) diff --git a/tests/testthat/test-filter-immundata.R b/tests/testthat/test-filter-immundata.R index ad9a020..f0fede2 100644 --- a/tests/testthat/test-filter-immundata.R +++ b/tests/testthat/test-filter-immundata.R @@ -1,5 +1,5 @@ test_that("filter() filters ImmunData by receptor-level conditions", { - idata <- get_test_idata_tsv_with_metadata() + idata <- get_test_idata() # Sanity check checkmate::expect_r6(idata, "ImmunData") @@ -17,7 +17,7 @@ test_that("filter() filters ImmunData by receptor-level conditions", { }) test_that("filter() filters ImmunData by annotation-level conditions (locus)", { - idata <- get_test_idata_tsv_no_metadata() + idata <- get_test_idata_tsv_no_manifest() # Let's say the annotation table has a column "locus" (common in TCR/BCR data) # We'll filter to "TRB". Adjust to an actual locus present in your data @@ -33,3 +33,99 @@ test_that("filter() filters ImmunData by annotation-level conditions (locus)", { # The receptor table should be smaller or the same size, never bigger expect_lte(filtered$receptors |> collect() |> nrow(), idata$receptors |> collect() |> nrow()) }) + +test_that("filters discard repertoire state when not preserving repertoires", { + annotations <- duckplyr::as_duckdb_tibble( + tibble::tibble( + imd_receptor_id = 1:4, + imd_barcode = paste0("bc", 1:4), + imd_chain_id = 1:4, + imd_n_chains = 1L, + cdr3_aa = c("AAA", "AAT", "AAA", "BBB"), + sample_id = c("S1", "S1", "S2", "S2") + ) + ) + idata <- ImmunData$new(schema = "cdr3_aa", annotations = annotations) |> + agg_repertoires("sample_id") |> + agg_strata("sample_id") + + repertoire_state_cols <- c( + imd_schema("repertoire"), + imd_schema("strata"), + imd_schema("strata_name"), + imd_schema("count"), + imd_schema("proportion"), + imd_schema("n_receptors"), + imd_schema("n_barcodes"), + imd_schema("n_repertoires") + ) + expect_true(all(c( + imd_schema("repertoire"), + imd_schema("strata"), + imd_schema("count"), + imd_schema("proportion"), + imd_schema("n_repertoires") + ) %in% names(idata$annotations))) + + filtered <- list( + filter_immundata(idata, sample_id == "S1", keep_repertoires = FALSE), + filter_barcodes(idata, "bc1", keep_repertoires = FALSE), + filter_receptors(idata, 1L, keep_repertoires = FALSE) + ) + + for (out in filtered) { + expect_null(out$repertoires) + expect_null(out$schema_repertoire) + expect_null(out$schema_strata) + expect_false(any(repertoire_state_cols %in% names(out$annotations))) + } +}) + +test_that("filters rebuild strata and retain strata labels", { + annotations <- duckplyr::as_duckdb_tibble( + tibble::tibble( + imd_receptor_id = 1:6, + imd_barcode = paste0("bc", 1:6), + imd_chain_id = 1:6, + imd_n_chains = 1L, + cdr3_aa = c("AAA", "AAT", "ABB", "BBB", "BBC", "BCC"), + sample_id = c("S1", "S1", "S2", "S2", "S3", "S3"), + response = c("R", "R", "NR", "NR", "R", "R") + ) + ) + idata <- ImmunData$new(schema = "cdr3_aa", annotations = annotations) |> + agg_repertoires(c("sample_id", "response")) |> + agg_strata("response") + + strata_col <- imd_schema("strata") + strata_name_col <- imd_schema("strata_name") + strata_names <- ifelse( + idata$strata$response == "R", + "Responder", + "Non-responder" + ) + idata <- rename_strata( + idata, + stats::setNames(strata_names, as.character(idata$strata[[strata_col]])) + ) + + filtered <- list( + filter_immundata(idata, sample_id != "S3"), + filter_barcodes(idata, paste0("bc", 1:4)), + filter_receptors(idata, 1:4) + ) + + for (out in filtered) { + expect_equal(out$schema_strata, "response") + expect_true(strata_col %in% names(out$annotations)) + expect_true(all(c(strata_col, strata_name_col) %in% names(out$repertoires))) + + rebuilt_strata <- out$strata |> + arrange(response) + expect_equal(rebuilt_strata$response, c("NR", "R")) + expect_equal( + rebuilt_strata[[strata_name_col]], + c("Non-responder", "Responder") + ) + } +}) diff --git a/tests/testthat/test-filter-receptors.R b/tests/testthat/test-filter-receptors.R index d031b3f..a7aa0f2 100644 --- a/tests/testthat/test-filter-receptors.R +++ b/tests/testthat/test-filter-receptors.R @@ -1,6 +1,6 @@ test_that("filter_receptors() filters ImmunData by a set of receptor identifiers", { - idata <- get_test_idata_tsv_no_metadata() + idata <- get_test_idata_tsv_no_manifest() all_receptors <- idata$annotations %>% distinct(imd_receptor_id) %>% diff --git a/tests/testthat/test-io-immundata-write.R b/tests/testthat/test-io-immundata-write.R deleted file mode 100644 index e69de29..0000000 diff --git a/tests/testthat/test-io-immundata.R b/tests/testthat/test-io-immundata.R new file mode 100644 index 0000000..1533d2f --- /dev/null +++ b/tests/testthat/test-io-immundata.R @@ -0,0 +1,969 @@ +test_that("read_immundata() upgrades legacy v1 metadata on the fly", { + legacy_path <- create_test_output_dir("legacy_v1_") + on.exit(cleanup_output_dir(legacy_path), add = TRUE) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = legacy_path, + preprocess = NULL, + postprocess = NULL + ) + + legacy_metadata_v1 <- list( + version = as.character(packageVersion("immundata")), + receptor_schema = list( + features = c("cdr3_aa", "v_call"), + chains = "TCRB" + ), + repertoire_schema = "imd_filename" + ) + jsonlite::write_json( + legacy_metadata_v1, + path = file.path(legacy_path, "metadata.json"), + auto_unbox = TRUE, + null = "null" + ) + + expect_warning( + idata <- read_immundata(legacy_path, verbose = FALSE), + "legacy v1 metadata" + ) + + checkmate::expect_r6(idata, classes = "ImmunData") + expect_true(length(names(idata$annotations)) > 0) + expect_equal(idata$schema_repertoire, "imd_filename") + expect_false(is.null(idata$repertoires)) + + prov <- get_provenance(idata) + expect_equal(prov$current_path, normalizePath(legacy_path, mustWork = FALSE)) + expect_null(prov$snapshot_id) + + written <- write_immundata(idata, output_folder = legacy_path) + expect_true(is.character(get_provenance(written)$snapshot_id)) +}) + +test_that("ImmunData$provenance is read-only and matches helper output", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + idata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + expect_identical(idata$provenance, get_provenance(idata)) + + expect_error( + idata$provenance <- list(), + "read-only" + ) +}) + +test_that("normalize_provenance applies canonical overrides and derives all paths", { + old_home <- create_test_output_dir("old_provenance_home_") + new_home <- create_test_output_dir("new_provenance_home_") + dir.create(old_home, recursive = TRUE) + dir.create(new_home, recursive = TRUE) + on.exit(cleanup_output_dir(old_home), add = TRUE) + on.exit(cleanup_output_dir(new_home), add = TRUE) + + current_path <- file.path( + new_home, + "snapshots", + "baseline", + "v003" + ) + dir.create(current_path, recursive = TRUE) + canonical_lineage <- list(list( + event = "snapshot", + snapshot_id = "canonical-id" + )) + provenance <- normalize_provenance( + provenance = list( + home_path = old_home, + current_path = old_home, + snapshot_root = file.path(old_home, "stale-snapshots"), + artifacts_root = file.path(old_home, "stale-artifacts"), + artifacts_path = file.path(old_home, "stale-artifact-path"), + snapshot_id = "stale-id", + lineage = list(list(event = "stale")) + ), + home_path = new_home, + current_path = current_path, + snapshot_id = "canonical-id", + lineage = canonical_lineage + ) + + normalized_home <- normalizePath(new_home, mustWork = TRUE) + expect_equal(provenance$home_path, normalized_home) + expect_equal( + provenance$current_path, + normalizePath( + file.path(normalized_home, "snapshots", "baseline", "v003"), + mustWork = FALSE + ) + ) + expect_equal( + provenance$snapshot_root, + normalizePath( + file.path(normalized_home, "snapshots"), + mustWork = FALSE + ) + ) + expect_equal( + provenance$artifacts_root, + normalizePath( + file.path(normalized_home, "artifacts"), + mustWork = FALSE + ) + ) + expect_equal( + provenance$artifacts_path, + normalizePath( + file.path(normalized_home, "artifacts", "baseline", "v003"), + mustWork = FALSE + ) + ) + expect_equal(provenance$snapshot_id, "canonical-id") + expect_identical(provenance$lineage, canonical_lineage) + + metadata_provenance <- provenance_paths_for_metadata(provenance) + expect_named( + metadata_provenance, + c( + "home_path", "current_path", "snapshot_root", + "artifacts_root", "artifacts_path" + ) + ) + expect_false("snapshot_id" %in% names(metadata_provenance)) + expect_false("lineage" %in% names(metadata_provenance)) +}) + +test_that("root ingestion exposes a shared artifacts root and root artifact path", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + idata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + normalized_output <- normalizePath(output_dir, mustWork = TRUE) + expected_root <- normalizePath( + file.path(normalized_output, "artifacts"), + mustWork = FALSE + ) + expected_path <- normalizePath( + file.path(expected_root, "root"), + mustWork = FALSE + ) + provenance <- idata$provenance + + expect_equal(provenance$artifacts_root, expected_root) + expect_equal(provenance$artifacts_path, expected_path) + expect_false(dir.exists(expected_path)) + + simulated_run <- file.path(provenance$artifacts_path, "distance", "run-001") + expect_true(dir.create(simulated_run, recursive = TRUE)) + expect_true(dir.exists(simulated_run)) + + metadata_json <- jsonlite::read_json( + file.path(output_dir, "metadata.json"), + simplifyVector = FALSE + ) + expect_equal(metadata_json$provenance$artifacts_root, expected_root) + expect_equal(metadata_json$provenance$artifacts_path, expected_path) + + reloaded <- read_immundata(output_dir, verbose = FALSE) + expect_equal(reloaded$provenance$artifacts_root, expected_root) + expect_equal(reloaded$provenance$artifacts_path, expected_path) + expect_true(dir.exists(file.path( + reloaded$provenance$artifacts_path, + "distance", + "run-001" + ))) +}) + +test_that("managed snapshots mirror tag and version beneath artifacts root", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + root_idata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + snapshot <- write_immundata( + root_idata, + output_folder = NULL, + tag = "baseline" + ) + + normalized_output <- normalizePath(output_dir, mustWork = TRUE) + expected_root <- normalizePath( + file.path(normalized_output, "artifacts"), + mustWork = FALSE + ) + expected_path <- normalizePath( + file.path(expected_root, "baseline", "v001"), + mustWork = FALSE + ) + + expect_equal(snapshot$provenance$artifacts_root, expected_root) + expect_equal(snapshot$provenance$artifacts_path, expected_path) + expect_false(dir.exists(expected_path)) + + simulated_run <- file.path( + snapshot$provenance$artifacts_path, + "distance", + "run-001" + ) + expect_true(dir.create(simulated_run, recursive = TRUE)) + expect_true(dir.exists(simulated_run)) + + snapshot_path <- file.path( + output_dir, + "snapshots", + "baseline", + "v001" + ) + metadata_json <- jsonlite::read_json( + file.path(snapshot_path, "metadata.json"), + simplifyVector = FALSE + ) + expect_equal(metadata_json$provenance$artifacts_root, expected_root) + expect_equal(metadata_json$provenance$artifacts_path, expected_path) + + reloaded <- read_immundata( + output_dir, + tag = "baseline", + version = 1, + verbose = FALSE + ) + expect_equal(reloaded$provenance$artifacts_root, expected_root) + expect_equal(reloaded$provenance$artifacts_path, expected_path) + expect_true(dir.exists(file.path( + reloaded$provenance$artifacts_path, + "distance", + "run-001" + ))) +}) + +test_that("read_repertoires() writes metadata with lineage array and provenance", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + metadata_path <- file.path(output_dir, "metadata.json") + metadata_json <- jsonlite::read_json(metadata_path, simplifyVector = FALSE) + + expect_equal(metadata_json$format_version, 2) + expect_true(is.character(metadata_json$snapshot_id)) + expect_equal(metadata_json$producer[["function"]], "read_repertoires") + expect_true(is.list(metadata_json$lineage)) + expect_length(metadata_json$lineage, 1) + expect_true(is.list(metadata_json$provenance)) + expect_false("snapshot_id" %in% names(metadata_json$provenance)) + expect_false("lineage" %in% names(metadata_json$provenance)) + + ingestion_event <- metadata_json$lineage[[1]] + expect_equal(ingestion_event$event, "ingestion") + expect_equal(ingestion_event$producer[["function"]], "read_repertoires") + expect_equal(ingestion_event$inputs$files, normalizePath(sample_file)) + expect_false(isTRUE(ingestion_event$inputs$manifest_joined)) + + normalized_out <- normalizePath(output_dir, mustWork = FALSE) + expect_equal(metadata_json$provenance$home_path, normalized_out) + expect_equal(metadata_json$provenance$current_path, normalized_out) + expect_equal( + normalizePath(metadata_json$provenance$snapshot_root, mustWork = FALSE), + normalizePath(file.path(normalized_out, "snapshots"), mustWork = FALSE) + ) + + loaded <- read_immundata(output_dir, verbose = FALSE) + loaded_provenance <- get_provenance(loaded) + expect_equal(loaded_provenance$snapshot_id, metadata_json$snapshot_id) + expect_length(loaded_provenance$lineage, length(metadata_json$lineage)) + expect_equal( + vapply(loaded_provenance$lineage, `[[`, character(1), "event"), + vapply(metadata_json$lineage, `[[`, character(1), "event") + ) + expect_equal( + vapply(loaded_provenance$lineage, `[[`, character(1), "snapshot_id"), + vapply(metadata_json$lineage, `[[`, character(1), "snapshot_id") + ) + + metadata_json$provenance$snapshot_id <- "stale-duplicated-id" + metadata_json$provenance$lineage <- list(list( + event = "stale-duplicated-event", + snapshot_id = "stale-duplicated-id" + )) + jsonlite::write_json( + metadata_json, + metadata_path, + auto_unbox = TRUE, + null = "null", + pretty = TRUE + ) + + loaded_duplicated_v2 <- read_immundata(output_dir, verbose = FALSE) + duplicated_v2_provenance <- get_provenance(loaded_duplicated_v2) + expect_equal(duplicated_v2_provenance$snapshot_id, metadata_json$snapshot_id) + expect_equal( + vapply(duplicated_v2_provenance$lineage, `[[`, character(1), "event"), + vapply(metadata_json$lineage, `[[`, character(1), "event") + ) +}) + +test_that("write_immundata() appends snapshot lineage event", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + idata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + metadata_path <- file.path(output_dir, "metadata.json") + before_json <- jsonlite::read_json(metadata_path, simplifyVector = FALSE) + expect_length(before_json$lineage, 1) + previous_snapshot_id <- before_json$snapshot_id + + write_immundata(idata, output_folder = output_dir) + after_json <- jsonlite::read_json(metadata_path, simplifyVector = FALSE) + + expect_equal(after_json$producer[["function"]], "write_immundata") + expect_length(after_json$lineage, 2) + expect_false(identical(after_json$snapshot_id, previous_snapshot_id)) + + snapshot_event <- after_json$lineage[[2]] + expect_equal(snapshot_event$event, "snapshot") + expect_equal(snapshot_event$producer[["function"]], "write_immundata") + + normalized_out <- normalizePath(output_dir, mustWork = FALSE) + expect_equal(snapshot_event$source_path, normalized_out) + expect_equal(snapshot_event$snapshot_path, normalized_out) +}) + +test_that("read_repertoires() writes manifest-derived files in ingestion lineage", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + + manifest_path <- system.file("extdata/tsv", "manifest.csv", package = "immundata") + manifest_df <- read_manifest(manifest_path) + + read_repertoires( + path = "", + schema = c("cdr3_aa", "v_call"), + manifest = manifest_df, + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + metadata_path <- file.path(output_dir, "metadata.json") + metadata_json <- jsonlite::read_json(metadata_path, simplifyVector = FALSE) + + expect_equal(metadata_json$producer[["function"]], "read_repertoires") + expect_length(metadata_json$lineage, 1) + + ingestion_event <- metadata_json$lineage[[1]] + expect_true(isTRUE(ingestion_event$inputs$manifest_joined)) + expect_equal( + unlist(ingestion_event$inputs$files, use.names = FALSE), + normalizePath(manifest_df$file) + ) + expect_equal(ingestion_event$args$manifest_file_col, "file") +}) + +test_that("write_immundata() auto-creates snapshot folders and increments versions", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + idata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + idata_v1 <- write_immundata(idata, output_folder = NULL, tag = "baseline") + idata_v2 <- write_immundata(idata_v1, output_folder = NULL, tag = "baseline") + + expect_true(dir.exists(file.path(output_dir, "snapshots", "baseline", "v001"))) + expect_true(dir.exists(file.path(output_dir, "snapshots", "baseline", "v002"))) + + prov_v2 <- get_provenance(idata_v2) + expect_equal( + prov_v2$current_path, + normalizePath(file.path(output_dir, "snapshots", "baseline", "v002"), mustWork = FALSE) + ) +}) + +test_that("snapshot tests use projectA/projectB tree in temporary snapshot root", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + rehome_dir <- layout$projectB + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + idata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + write_immundata(idata, output_folder = NULL, tag = "baseline") + write_immundata( + read_immundata(output_dir, tag = "baseline", version = 1), + output_folder = NULL, + tag = "baseline" + ) + write_immundata(idata, output_folder = NULL, tag = "treated") + write_immundata(idata, output_folder = rehome_dir, rehome = TRUE) + + expect_true(file.exists(file.path(output_dir, "annotations.parquet"))) + expect_true(file.exists(file.path(output_dir, "metadata.json"))) + expect_true(dir.exists(file.path(output_dir, "snapshots", "baseline", "v001"))) + expect_true(dir.exists(file.path(output_dir, "snapshots", "baseline", "v002"))) + expect_true(dir.exists(file.path(output_dir, "snapshots", "treated", "v001"))) + expect_true(file.exists(file.path(rehome_dir, "annotations.parquet"))) + expect_true(file.exists(file.path(rehome_dir, "metadata.json"))) +}) + +test_that("read_immundata() resolves tag latest and specific versions", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + idata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + write_immundata(idata, output_folder = NULL, tag = "baseline") + write_immundata(read_immundata(output_dir, tag = "baseline", version = 1), output_folder = NULL, tag = "baseline") + + latest <- read_immundata(output_dir, tag = "baseline") + latest_prov <- get_provenance(latest) + expect_equal( + latest_prov$current_path, + normalizePath(file.path(output_dir, "snapshots", "baseline", "v002"), mustWork = FALSE) + ) + + version1 <- read_immundata(output_dir, tag = "baseline", version = 1) + v1_prov <- get_provenance(version1) + expect_equal( + v1_prov$current_path, + normalizePath(file.path(output_dir, "snapshots", "baseline", "v001"), mustWork = FALSE) + ) +}) + +test_that("snapshot path resolution validates missing tags and versions", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + idata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + write_immundata(idata, output_folder = NULL, tag = "baseline") + + expect_error( + read_immundata(output_dir, tag = "ghost"), + "not found" + ) + + expect_error( + read_immundata(output_dir, tag = "baseline", version = 99), + "not found" + ) + + expect_error( + read_immundata(output_dir, version = 1), + "only.*tag" + ) + + expect_error( + read_immundata(file.path(output_dir, "snapshots", "baseline", "v001"), tag = "baseline"), + "already points" + ) + + expect_error( + read_immundata(output_dir, tag = "../bad"), + "must not include path separators" + ) + + expect_error( + read_immundata(output_dir, tag = "bad tag"), + "may only contain" + ) + + expect_error( + read_immundata(output_dir, tag = "root"), + "reserved for the original ingestion state" + ) + + expect_error( + read_immundata(output_dir, tag = "ROOT"), + "reserved for the original ingestion state" + ) +}) + +test_that("in-memory provenance reads are stable and snapshot IDs are created by writes", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + + idata <- get_test_idata_tsv_no_manifest() + idata_no_provenance <- ImmunData$new( + schema = idata$schema_receptor, + annotations = idata$annotations + ) + + first_provenance <- get_provenance(idata_no_provenance) + second_provenance <- get_provenance(idata_no_provenance) + expect_identical(first_provenance, second_provenance) + expect_null(first_provenance$snapshot_id) + expect_null(idata_no_provenance$.__enclos_env__$private$.provenance) + + expect_error( + write_immundata(idata_no_provenance, output_folder = NULL), + "Cannot infer snapshot home path" + ) + + written <- write_immundata(idata_no_provenance, output_folder = output_dir) + metadata_json <- jsonlite::read_json( + file.path(output_dir, "metadata.json"), + simplifyVector = FALSE + ) + expect_true(is.character(metadata_json$snapshot_id)) + expect_equal(get_provenance(written)$snapshot_id, metadata_json$snapshot_id) + + expect_error( + write_immundata(idata, output_folder = NULL, tag = "../bad"), + "must not include path separators" + ) +}) + +test_that("write_immundata() rehome controls future auto-snapshot root", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + alt_output_dir <- layout$projectB + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + idata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + moved_without_rehome <- write_immundata(idata, output_folder = alt_output_dir, rehome = FALSE) + auto_from_old_home <- write_immundata(moved_without_rehome, output_folder = NULL, tag = "baseline") + prov_old_home <- get_provenance(auto_from_old_home) + expect_equal( + prov_old_home$current_path, + normalizePath(file.path(output_dir, "snapshots", "baseline", "v001"), mustWork = FALSE) + ) + + moved_with_rehome <- write_immundata(idata, output_folder = alt_output_dir, rehome = TRUE) + auto_from_new_home <- write_immundata(moved_with_rehome, output_folder = NULL, tag = "baseline") + prov_new_home <- get_provenance(auto_from_new_home) + expect_equal( + prov_new_home$current_path, + normalizePath(file.path(alt_output_dir, "snapshots", "baseline", "v001"), mustWork = FALSE) + ) +}) + +test_that("operation outputs preserve provenance for auto-snapshots", { + layout <- create_snapshot_test_layout() + on.exit(cleanup_snapshot_test_root()) + output_dir <- layout$projectA + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + idata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + filtered <- filter_immundata(idata, TRUE) + snap <- write_immundata(filtered, output_folder = NULL, tag = "ops") + prov <- get_provenance(snap) + + expect_equal( + prov$current_path, + normalizePath(file.path(output_dir, "snapshots", "ops", "v001"), mustWork = FALSE) + ) + + aggregated <- agg_repertoires(idata, "imd_filename") + downsampled <- downsample_immundata(aggregated, n = 0.5, seed = 1) + downsampled_snap <- write_immundata(downsampled, output_folder = NULL, tag = "downsample") + downsampled_prov <- get_provenance(downsampled_snap) + + expect_equal( + downsampled_prov$current_path, + normalizePath(file.path(output_dir, "snapshots", "downsample", "v001"), mustWork = FALSE) + ) +}) + +test_that("write/read roundtrip preserves repertoire and strata state from metadata.json", { + output_dir <- create_test_output_dir("strata_roundtrip_") + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) |> + agg_strata(schema = "Response") + + strata_ids <- sort(unique(idata$repertoires[[imd_schema("strata")]])) + custom_names <- paste0("Custom_", strata_ids) + names(custom_names) <- as.character(strata_ids) + idata <- rename_strata(idata, names = custom_names) + + repertoire_col <- imd_schema("repertoire") + strata_col <- imd_schema("strata") + shifted_annotations <- idata$annotations |> + dplyr::mutate( + !!rlang::sym(repertoire_col) := !!rlang::sym(repertoire_col) + 1000L, + !!rlang::sym(strata_col) := !!rlang::sym(strata_col) + 100L + ) + shifted_repertoires <- idata$repertoires + shifted_repertoires[[repertoire_col]] <- shifted_repertoires[[repertoire_col]] + 1000L + shifted_repertoires[[strata_col]] <- shifted_repertoires[[strata_col]] + 100L + shifted_repertoires$json_flag <- rep(c(TRUE, NA), length.out = nrow(shifted_repertoires)) + shifted_repertoires$json_score <- rep(c(1.5, NA_real_), length.out = nrow(shifted_repertoires)) + shifted_repertoires$json_label <- rep(c("A", NA_character_), length.out = nrow(shifted_repertoires)) + shifted_annotations <- shifted_annotations |> + dplyr::left_join( + duckplyr::as_duckdb_tibble( + shifted_repertoires |> + dplyr::select(all_of(c( + repertoire_col, + "json_flag", + "json_score", + "json_label" + ))) + ), + by = repertoire_col + ) + shifted_strata <- shifted_repertoires |> + dplyr::select(all_of(c( + strata_col, + imd_schema("strata_name"), + idata$schema_strata + ))) |> + dplyr::distinct() |> + duckplyr::as_duckdb_tibble() + + original <- ImmunData$new( + schema = idata$schema_receptor, + annotations = shifted_annotations, + repertoires = duckplyr::as_duckdb_tibble(shifted_repertoires), + strata = shifted_strata, + provenance = get_provenance(idata) + ) + + write_immundata(original, output_folder = output_dir) + loaded <- read_immundata(output_dir, verbose = FALSE) + + metadata_json <- jsonlite::read_json( + file.path(output_dir, "metadata.json"), + simplifyVector = FALSE + ) + expect_equal( + unlist(metadata_json$schema_strata, use.names = FALSE), + original$schema_strata + ) + expect_true(is.list(metadata_json$repertoires)) + expect_equal(names(metadata_json$repertoires), names(original$repertoires)) + + expect_equal(loaded$schema_repertoire, original$schema_repertoire) + expect_equal(loaded$schema_strata, original$schema_strata) + expect_equal( + as.data.frame(loaded$repertoires), + as.data.frame(original$repertoires), + ignore_attr = TRUE + ) + expect_equal( + as.data.frame(loaded$strata), + as.data.frame(original$strata), + ignore_attr = TRUE + ) + + loaded_ids <- loaded$annotations |> + dplyr::select(all_of(c(repertoire_col, strata_col))) |> + dplyr::distinct() |> + dplyr::collect() |> + dplyr::arrange(.data[[repertoire_col]]) + original_ids <- original$annotations |> + dplyr::select(all_of(c(repertoire_col, strata_col))) |> + dplyr::distinct() |> + dplyr::collect() |> + dplyr::arrange(.data[[repertoire_col]]) + expect_equal(loaded_ids, original_ids, ignore_attr = TRUE) +}) + +test_that("read_immundata errors when repertoire schema lacks serialized repertoires", { + output_dir <- create_test_output_dir("missing_repertoires_") + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + repertoire_schema = NULL, + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + metadata_path <- file.path(output_dir, "metadata.json") + metadata_json <- jsonlite::read_json(metadata_path, simplifyVector = FALSE) + metadata_json$schema_repertoire <- "imd_filename" + metadata_json["repertoires"] <- list(NULL) + jsonlite::write_json( + metadata_json, + metadata_path, + auto_unbox = TRUE, + null = "null", + pretty = TRUE + ) + + expect_error( + read_immundata(output_dir, verbose = FALSE), + "declares a repertoire schema.*does not contain serialized repertoire data" + ) +}) + +test_that("read_immundata reports missing Parquet columns without collecting annotations", { + output_dir <- create_test_output_dir("snapshot_schema_source_") + broken_dir <- create_test_output_dir("snapshot_schema_broken_") + on.exit(cleanup_output_dir(output_dir), add = TRUE) + on.exit(cleanup_output_dir(broken_dir), add = TRUE) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + repertoire_schema = NULL, + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + dir.create(broken_dir, recursive = TRUE) + file.copy(file.path(output_dir, "metadata.json"), broken_dir) + duckplyr::read_parquet_duckdb(file.path(output_dir, "annotations.parquet")) |> + dplyr::select(-imd_receptor_id, -cdr3_aa) |> + duckplyr::compute_parquet(file.path(broken_dir, "annotations.parquet")) + + error <- tryCatch( + read_immundata(broken_dir, verbose = FALSE), + error = identity + ) + expect_s3_class(error, "error") + expect_match(conditionMessage(error), "Cannot load ImmunData snapshot", fixed = TRUE) + expect_match(conditionMessage(error), "imd_receptor_id", fixed = TRUE) + expect_match(conditionMessage(error), "cdr3_aa", fixed = TRUE) + expect_match(conditionMessage(error), "No data was loaded", fixed = TRUE) +}) + +test_that("read_immundata validates declared repertoire columns", { + output_dir <- create_test_output_dir("snapshot_repertoire_columns_") + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + repertoire_schema = "imd_filename", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + metadata_path <- file.path(output_dir, "metadata.json") + metadata_json <- jsonlite::read_json(metadata_path, simplifyVector = FALSE) + metadata_json$schema_repertoire <- c(metadata_json$schema_repertoire, "repertoire_only") + metadata_json$repertoires$repertoire_only <- metadata_json$repertoires$imd_filename + metadata_json$repertoires$n_receptors <- NULL + jsonlite::write_json( + metadata_json, + metadata_path, + auto_unbox = TRUE, + null = "null", + pretty = TRUE + ) + + error <- tryCatch( + read_immundata(output_dir, verbose = FALSE), + error = identity + ) + expect_match(conditionMessage(error), "repertoire_only", fixed = TRUE) + expect_match(conditionMessage(error), "n_receptors", fixed = TRUE) +}) + +test_that("read_immundata validates malformed and incomplete metadata fields", { + output_dir <- create_test_output_dir("snapshot_metadata_boundary_") + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + metadata_path <- file.path(output_dir, "metadata.json") + valid_metadata <- jsonlite::read_json(metadata_path, simplifyVector = FALSE) + + malformed_metadata <- valid_metadata + malformed_metadata$producer <- "not a metadata object" + jsonlite::write_json( + malformed_metadata, + metadata_path, + auto_unbox = TRUE, + null = "null", + pretty = TRUE + ) + expect_error(read_immundata(output_dir, verbose = FALSE)) + + incomplete_metadata <- valid_metadata + incomplete_metadata$extensions <- NULL + jsonlite::write_json( + incomplete_metadata, + metadata_path, + auto_unbox = TRUE, + null = "null", + pretty = TRUE + ) + expect_error( + read_immundata(output_dir, verbose = FALSE), + "missing required field" + ) +}) + +test_that("read_immundata rejects unsupported snapshot format versions", { + output_dir <- create_test_output_dir("snapshot_unsupported_version_") + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + metadata_path <- file.path(output_dir, "metadata.json") + metadata_json <- jsonlite::read_json(metadata_path, simplifyVector = FALSE) + metadata_json$format_version <- 99L + jsonlite::write_json( + metadata_json, + metadata_path, + auto_unbox = TRUE, + null = "null", + pretty = TRUE + ) + + expect_error( + read_immundata(output_dir, verbose = FALSE), + "Unsupported.*format version" + ) +}) + +test_that("read_immundata rejects unreadable annotation snapshots", { + output_dir <- create_test_output_dir("snapshot_corrupt_annotations_") + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + writeLines("not a parquet file", file.path(output_dir, "annotations.parquet")) + expect_error(read_immundata(output_dir, verbose = FALSE)) +}) + +test_that("read_immundata validates declared strata columns", { + output_dir <- create_test_output_dir("snapshot_strata_columns_") + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + idata <- get_test_idata() |> + agg_repertoires(c("Response", "Therapy")) |> + agg_strata(schema = "Response") + write_immundata(idata, output_folder = output_dir) + + metadata_path <- file.path(output_dir, "metadata.json") + metadata_json <- jsonlite::read_json(metadata_path, simplifyVector = FALSE) + metadata_json$repertoires[[imd_schema("strata")]] <- NULL + jsonlite::write_json( + metadata_json, + metadata_path, + auto_unbox = TRUE, + null = "null", + pretty = TRUE + ) + + expect_error( + read_immundata(output_dir, verbose = FALSE), + imd_schema("strata") + ) +}) diff --git a/tests/testthat/test-io-repertoires-agg.R b/tests/testthat/test-io-repertoires-agg.R new file mode 100644 index 0000000..888e6eb --- /dev/null +++ b/tests/testthat/test-io-repertoires-agg.R @@ -0,0 +1,530 @@ +# ============================================================================ +# TEST 1: Single-chain data with duplicated receptors +# ============================================================================ + +test_that("agg_repertoires counts single-chain receptors correctly", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create synthetic single-chain data + # Sample1: 3 cells, 2 unique receptors (cells 1&2 share receptor) + # Sample2: 2 cells, 2 unique receptors + test_data <- make_single_chain_shared_receptor_test_data() + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + # Read and aggregate + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = "IGH" + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + # Aggregate repertoires by sample_id + idata_agg <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "single-chain duplicated receptors" + ) + + repertoires <- idata_agg$repertoires |> collect() + annotations <- idata_agg$annotations |> collect() + + # Expected results: + # Sample1: n_barcodes = 3, n_receptors = 2 (cells 1&2 share receptor) + # Sample2: n_barcodes = 2, n_receptors = 2 + + sample1_stats <- repertoires |> filter(sample_id == "Sample1") + sample2_stats <- repertoires |> filter(sample_id == "Sample2") + + expect_equal(sample1_stats$n_barcodes, 3, + info = "Sample1 should have 3 barcodes" + ) + expect_equal(sample1_stats$n_receptors, 2, + info = "Sample1 should have 2 unique receptors" + ) + expect_equal(sample2_stats$n_barcodes, 2, + info = "Sample2 should have 2 barcodes" + ) + expect_equal(sample2_stats$n_receptors, 2, + info = "Sample2 should have 2 unique receptors" + ) + + # Verify receptor counts within repertoires + receptor_counts <- annotations |> + group_by(sample_id, imd_receptor_id) |> + summarise( + total_count1 = first(imd_count), + total_count2 = last(imd_count), + .groups = "drop" + ) + + # The shared receptor in Sample1 should have count = 2 + sample1_counts <- receptor_counts |> + filter(sample_id == "Sample1") |> + arrange(desc(total_count1)) + + expect_equal(sample1_counts$total_count1[1], 2, + info = "Shared receptor should have count of 2" + ) + + sample1_counts2 <- receptor_counts |> + filter(sample_id == "Sample1") |> + arrange(desc(total_count2)) + + expect_equal(sample1_counts$total_count2[1], 2, + info = "Shared receptor should have count of 2" + ) +}) + +# ============================================================================ +# TEST 2: Paired-chain data +# ============================================================================ + +test_that("agg_repertoires counts paired-chain barcodes correctly", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create synthetic paired-chain data + # Each receptor appears TWICE (once per chain) + test_data <- data.frame( + cell_id = c("cell1", "cell1", "cell2", "cell2", "cell3", "cell3"), + sample_id = c("Sample1", "Sample1", "Sample1", "Sample1", "Sample2", "Sample2"), + v_call = c("IGHV1", "IGLV1", "IGHV2", "IGLV2", "IGHV3", "IGLV3"), + j_call = c("IGHJ1", "IGLJ1", "IGHJ2", "IGLJ2", "IGHJ3", "IGLJ3"), + junction_aa = c("CARW", "CASL", "CBRW", "CBSL", "CCRW", "CCSL"), + locus = c("IGH", "IGL", "IGH", "IGL", "IGH", "IGL"), + umi_count = c(100, 100, 150, 150, 200, 200) # Same count for both chains + ) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + # Read with paired-chain schema + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGL") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + # Aggregate repertoires + idata_agg <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "paired-chain barcode counts" + ) + + repertoires <- idata_agg$repertoires |> collect() + annotations <- idata_agg$annotations |> collect() + + # Expected results: + # Sample1: n_barcodes = 2 (actual cells) + # Sample2: n_barcodes = 1 (actual cell) + + sample1_stats <- repertoires |> filter(sample_id == "Sample1") + sample2_stats <- repertoires |> filter(sample_id == "Sample2") + + # Test for n_receptors + expect_equal(sample1_stats$n_receptors, 2, + info = "Sample1 should have 2 unique receptors" + ) + expect_equal(sample2_stats$n_receptors, 1, + info = "Sample2 should have 1 unique receptor" + ) + + expect_equal(sample1_stats$n_barcodes, 2) + expect_equal(sample2_stats$n_barcodes, 1) +}) + +# ============================================================================ +# TEST 3: Mixed scenario - paired-chain with shared receptors +# ============================================================================ + +test_that("agg_repertoires handles paired-chain data with shared receptors", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create paired-chain data where some cells share receptors + test_data <- data.frame( + cell_id = c("cell1", "cell1", "cell2", "cell2", "cell3", "cell3"), + sample_id = c("Sample1", "Sample1", "Sample1", "Sample1", "Sample1", "Sample1"), + v_call = c("IGHV1", "IGLV1", "IGHV1", "IGLV1", "IGHV2", "IGLV2"), + j_call = c("IGHJ1", "IGLJ1", "IGHJ1", "IGLJ1", "IGHJ2", "IGLJ2"), + junction_aa = c("CARW", "CASL", "CARW", "CASL", "CBRW", "CBSL"), + locus = c("IGH", "IGL", "IGH", "IGL", "IGH", "IGL"), + umi_count = c(100, 100, 150, 150, 200, 200) + ) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGL") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + idata_agg <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "paired-chain shared receptors" + ) + + repertoires <- idata_agg$repertoires |> collect() + annotations <- idata_agg$annotations |> collect() + + # Expected (correct) results: + # - 3 actual cells + # - 2 unique receptors (cells 1&2 share) + + expect_equal(repertoires$n_receptors, 2, + info = "Should have 2 unique receptors" + ) + # Check receptor counts + receptor_counts <- annotations |> + select(imd_receptor_id, imd_count) |> + distinct() + + # The shared receptor should have count = 2 (two cells sharing the receptor) + shared_receptor_count <- max(receptor_counts$imd_count) + expect_equal(shared_receptor_count, 2) +}) + +# ============================================================================ +# TEST 4: Proportions calculation +# ============================================================================ + +test_that("agg_repertoires calculates paired-chain proportions correctly", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Simple paired-chain data for clear proportion calculation + test_data <- data.frame( + cell_id = c("cell1", "cell1", "cell2", "cell2"), + sample_id = c("Sample1", "Sample1", "Sample1", "Sample1"), + v_call = c("IGHV1", "IGLV1", "IGHV2", "IGLV2"), + j_call = c("IGHJ1", "IGLJ1", "IGHJ2", "IGLJ2"), + junction_aa = c("CARW", "CASL", "CBRW", "CBSL"), + locus = c("IGH", "IGL", "IGH", "IGL"), + umi_count = c(100, 100, 100, 100) # Equal counts + ) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGL") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + idata_agg <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "paired-chain proportions" + ) + annotations <- idata_agg$annotations |> collect() + + # Get unique proportions + proportions <- annotations |> + select(imd_receptor_id, imd_proportion) |> + distinct() + + expect_equal(unique(proportions$imd_proportion), 0.5) +}) + +# ============================================================================ +# TEST 5: Verify n_repertoires calculation (should be correct) +# ============================================================================ + +test_that("agg_repertoires n_repertoires calculation is correct for paired data", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create data where one receptor appears in multiple samples + test_data <- data.frame( + cell_id = c("cell1", "cell1", "cell2", "cell2", "cell3", "cell3"), + sample_id = c("Sample1", "Sample1", "Sample2", "Sample2", "Sample3", "Sample3"), + v_call = c("IGHV1", "IGLV1", "IGHV1", "IGLV1", "IGHV2", "IGLV2"), + j_call = c("IGHJ1", "IGLJ1", "IGHJ1", "IGLJ1", "IGHJ2", "IGLJ2"), + junction_aa = c("CARW", "CASL", "CARW", "CASL", "CBRW", "CBSL"), + locus = c("IGH", "IGL", "IGH", "IGL", "IGH", "IGL"), + umi_count = rep(100, 6) + ) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGL") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + idata_agg <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "paired-chain n_repertoires" + ) + annotations <- idata_agg$annotations |> collect() + + # Check n_repertoires for the shared receptor + n_repertoires_vals <- annotations |> + select(imd_receptor_id, n_repertoires) |> + distinct() + + # First receptor appears in Sample1 and Sample2 (n_repertoires = 2) + # Second receptor appears only in Sample3 (n_repertoires = 1) + + expect_true(2 %in% n_repertoires_vals$n_repertoires, + info = "Shared receptor should appear in 2 repertoires" + ) + expect_true(1 %in% n_repertoires_vals$n_repertoires, + info = "Unique receptor should appear in 1 repertoire" + ) +}) + +test_that("agg_repertoires preserves second chain data (no NAs)", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create paired-chain data + # Cell1: IGH + IGL + test_data <- data.frame( + cell_id = c("cell1", "cell1"), + sample_id = c("Sample1", "Sample1"), + v_call = c("IGHV1", "IGLV1"), + j_call = c("IGHJ1", "IGLJ1"), + junction_aa = c("CARW", "CASL"), + locus = c("IGH", "IGL"), + umi_count = c(100, 100) + ) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGL") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir + ) + + # Run aggregation + idata_agg <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "paired-chain no-NA preservation" + ) + annotations <- idata_agg$annotations |> collect() + + # ---------------------------------------------------------------------- + # CHECK 1: Row Count + # The input had 2 rows (1 cell x 2 chains). + # The output MUST have 2 rows. + # (The bug caused this to drop to 1 row). + # ---------------------------------------------------------------------- + expect_equal(nrow(annotations), 2, + info = "Annotation table should retain rows for both chains (IGH and IGL)" + ) + + # ---------------------------------------------------------------------- + # CHECK 2: Second Chain Data Presence + # Filter for the Light Chain row and ensure V-call is present (not NA) + # ---------------------------------------------------------------------- + igl_row <- annotations |> filter(locus == "IGL") + + expect_equal(nrow(igl_row), 1, + info = "Should find exactly one row for the IGL chain" + ) + + expect_false(is.na(igl_row$v_call), + info = "IGL row should have valid v_call data (not NA)" + ) + expect_equal(igl_row$v_call, "IGLV1", + info = "IGL v_call should match input data" + ) + + # ---------------------------------------------------------------------- + # CHECK 3: Statistics Mapping + # Ensure the calculated stats (which are per-receptor) are mapped + # correctly to BOTH chain rows. + # ---------------------------------------------------------------------- + # Both IGH and IGL rows belong to the same receptor, so they should + # both have the same imd_count and imd_proportion. + + igh_row <- annotations |> filter(locus == "IGH") + + expect_equal(igh_row$imd_count, igl_row$imd_count, + info = "Both chains of the same receptor should share the same count stats" + ) + expect_equal(igh_row$imd_repertoire_id, igl_row$imd_repertoire_id, + info = "Both chains should belong to the same repertoire ID" + ) +}) + +test_that("agg_repertoires keeps repertoire IDs aligned with their schema", { + output_dir <- create_test_output_dir("repertoire_mapping_") + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + set.seed(1) + + n_repertoires <- 128L + n_barcodes <- 1000L + n_rows <- n_repertoires * n_barcodes + + annotations <- tibble::tibble( + sample_id = rep( + sprintf("sample_%03d", seq_len(n_repertoires)), + each = n_barcodes + ), + imd_receptor_id = rep( + seq_len(n_repertoires), + each = n_barcodes + ), + imd_barcode = seq_len(n_rows), + imd_chain_id = seq_len(n_rows), + imd_n_chains = 1L, + cdr3_aa = rep( + sprintf("CASS%03d", seq_len(n_repertoires)), + each = n_barcodes + ) + ) |> + dplyr::slice_sample(prop = 1) |> + duckplyr::as_duckdb_tibble() + + idata <- ImmunData$new( + schema = "cdr3_aa", + annotations = annotations + ) + + aggregated <- agg_repertoires_with_integrity( + idata, + schema = "sample_id", + context = "repertoire ID and schema mapping" + ) + + expected_mapping <- tibble::tibble( + imd_repertoire_id = seq_len(n_repertoires), + sample_id = sprintf("sample_%03d", seq_len(n_repertoires)) + ) + + actual_mapping <- aggregated$repertoires |> + dplyr::select(imd_repertoire_id, sample_id) |> + dplyr::arrange(imd_repertoire_id) |> + dplyr::collect() + + expect_equal(actual_mapping, expected_mapping) + + write_immundata(aggregated, output_folder = output_dir) + loaded <- read_immundata(output_dir, verbose = FALSE) + + expect_agg_repertoires_integrity( + loaded, + context = "repertoire ID and schema mapping after snapshot roundtrip", + schema = "sample_id" + ) + + loaded_mapping <- loaded$repertoires |> + dplyr::select(imd_repertoire_id, sample_id) |> + dplyr::arrange(imd_repertoire_id) |> + dplyr::collect() + + expect_equal(loaded_mapping, expected_mapping) +}) + +test_that("agg_repertoires orders IDs by a composite schema", { + annotations <- tibble::tibble( + sample_id = c("sample_b", "sample_a", "sample_b", "sample_a"), + timepoint = c("day_2", "day_2", "day_1", "day_1"), + imd_receptor_id = seq_len(4L), + imd_barcode = seq_len(4L), + imd_chain_id = seq_len(4L), + imd_n_chains = 1L, + cdr3_aa = paste0("CASS", seq_len(4L)) + ) |> + duckplyr::as_duckdb_tibble() + + idata <- ImmunData$new( + schema = "cdr3_aa", + annotations = annotations + ) + + aggregated <- agg_repertoires_with_integrity( + idata, + schema = c("sample_id", "timepoint"), + context = "composite repertoire schema mapping" + ) + + actual_mapping <- aggregated$repertoires |> + dplyr::select(imd_repertoire_id, sample_id, timepoint) |> + dplyr::arrange(imd_repertoire_id) |> + dplyr::collect() + + expected_mapping <- tibble::tribble( + ~imd_repertoire_id, ~sample_id, ~timepoint, + 1L, "sample_a", "day_1", + 2L, "sample_a", "day_2", + 3L, "sample_b", "day_1", + 4L, "sample_b", "day_2" + ) + + expect_equal(actual_mapping, expected_mapping) +}) diff --git a/tests/testthat/test-io-repertoires-counts.R b/tests/testthat/test-io-repertoires-counts.R new file mode 100644 index 0000000..0d1b990 --- /dev/null +++ b/tests/testthat/test-io-repertoires-counts.R @@ -0,0 +1,402 @@ +# ============================================================================ +# USE CASE 1: SINGLE CHAIN +# ============================================================================ + +test_that("Single chain: correct barcode and receptor counts", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create toy data: 5 cells, all with IGH + test_data <- make_single_chain_shared_receptor_test_data() + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = "IGH" + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + annotations <- idata$annotations |> collect() + + # Expected counts: + # - 5 barcodes (all cells have IGH) + # - 4 unique receptors (cell1 and cell3 have identical receptors) + + expect_equal(n_distinct(annotations$imd_barcode), 5) + expect_equal(n_distinct(annotations$imd_receptor_id), 4) + + # Verify receptor sharing + receptor_counts <- annotations |> + group_by(imd_receptor_id) |> + summarise(n_cells = n_distinct(imd_barcode), .groups = "drop") + + expect_true(any(receptor_counts$n_cells == 2)) +}) + +test_that("Single chain with mixed loci: filters correctly", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create toy data with mixed loci + test_data <- data.frame( + cell_id = c("cell1", "cell2", "cell3", "cell4", "cell5"), + v_call = c("IGHV1", "IGLV1", "IGHV2", "IGKV1", "IGHV3"), + j_call = c("IGHJ1", "IGLJ1", "IGHJ2", "IGKJ1", "IGHJ3"), + junction_aa = c("CARW", "CASL", "CBRW", "CASK", "CCRW"), + locus = c("IGH", "IGL", "IGH", "IGK", "IGH"), + umi_count = c(100, 200, 150, 300, 250) + ) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + # Request only IGH chains + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = "IGH" + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + annotations <- idata$annotations |> collect() + + # Expected: Only cells 1, 3, and 5 (those with IGH) + expect_equal(n_distinct(annotations$imd_barcode), 3, + info = "Should have only 3 cells with IGH" + ) + expect_setequal(annotations$imd_barcode, c("cell1", "cell3", "cell5")) +}) + +# ============================================================================ +# USE CASE 2: STRICT PAIRING +# ============================================================================ + +test_that("Strict pairing: correct counts for perfect pairs", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create toy data with perfect IGH+IGL pairs + test_data <- data.frame( + cell_id = c("cell1", "cell1", "cell2", "cell2", "cell3", "cell3"), + v_call = c("IGHV1", "IGLV1", "IGHV2", "IGLV2", "IGHV1", "IGLV1"), + j_call = c("IGHJ1", "IGLJ1", "IGHJ2", "IGLJ2", "IGHJ1", "IGLJ1"), + junction_aa = c("CARW", "CASL", "CBRW", "CBSL", "CARW", "CASL"), + locus = c("IGH", "IGL", "IGH", "IGL", "IGH", "IGL"), + umi_count = c(100, 80, 120, 90, 110, 85) + ) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGL") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + annotations <- idata$annotations |> collect() + + # Expected: + # - 3 barcodes (all have complete pairs) + # - 2 unique receptors (cell1 and cell3 share the same receptor) + + expect_equal(n_distinct(annotations$imd_barcode), 3) + expect_equal(n_distinct(annotations$imd_receptor_id), 2) +}) + +test_that("Strict pairing: excludes incomplete pairs", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create toy data with some incomplete pairs + test_data <- data.frame( + cell_id = c("cell1", "cell1", "cell2", "cell3", "cell4", "cell4"), + v_call = c("IGHV1", "IGLV1", "IGHV2", "IGLV3", "IGHV4", "IGKV4"), + j_call = c("IGHJ1", "IGLJ1", "IGHJ2", "IGLJ3", "IGHJ4", "IGKJ4"), + junction_aa = c("CARW", "CASL", "CBRW", "CCSL", "CDRW", "CDSK"), + locus = c("IGH", "IGL", "IGH", "IGL", "IGH", "IGK"), + umi_count = c(100, 80, 120, 90, 110, 85) + ) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + # Request strict IGH+IGL pairing + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGL") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + annotations <- idata$annotations |> collect() + + # Expected: Only cell1 has both IGH and IGL + # cell2 has only IGH, cell3 has only IGL, cell4 has IGH+IGK (wrong light chain) + + expect_equal(n_distinct(annotations$imd_barcode), 1) + expect_equal(unique(annotations$imd_barcode), "cell1") +}) + +test_that("Strict pairing: handles duplicate chains correctly", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create data with multiple chains of same locus per cell + test_data <- make_duplicate_chain_test_data() + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGL") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + annotations <- idata$annotations |> collect() + + # Should select highest UMI for each locus + expect_equal(nrow(annotations), 2) + + igh_chain <- annotations |> filter(locus == "IGH") + igl_chain <- annotations |> filter(locus == "IGL") + + expect_equal(igh_chain$v_call, "IGHV2") + expect_equal(igl_chain$v_call, "IGLV1") +}) + +# ============================================================================ +# USE CASE 3: RELAXED PAIRING WITH ARTIFACT EXCLUSION +# ============================================================================ + +test_that("Relaxed pairing: correct counts with artifact exclusion", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create toy data with various scenarios + test_data <- make_relaxed_pairing_test_data() + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGL|IGK") # Relaxed pairing + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + annotations <- idata$annotations |> collect() + + # Expected results: + # - normal_igl: IGH + IGL ✓ (valid) + # - normal_igk: IGH + IGK ✓ (valid) + # - artifact: EXCLUDED (has both IGL and IGK) + # - heavy_only: EXCLUDED (missing light chain) + # - light_only: EXCLUDED (missing heavy chain) + # - two_lights: EXCLUDED (missing heavy chain) + + expect_equal(n_distinct(annotations$imd_barcode), 2) + expect_setequal(unique(annotations$imd_barcode), c("normal_igl", "normal_igk")) + expect_equal(n_distinct(annotations$imd_receptor_id), 2) +}) + +# ============================================================================ +# USE CASE 3.5: BULK DATA (no barcodes, with counts) +# ============================================================================ + +test_that("Bulk data: correct receptor counts without barcodes", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create bulk data (no cell IDs, but with clone counts) + test_data <- data.frame( + v_call = c("IGHV1", "IGHV1", "IGHV2", "IGHV3"), + j_call = c("IGHJ1", "IGHJ1", "IGHJ2", "IGHJ3"), + junction_aa = c("CARW", "CARW", "CBRW", "CCRW"), + locus = c("IGH", "IGH", "IGH", "IGH"), + clone_count = c(100, 50, 200, 75) # Frequency counts + ) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + idata <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = "IGH" + ), + count_col = "clone_count", # Bulk mode with counts + locus_col = "locus", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + annotations <- idata$annotations |> collect() + + # Expected: + # - 4 synthetic barcodes (one per row) + # - 3 unique receptors (rows 1 and 2 have identical receptors) + # - Chain counts should match clone_count values + + expect_equal(n_distinct(annotations$imd_barcode), 4) + expect_equal(n_distinct(annotations$imd_receptor_id), 3) + expect_equal(annotations$imd_n_chains, test_data$clone_count) +}) + +test_that("agg_receptors rejects negative bulk counts", { + dataset <- duckplyr::as_duckdb_tibble( + tibble::tibble( + v_call = c("IGHV1", "IGHV2"), + j_call = c("IGHJ1", "IGHJ2"), + junction_aa = c("CARW", "CBRW"), + clone_count = c(10, -1) + ) + ) + + expect_error( + agg_receptors( + dataset = dataset, + schema = c("v_call", "j_call", "junction_aa"), + count_col = "clone_count" + ), + "non-negative|negative" + ) +}) + +# ============================================================================ +# COMPARISON TESTS +# ============================================================================ + +test_that("Comparison: relaxed vs strict pairing counts", { + output_dir_strict <- create_test_output_dir() + output_dir_relaxed <- create_test_output_dir() + on.exit({ + cleanup_output_dir(output_dir_strict) + cleanup_output_dir(output_dir_relaxed) + }) + + test_data <- make_relaxed_pairing_test_data() + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + # Strict IGH + IGL + idata_strict <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGL") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir_strict, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + # Relaxed IGH + (IGL|IGK) + idata_relaxed <- read_repertoires( + path = temp_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("IGH", "IGL|IGK") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir_relaxed, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + strict_cells <- idata_strict$annotations |> + collect() |> + pull(imd_barcode) |> + unique() + relaxed_cells <- idata_relaxed$annotations |> + collect() |> + pull(imd_barcode) |> + unique() + + # Strict should include: normal_igl, artifact (picking IGL) + # Relaxed should include: normal_igl, normal_igk (excluding artifact) + + expect_true("artifact" %in% strict_cells, + info = "Strict pairing includes artifact cell" + ) + expect_false("artifact" %in% relaxed_cells, + info = "Relaxed pairing excludes artifact cell" + ) + expect_equal(length(relaxed_cells), 2, + info = "Relaxed should have 2 valid cells" + ) +}) diff --git a/tests/testthat/test-io-repertoires-files.R b/tests/testthat/test-io-repertoires-files.R new file mode 100644 index 0000000..2e68b04 --- /dev/null +++ b/tests/testthat/test-io-repertoires-files.R @@ -0,0 +1,626 @@ +test_that("read_repertoires() fails if path doesn't exist", { + expect_error( + read_repertoires( + path = "nonexistent_file.tsv", + schema = c("cdr3_aa", "v_call") + ), + "No file provided|does not exist|cannot find" + ) +}) + +test_that("read_repertoires() works with single file input", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Single file as documented + inp_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + idata <- read_repertoires( + path = inp_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, # Disable for testing + postprocess = NULL + ) + + # Verify result + expect_s3_class(idata, "ImmunData") + expect_true(file.exists(file.path(output_dir, "annotations.parquet"))) + expect_true(file.exists(file.path(output_dir, "metadata.json"))) + + # Check data was loaded + annotations <- idata$annotations |> collect() + expect_gt(nrow(annotations), 0) + + # Check required columns exist + expect_true("imd_receptor_id" %in% colnames(annotations)) + expect_true("cdr3_aa" %in% colnames(annotations)) + expect_true("v_call" %in% colnames(annotations)) +}) + +test_that("read_repertoires() works with vector of file names", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Vector of files as documented + inp_file1 <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + inp_file2 <- system.file("extdata/tsv", "sample_1k_2k.tsv", package = "immundata") + file_vec <- c(inp_file1, inp_file2) + + idata <- read_repertoires( + path = file_vec, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + # Verify result + expect_s3_class(idata, "ImmunData") + + # Check that data from both files is present + annotations <- idata$annotations |> collect() + expect_gt(nrow(annotations), 0) + + # Should have data from both files + if ("imd_filename" %in% colnames(annotations)) { + unique_files <- unique(basename(annotations$imd_filename)) + expect_true("sample_0_1k.tsv" %in% unique_files || length(unique_files) > 0) + } +}) + +test_that("text input is prematerialized by default and the temporary file is removed", { + output_dir <- create_test_output_dir() + prematerialize_dir <- tempfile("test-prematerialize-") + dir.create(prematerialize_dir) + on.exit(cleanup_output_dir(output_dir), add = TRUE) + on.exit(cleanup_output_dir(prematerialize_dir), add = TRUE) + + input_files <- c( + system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata"), + system.file("extdata/tsv", "sample_1k_2k.tsv", package = "immundata") + ) + + idata <- read_repertoires( + path = input_files, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + prematerialize_folder = prematerialize_dir, + preprocess = NULL, + postprocess = NULL, + verbose = FALSE + ) + + observed_files <- idata$annotations |> + collect() |> + distinct(imd_filename) |> + pull(imd_filename) + + expect_setequal(observed_files, normalizePath(input_files)) + expect_length(list.files(prematerialize_dir, all.files = TRUE, no.. = TRUE), 0L) + + metadata <- jsonlite::read_json( + file.path(output_dir, "metadata.json"), + simplifyVector = FALSE + ) + ingestion_event <- metadata$lineage[[1]] + expect_equal( + unlist(ingestion_event$inputs$files, use.names = FALSE), + normalizePath(input_files) + ) + expect_true(isTRUE(ingestion_event$pipeline$prematerialize$requested)) + expect_true(isTRUE(ingestion_event$pipeline$prematerialize$applied)) +}) + +test_that("Parquet input skips prematerialization", { + input_file <- tempfile(fileext = ".parquet") + output_dir <- create_test_output_dir() + unused_prematerialize_dir <- tempfile("test-unused-prematerialize-") + on.exit(unlink(input_file), add = TRUE) + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + input_data <- duckplyr::duckdb_tibble(tibble::tibble( + cdr3_aa = c("CASSA", "CASSB"), + v_call = c("TRBV1", "TRBV2") + )) + suppressMessages(duckplyr::compute_parquet(input_data, input_file)) + + read_repertoires( + path = input_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + prematerialize_folder = unused_prematerialize_dir, + preprocess = NULL, + postprocess = NULL, + verbose = FALSE + ) + + expect_false(dir.exists(unused_prematerialize_dir)) + + metadata <- jsonlite::read_json( + file.path(output_dir, "metadata.json"), + simplifyVector = FALSE + ) + prematerialize_event <- metadata$lineage[[1]]$pipeline$prematerialize + expect_true(isTRUE(prematerialize_event$requested)) + expect_false(isTRUE(prematerialize_event$applied)) +}) + +test_that("an unusable explicit prematerialization folder errors without fallback", { + output_dir <- create_test_output_dir() + input_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + expect_error( + read_repertoires( + path = input_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + prematerialize_folder = input_file, + preprocess = NULL, + postprocess = NULL, + verbose = FALSE + ), + "Cannot create.*prematerialize_folder" + ) +}) + +test_that("the prematerialized file is removed after a downstream error", { + output_dir <- create_test_output_dir() + prematerialize_dir <- tempfile("test-prematerialize-error-") + dir.create(prematerialize_dir) + on.exit(cleanup_output_dir(output_dir), add = TRUE) + on.exit(cleanup_output_dir(prematerialize_dir), add = TRUE) + + input_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + expect_error( + read_repertoires( + path = input_file, + schema = "missing_receptor_column", + output_folder = output_dir, + prematerialize_folder = prematerialize_dir, + preprocess = NULL, + postprocess = NULL, + verbose = FALSE + ), + "Missing receptor feature column" + ) + + expect_length(list.files(prematerialize_dir, all.files = TRUE, no.. = TRUE), 0L) +}) + +test_that("single-cell pairing does not combine equal barcodes from different files", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + sample_a_file <- tempfile("sample_A_", fileext = ".tsv") + sample_b_file <- tempfile("sample_B_", fileext = ".tsv") + on.exit(unlink(c(sample_a_file, sample_b_file)), add = TRUE) + + # The AAAC-1 rows come from independent libraries. Although their raw + # barcodes match, neither AAAC-1 cell has a complete TRA/TRB receptor. + # VALID-A is an unrelated complete receptor that keeps the result non-empty. + readr::write_tsv( + tibble::tibble( + barcode = c("AAAC-1", "VALID-A", "VALID-A"), + locus = c("TRA", "TRA", "TRB"), + UMI = c(12L, 10L, 11L), + v_call = c("TRAV1", "TRAV2", "TRBV2"), + j_call = c("TRAJ1", "TRAJ2", "TRBJ2"), + junction_aa = c("CAVR", "CAVVALID", "CASSVALID") + ), + sample_a_file + ) + readr::write_tsv( + tibble::tibble( + barcode = "AAAC-1", + locus = "TRB", + UMI = 18L, + v_call = "TRBV1", + j_call = "TRBJ1", + junction_aa = "CASSR" + ), + sample_b_file + ) + + idata <- read_repertoires( + path = c(sample_a_file, sample_b_file), + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = c("TRA", "TRB") + ), + barcode_col = "barcode", + locus_col = "locus", + umi_col = "UMI", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + cross_file_receptors <- idata$annotations |> + collect() |> + summarise( + n_source_files = n_distinct(imd_filename), + .by = imd_receptor_id + ) |> + filter(n_source_files > 1L) + + expect_equal( + nrow(cross_file_receptors), + 0L, + info = "a receptor must never contain chains from distinct input files" + ) +}) + +test_that("single-chain selection treats equal barcodes from different files independently", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + sample_a_file <- tempfile("sample_A_", fileext = ".tsv") + sample_b_file <- tempfile("sample_B_", fileext = ".tsv") + on.exit(unlink(c(sample_a_file, sample_b_file)), add = TRUE) + + readr::write_tsv( + tibble::tibble( + barcode = "AAAC-1", + locus = "TRA", + UMI = 12L, + v_call = "TRAV1", + j_call = "TRAJ1", + junction_aa = "CAVA" + ), + sample_a_file + ) + readr::write_tsv( + tibble::tibble( + barcode = "AAAC-1", + locus = "TRA", + UMI = 20L, + v_call = "TRAV2", + j_call = "TRAJ2", + junction_aa = "CAVB" + ), + sample_b_file + ) + + idata <- read_repertoires( + path = c(sample_a_file, sample_b_file), + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = "TRA" + ), + barcode_col = "barcode", + locus_col = "locus", + umi_col = "UMI", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + observed_source_files <- idata$annotations |> + collect() |> + distinct(imd_filename) |> + pull(imd_filename) + + expect_setequal( + observed_source_files, + normalizePath(c(sample_a_file, sample_b_file)) + ) +}) + +test_that("read_repertoires() works with glob pattern", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Glob pattern as documented + folder_with_files <- system.file("extdata/tsv", package = "immundata") + glob_files <- file.path(folder_with_files, "sample*.tsv") + + # Verify glob expands to actual files + expanded_files <- Sys.glob(glob_files) + expect_gt(length(expanded_files), 0) + + idata <- read_repertoires( + path = glob_files, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + # Verify result + expect_s3_class(idata, "ImmunData") + annotations <- idata$annotations |> collect() + expect_gt(nrow(annotations), 0) +}) + +test_that("read_repertoires() works with manifest table and file vector", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Load manifest + manifest_path <- system.file("extdata/tsv", "manifest.csv", package = "immundata") + manifest_df <- read_manifest(manifest_path) + + # Get sample files + sample_files <- c( + system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata"), + system.file("extdata/tsv", "sample_1k_2k.tsv", package = "immundata") + ) + + idata <- read_repertoires( + path = sample_files, + schema = c("cdr3_aa", "v_call"), + manifest = manifest_df, + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + # Verify result + expect_s3_class(idata, "ImmunData") + expect_true(file.exists(file.path(output_dir, "annotations.parquet"))) + expect_true(file.exists(file.path(output_dir, "metadata.json"))) + + # Check manifest annotations were joined + annotations <- idata$annotations |> collect() + if (!is.null(manifest_df) && "Therapy" %in% colnames(manifest_df)) { + expect_true("Therapy" %in% colnames(annotations)) + expect_true("Response" %in% colnames(annotations)) + } +}) + +test_that("read_repertoires() works with directive", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Load manifest with proper file paths + manifest_path <- system.file("extdata/tsv", "manifest.csv", package = "immundata") + manifest_df <- read_manifest(manifest_path) + + idata <- read_repertoires( + path = "", + schema = c("cdr3_aa", "v_call"), + manifest = manifest_df, + manifest_file_col = "file", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + # Verify result + expect_s3_class(idata, "ImmunData") + + # Check manifest columns are present + annotations <- idata$annotations |> collect() + expect_true("Therapy" %in% colnames(annotations)) + expect_true("Response" %in% colnames(annotations)) + expect_true("Prefix" %in% colnames(annotations)) +}) + +test_that("read_repertoires() creates one repertoire per manifest row with repertoire_schema", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + manifest_path <- system.file("extdata/tsv", "manifest.csv", package = "immundata") + manifest_df <- read_manifest(manifest_path) + + idata <- read_repertoires( + path = "", + schema = c("cdr3_aa", "v_call"), + manifest = manifest_df, + manifest_file_col = "file", + repertoire_schema = "", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + expect_s3_class(idata, "ImmunData") + expect_false(is.null(idata$repertoires)) + + annotations <- idata$annotations |> collect() + repertoires <- idata$repertoires |> collect() + + expect_equal(nrow(repertoires), nrow(manifest_df)) + expect_true("imd_filename" %in% colnames(repertoires)) + expect_true("imd_filename" %in% idata$schema_repertoire) + expect_true(all(colnames(manifest_df) %in% idata$schema_repertoire)) + expect_true(all(colnames(manifest_df) %in% colnames(repertoires))) + + file_to_repertoire <- annotations |> + dplyr::summarise( + n_repertoires = dplyr::n_distinct(imd_repertoire_id), + .by = imd_filename + ) + + expect_equal(nrow(file_to_repertoire), nrow(manifest_df)) + expect_true(all(file_to_repertoire$n_repertoires == 1)) +}) + +test_that("read_repertoires() rejects repeated manifest paths", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + input_file <- tempfile(fileext = ".tsv") + on.exit(unlink(input_file), add = TRUE) + readr::write_tsv( + data.frame(cdr3_aa = "AAA", v_call = "V1"), + input_file + ) + + normalized_input_file <- normalizePath(input_file) + equivalent_input_file <- file.path( + dirname(normalized_input_file), + ".", + basename(normalized_input_file) + ) + + manifests <- list( + exact = data.frame( + file = rep(normalized_input_file, 2), + sample_id = c("S1", "S2") + ), + normalized = data.frame( + file = c(normalized_input_file, equivalent_input_file), + sample_id = c("S1", "S2") + ) + ) + + for (manifest_name in names(manifests)) { + expect_error( + read_repertoires( + path = "", + schema = c("cdr3_aa", "v_call"), + manifest = manifests[[manifest_name]], + repertoire_schema = "", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ), + "duplicated repertoire file paths after normalization", + info = manifest_name + ) + } + + explicit_path_manifest <- data.frame( + imd_filename = rep(normalized_input_file, 2), + sample_id = c("S1", "S2") + ) + + expect_error( + read_repertoires( + path = normalized_input_file, + schema = c("cdr3_aa", "v_call"), + manifest = explicit_path_manifest, + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ), + "duplicated repertoire file paths after normalization" + ) +}) + +test_that("read_repertoires() uses all manifest columns when path is ", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + manifest_path <- system.file("extdata/tsv", "manifest.csv", package = "immundata") + manifest_df <- read_manifest(manifest_path) + + idata <- read_repertoires( + path = "", + schema = c("cdr3_aa", "v_call"), + manifest = manifest_df, + manifest_file_col = "file", + repertoire_schema = "", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + repertoires <- idata$repertoires |> collect() + + expect_equal(nrow(repertoires), nrow(manifest_df)) + expect_true(all(colnames(manifest_df) %in% idata$schema_repertoire)) + expect_true(all(colnames(manifest_df) %in% colnames(repertoires))) + expect_true("imd_filename" %in% idata$schema_repertoire) +}) + +test_that("read_repertoires() creates one repertoire per file without manifest", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + inp_file1 <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + inp_file2 <- system.file("extdata/tsv", "sample_1k_2k.tsv", package = "immundata") + file_vec <- c(inp_file1, inp_file2) + + idata <- read_repertoires( + path = file_vec, + schema = c("cdr3_aa", "v_call"), + repertoire_schema = "", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + expect_s3_class(idata, "ImmunData") + expect_equal(idata$schema_repertoire, "imd_filename") + + annotations <- idata$annotations |> collect() + repertoires <- idata$repertoires |> collect() + + expect_equal(nrow(repertoires), length(file_vec)) + expect_true("imd_filename" %in% colnames(repertoires)) + + file_to_repertoire <- annotations |> + dplyr::summarise( + n_repertoires = dplyr::n_distinct(imd_repertoire_id), + .by = imd_filename + ) + + expect_equal(nrow(file_to_repertoire), length(file_vec)) + expect_true(all(file_to_repertoire$n_repertoires == 1)) +}) + +test_that("read_manifest() rejects old metadata filenames", { + manifest_dir <- tempfile("old_manifest_name_") + dir.create(manifest_dir) + on.exit(unlink(manifest_dir, recursive = TRUE), add = TRUE) + + old_path <- file.path(manifest_dir, "metadata.tsv") + writeLines(c("file", "sample_0_1k.tsv"), old_path) + + expect_error( + read_manifest(old_path), + "repertoire metadata tables are now manifests" + ) +}) + +test_that("read_repertoires() fails with when no manifest provided", { + expect_error( + read_repertoires( + path = "", + schema = c("cdr3_aa", "v_call"), + manifest = NULL + ), + "no `manifest` table provided" + ) +}) + +test_that("read_repertoires() handles custom manifest_file_col", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Create custom manifest with different column name + base_dir <- system.file("extdata/tsv", package = "immundata") + custom_manifest <- data.frame( + FilePath = c( + file.path(base_dir, "sample_0_1k.tsv"), + file.path(base_dir, "sample_1k_2k.tsv") + ), + SampleID = c("S1", "S2"), + Treatment = c("A", "B") + ) + + idata <- read_repertoires( + path = "", + schema = c("cdr3_aa", "v_call"), + manifest = custom_manifest, + manifest_file_col = "FilePath", # Custom column name + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + # Verify result + expect_s3_class(idata, "ImmunData") + annotations <- idata$annotations |> collect() + expect_true("SampleID" %in% colnames(annotations)) + expect_true("Treatment" %in% colnames(annotations)) +}) diff --git a/tests/testthat/test-io-repertoires-processing.R b/tests/testthat/test-io-repertoires-processing.R new file mode 100644 index 0000000..682f3dd --- /dev/null +++ b/tests/testthat/test-io-repertoires-processing.R @@ -0,0 +1,158 @@ +test_that("read_repertoires() excludes specified columns", { + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + exclude_cols <- c("sequence", "fwr1", "cdr1") + + imdata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), # columns that do exist + preprocess = list( + exclude_columns = make_exclude_columns(cols = exclude_cols) + ), + output_folder = file.path(tempdir(), "test-exclude") + ) + + ann_cols <- colnames(imdata$annotations) + + for (col in exclude_cols) { + expect_false( + col %in% ann_cols, + info = paste("Column", col, "should have been excluded but is still present.") + ) + } +}) + +test_that("read_repertoires() correctly renames columns (v_call -> v_gene)", { + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + rename_map <- c("v_gene" = "v_call") + + imdata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_gene"), + rename_columns = rename_map, + output_folder = file.path(tempdir(), "test-rename") + ) + + ann_cols <- colnames(imdata$annotations) + + expect_true( + "v_gene" %in% ann_cols, + info = "Renamed column 'v_call' -> 'v_gene' should appear in the annotation." + ) + expect_false( + "v_call" %in% ann_cols, + info = "Original column 'v_call' should be removed after rename." + ) +}) + +test_that("read_repertoires() excludes columns AND renames simultaneously", { + sample_file <- system.file("extdata/tsv", "sample_1k_2k.tsv", package = "immundata") + + # Suppose the data has columns "j_call" and we want to rename it to "j_gene" + rename_map <- c("j_gene" = "j_call") + exclude_cols <- c("cdr2", "fwr2") # must exist in sample_1k_2k.tsv for the test to pass + + imdata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call", "j_gene"), + preprocess = list( + exclude_columns = make_exclude_columns(cols = exclude_cols) + ), + rename_columns = rename_map, + output_folder = file.path(tempdir(), "test-exclude-rename") + ) + + ann_cols <- colnames(imdata$annotations) + + for (col in exclude_cols) { + expect_false( + col %in% ann_cols, + info = paste("Column", col, "should have been excluded.") + ) + } + + expect_true( + "j_gene" %in% ann_cols, + info = "Renamed column 'j_call' -> 'j_gene' should appear." + ) + expect_false( + "j_call" %in% ann_cols, + info = "Original column 'j_call' should be gone." + ) +}) + +test_that("read_repertoires() removes non-productive", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + imdata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir + ) + imdata_rows <- imdata |> + count() |> + pull() + + df <- readr::read_tsv(sample_file) + n_prod <- sum(df$productive) + + expect_equal( + imdata_rows, n_prod + ) +}) + +test_that("read_repertoires() correctly reads non-productive", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + imdata <- read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + preprocess = NULL, + output_folder = output_dir + ) + imdata_rows <- imdata |> + count() |> + pull() + + df <- readr::read_tsv(sample_file) + n_all <- df |> + count() |> + pull() + + expect_equal( + imdata_rows, + n_all + ) +}) + +test_that("read_repertoires() with repertoire_schema creates repertoires", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + manifest_path <- system.file("extdata/tsv", "manifest.csv", package = "immundata") + manifest_df <- read_manifest(manifest_path) + + idata <- read_repertoires( + path = "", + schema = c("cdr3_aa", "v_call"), + manifest = manifest_df, + repertoire_schema = "Therapy", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL + ) + + expect_true(!is.null(idata$repertoires)) + + if (!is.null(idata$repertoires)) { + repertoires <- idata$repertoires |> collect() + expect_gt(nrow(repertoires), 0) + expect_true("Therapy" %in% colnames(repertoires)) + } +}) diff --git a/tests/testthat/test-io-immundata-read.R b/tests/testthat/test-io-repertoires-schema-bulk.R similarity index 100% rename from tests/testthat/test-io-immundata-read.R rename to tests/testthat/test-io-repertoires-schema-bulk.R diff --git a/tests/testthat/test-io-repertoires-schema-paired.R b/tests/testthat/test-io-repertoires-schema-paired.R new file mode 100644 index 0000000..1edbe6b --- /dev/null +++ b/tests/testthat/test-io-repertoires-schema-paired.R @@ -0,0 +1,208 @@ +test_that("Case 3.2a: read_repertoires handles strict pairing correctly", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + schema_features <- c("v_call", "j_call", "junction_aa") + + sample_file <- test_ig_data() + + idata <- read_repertoires( + path = sample_file, + schema = make_receptor_schema( + features = schema_features, + chains = c("IGH", "IGK") + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "duplicate_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + annotations <- idata$annotations |> collect() + receptors <- idata$receptors |> collect() + + # Sanity check + expect_false(nrow(annotations) == 0) + + expect_setequal(colnames(idata$receptors), c(do.call(paste0, expand.grid(c(schema_features, "locus"), c(".x", ".y"))), imd_schema("receptor"))) + + expect_equal(receptors |> select(-imd_receptor_id) |> distinct() |> nrow(), nrow(receptors)) + expect_equal(receptors |> select(imd_receptor_id) |> distinct() |> nrow(), nrow(receptors)) + + chains_stats <- annotations |> + summarise(n_chains = n(), .by = imd_barcode) |> + summarise(n = n(), .by = n_chains) + + # We should see the receptors with two chains only, nothing more, nothing less + expect_setequal(2, chains_stats$n_chains) + + # Each cell should have both IGH and IGK + cell_loci <- annotations |> + group_by(imd_barcode) |> + summarise( + loci = list(sort(unique(locus))), + n_loci = n_distinct(locus), + .groups = "drop" + ) + + # All cells should have exactly 2 loci + expect_true(all(cell_loci$n_loci == 2)) + + # All cells should have both IGH and IGK + res <- cell_loci |> + distinct(loci) |> + pull(loci) + expect_equal(res[[1]], c("IGH", "IGK")) +}) + +test_that("Case 3.2b: read_repertoires() handles relaxed pairing and excludes artifacts", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + # Test relaxed IGH + (IGK|IGL) pairing + schema_features <- c("v_call", "j_call", "junction_aa") + + sample_file <- test_ig_data() + + idata <- read_repertoires( + path = sample_file, + schema = make_receptor_schema( + features = schema_features, + chains = c("IGH", "IGK|IGL") # Relaxed pairing syntax + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "duplicate_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + expect_setequal(colnames(idata$receptors), c(do.call(paste0, expand.grid(c(schema_features, "locus"), c(".x", ".y"))), imd_schema("receptor"))) + + # Sanity check + annotations <- idata$annotations |> collect() + receptors <- idata$receptors |> collect() + + expect_false(nrow(annotations) == 0) + + expect_equal(receptors |> select(-imd_receptor_id) |> distinct() |> nrow(), nrow(receptors)) + expect_equal(receptors |> select(imd_receptor_id) |> distinct() |> nrow(), nrow(receptors)) + + chains_stats <- annotations |> + summarise(n_chains = n(), .by = imd_barcode) |> + summarise(n = n(), .by = n_chains) + + # We should see the receptors with two chains only, nothing more, nothing less + expect_setequal(2, chains_stats$n_chains) + + # Each cell should have both IGH and IGK + cell_loci <- annotations |> + group_by(imd_barcode) |> + summarise( + loci = list(sort(unique(locus))), + n_loci = n_distinct(locus), + .groups = "drop" + ) + + # All cells should have exactly 2 loci + expect_true(all(cell_loci$n_loci == 2)) + + # All cells should have both IGH and IGK + res <- cell_loci |> + distinct(loci) |> + pull(loci) + expect_setequal(res, list(c("IGH", "IGK"), c("IGH", "IGL"))) + + # Identify artifact cells from original data (those with IGH + IGK + IGL) + original_data <- readr::read_tsv(sample_file, show_col_types = FALSE) + artifact_cells <- original_data |> + group_by(cell_id) |> + summarise( + has_igh = "IGH" %in% locus, + has_igk = "IGK" %in% locus, + has_igl = "IGL" %in% locus, + .groups = "drop" + ) |> + filter(!((has_igh & has_igk & !has_igl) | (has_igh & !has_igk & has_igl))) |> + pull(cell_id) + + cells_in_result <- unique(annotations$imd_barcode) + + # Artifact cells should NOT be in the result + expect_false(any(artifact_cells %in% cells_in_result)) +}) + +test_that("read_repertoires handles duplicate paired-chain entries correctly", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + test_data <- make_duplicate_chain_test_data() + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + schema_features <- c("v_call", "j_call", "junction_aa") + + idata_strict <- read_repertoires( + path = temp_file, + schema = make_receptor_schema(features = schema_features, chains = c("IGH", "IGL")), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + annotations <- idata_strict$annotations |> collect() + + # Should select highest UMI count for each locus + igh_chains <- annotations |> filter(locus == "IGH") + igl_chains <- annotations |> filter(locus == "IGL") + + # Should have selected the chains with highest UMI + expect_equal(igh_chains$junction_aa, "CBRW") # 150 UMI + expect_equal(igl_chains$junction_aa, "CASL") # 80 UMI +}) + +test_that("read_repertoires handles tied max UMI counts for paired chains", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + test_data <- make_duplicate_chain_test_data(tied = TRUE) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + schema_features <- c("v_call", "j_call", "junction_aa") + + idata_strict <- read_repertoires( + path = temp_file, + schema = make_receptor_schema(features = schema_features, chains = c("IGH", "IGL")), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + annotations <- idata_strict$annotations |> collect() + + # Should select highest UMI count for each locus + igh_chains <- annotations |> filter(locus == "IGH") + igl_chains <- annotations |> filter(locus == "IGL") + + # Should have selected the chains with highest UMI + expect_equal(igh_chains$junction_aa, "CARW") # first one + expect_equal(igl_chains$junction_aa, "CASL") # first one as well +}) diff --git a/tests/testthat/test-io-repertoires-schema-single.R b/tests/testthat/test-io-repertoires-schema-single.R new file mode 100644 index 0000000..a449398 --- /dev/null +++ b/tests/testthat/test-io-repertoires-schema-single.R @@ -0,0 +1,120 @@ +test_that("Case 3.1: read_repertoires() handles single chain correctly", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + schema_features <- c("v_call", "j_call", "junction_aa") + + sample_file <- test_ig_data() + + idata <- read_repertoires( + path = sample_file, + schema = make_receptor_schema( + features = schema_features, + chains = "IGH" + ), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "duplicate_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + # Tests + expect_s3_class(idata, "ImmunData") + + annotations <- idata$annotations |> collect() + receptors <- idata$receptors |> collect() + + expect_false(nrow(annotations) == 0) + + expect_equal(receptors |> select(-imd_receptor_id) |> distinct() |> nrow(), nrow(receptors)) + expect_equal(receptors |> select(imd_receptor_id) |> distinct() |> nrow(), nrow(receptors)) + + expect_equal(unique(annotations$locus), "IGH") + + # Verify all valid IGH-containing cells from original data are represented + cells_in_result <- unique(annotations$imd_barcode) + original_data <- readr::read_tsv(sample_file, show_col_types = FALSE) + + cells_with_igh <- original_data |> + group_by(cell_id) |> + summarise( + has_igh = "IGH" %in% locus, + has_igk = "IGK" %in% locus, + has_igl = "IGL" %in% locus, + .groups = "drop" + ) |> + filter(has_igh) |> + pull(cell_id) |> + unique() + + expect_setequal(cells_in_result, cells_with_igh) +}) + +test_that("read_repertoires handles duplicate single-chain entries correctly", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + test_data <- make_duplicate_chain_test_data() + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + schema_features <- c("v_call", "j_call", "junction_aa") + + idata_strict <- read_repertoires( + path = temp_file, + schema = make_receptor_schema(features = schema_features, chains = c("IGH")), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + annotations <- idata_strict$annotations |> collect() + + # Should select highest UMI count for each locus + igh_chains <- annotations |> filter(locus == "IGH") + + # Should have selected the chains with highest UMI + expect_equal(igh_chains$junction_aa, "CBRW") # 150 UMI +}) + +test_that("read_repertoires handles tied max UMI counts for a single chain", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + test_data <- make_duplicate_chain_test_data(tied = TRUE) + + temp_file <- tempfile(fileext = ".tsv") + readr::write_tsv(test_data, temp_file) + on.exit(unlink(temp_file), add = TRUE) + + schema_features <- c("v_call", "j_call", "junction_aa") + + idata_strict <- read_repertoires( + path = temp_file, + schema = make_receptor_schema(features = schema_features, chains = c("IGH")), + barcode_col = "cell_id", + locus_col = "locus", + umi_col = "umi_count", + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL + ) + + annotations <- idata_strict$annotations |> collect() + + # Should select highest UMI count for each locus + igh_chains <- annotations |> filter(locus == "IGH") + + # Should have selected the chains with highest UMI + expect_equal(igh_chains$junction_aa, "CARW") # first one +}) diff --git a/tests/testthat/test-io-repertoires-schema-table.R b/tests/testthat/test-io-repertoires-schema-table.R new file mode 100644 index 0000000..1be3144 --- /dev/null +++ b/tests/testthat/test-io-repertoires-schema-table.R @@ -0,0 +1,30 @@ +test_that("Case 1: read_repertoires() handles table case correctly", { + small_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + # Create a fresh temp folder + outdir <- file.path(tempdir(), "test-no-barcodes") + dir.create(outdir, showWarnings = FALSE) + + schema_features <- c("cdr3_aa", "v_call") + + idata <- read_repertoires( + path = small_file, + schema = schema_features, + output_folder = outdir + ) + + annotations <- idata$annotations |> collect() + receptors <- idata$receptors |> collect() + + expect_false(nrow(annotations) == 0) + + expect_setequal(colnames(idata$receptors), c(schema_features, imd_schema("receptor"))) + + expect_equal(receptors |> select(-imd_receptor_id) |> distinct() |> nrow(), nrow(receptors)) + expect_equal(receptors |> select(imd_receptor_id) |> distinct() |> nrow(), nrow(receptors)) + + checkmate::expect_r6(idata, classes = "ImmunData") + + expect_true(file.exists(file.path(outdir, imd_files()$metadata))) + expect_true(file.exists(file.path(outdir, imd_files()$annotations))) +}) diff --git a/tests/testthat/test-io-repertoires-schema.R b/tests/testthat/test-io-repertoires-schema.R new file mode 100644 index 0000000..e8be987 --- /dev/null +++ b/tests/testthat/test-io-repertoires-schema.R @@ -0,0 +1,211 @@ +test_that("read_repertoires() errors when both barcode_col and count_col are set", { + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + expect_error( + read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + barcode_col = "barcode", + count_col = "count_col" # Not actually in the file, but we want the code path tested + ), + "either .*barcode_col.*count_col" + ) +}) + +test_that("read_repertoires() fails if missing columns in the receptor schema", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + bad_schema <- c("cdr3_aa", "v_call", "some_missing_col") + + expect_error( + read_repertoires( + path = sample_file, + schema = bad_schema, + output_folder = output_dir + ), + "Missing receptor feature column\\(s\\)" + ) +}) + +test_that("read_repertoires() fails with a clear message when barcode column is missing", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + expect_error( + read_repertoires( + path = sample_file, + schema = make_receptor_schema( + features = c("v_call", "j_call", "junction_aa"), + chains = "IGH" + ), + barcode_col = "missing_barcode_col", + locus_col = "locus", + umi_col = "counts", + output_folder = output_dir + ), + "Missing column\\(s\\) referenced by arguments: \\[missing_barcode_col\\]" + ) +}) + +test_that("read_repertoires() fails with a clear message when manifest file column is missing", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + manifest <- data.frame( + WrongFileCol = "some/path.tsv", + stringsAsFactors = FALSE + ) + + expect_error( + read_repertoires( + path = "", + manifest = manifest, + manifest_file_col = "file", + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir + ), + "has no column" + ) +}) + +test_that("read_repertoires() fails when single-cell mode is requested without umi_col", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + expect_error( + read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + barcode_col = "sequence_id", + output_folder = output_dir + ), + "Single-cell mode requires .*umi_col" + ) +}) + +test_that("read_repertoires() fails when chained schema is passed without locus_col", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + expect_error( + read_repertoires( + path = sample_file, + schema = make_receptor_schema( + features = c("cdr3_aa", "v_call"), + chains = "IGH" + ), + output_folder = output_dir + ), + "locus_col.*NULL" + ) +}) + +test_that("read_repertoires() fails when paired schema is passed without barcode_col", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + expect_error( + read_repertoires( + path = sample_file, + schema = make_receptor_schema( + features = c("cdr3_aa", "v_call"), + chains = c("IGH", "IGL") + ), + locus_col = "locus", + output_folder = output_dir + ), + "barcode_col.*NULL" + ) +}) + +test_that("read_repertoires() fails with a clear message when count_col is missing", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + expect_error( + read_repertoires( + path = sample_file, + schema = c("cdr3_aa", "v_call"), + count_col = "missing_count_col", + output_folder = output_dir + ), + "Missing column\\(s\\) referenced by arguments: \\[missing_count_col\\]" + ) +}) + +test_that("read_repertoires() fails with a clear message when locus_col is missing", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + expect_error( + read_repertoires( + path = sample_file, + schema = make_receptor_schema( + features = c("cdr3_aa", "v_call"), + chains = "IGH" + ), + barcode_col = "sequence_id", + locus_col = "missing_locus_col", + umi_col = "counts", + output_folder = output_dir + ), + "Missing column\\(s\\) referenced by arguments: \\[missing_locus_col\\]" + ) +}) + +test_that("read_repertoires() fails with a clear message when umi_col is missing", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") + + expect_error( + read_repertoires( + path = sample_file, + schema = make_receptor_schema( + features = c("cdr3_aa", "v_call"), + chains = "IGH" + ), + barcode_col = "sequence_id", + locus_col = "locus", + umi_col = "missing_umi_col", + output_folder = output_dir + ), + "Missing column\\(s\\) referenced by arguments: \\[missing_umi_col\\]" + ) +}) + +test_that("read_repertoires() fails when manifest paths are empty or NA", { + output_dir <- create_test_output_dir() + on.exit(cleanup_output_dir(output_dir)) + + manifest <- data.frame( + file = c("", NA), + stringsAsFactors = FALSE + ) + + expect_error( + read_repertoires( + path = "", + manifest = manifest, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir + ), + "contains empty/NA paths" + ) +}) diff --git a/tests/testthat/test-io-repertoires.R b/tests/testthat/test-io-repertoires.R deleted file mode 100644 index 35cf076..0000000 --- a/tests/testthat/test-io-repertoires.R +++ /dev/null @@ -1,213 +0,0 @@ -test_that("read_repertoires() fails if path doesn't exist", { - expect_error( - read_repertoires(path = "nonexistent_file.tsv", schema = c("cdr3_aa", "v_call")), - "No file provided" - ) -}) - -test_that("read_repertoires() works with sample data and merges metadata", { - # Load example data shipped with your package - md_path <- system.file("extdata/tsv", "metadata.tsv", package = "immundata") - sample_files <- c( - system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata"), - system.file("extdata/tsv", "sample_1k_2k.tsv", package = "immundata") - ) - - # This function presumably reads the metadata file - # (If you have a 'read_metadata()' or 'load_metadata()' function.) - metadata_df <- read_metadata(md_path) - - outdir <- tempdir() - - # Run the function - imdata <- read_repertoires( - path = sample_files, - schema = c("cdr3_aa", "v_call"), - metadata = metadata_df, - output_folder = outdir - ) - - # Basic check: Did we get an object back? - expect_true(!is.null(imdata)) - - checkmate::expect_r6(imdata, classes = "ImmunData") - - # Check if the output files exist - expect_true(file.exists(file.path(outdir, imd_files()$metadata))) - expect_true(file.exists(file.path(outdir, imd_files()$annotations))) -}) - -test_that("read_repertoires() works with ", { - # Load example data shipped with your package - md_path <- system.file("extdata/tsv", "metadata.tsv", package = "immundata") - - # This function presumably reads the metadata file - # (If you have a 'read_metadata()' or 'load_metadata()' function.) - metadata_df <- read_metadata(md_path) - - outdir <- tempdir() - - # Run the function - imdata <- read_repertoires( - path = "", - schema = c("cdr3_aa", "v_call"), - metadata = metadata_df, - output_folder = outdir - ) - - # Basic check: Did we get an object back? - expect_true(!is.null(imdata)) - - checkmate::expect_r6(imdata, classes = "ImmunData") - - # Check if the output files exist - expect_true(file.exists(file.path(outdir, imd_files()$metadata))) - expect_true(file.exists(file.path(outdir, imd_files()$annotations))) -}) - -test_that("read_repertoires() case 1: no barcode_col and no count_col", { - # Provide test data that doesn't have barcodes or counts - # e.g. a minimal TSV with just 'cdr3_aa' and 'v_call' - small_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") - - # Create a fresh temp folder - outdir <- file.path(tempdir(), "test-no-barcodes") - dir.create(outdir, showWarnings = FALSE) - - # If you want to ensure no barcodes: - # We'll skip 'cell_id_col' and 'count_col' - imdata <- read_repertoires( - path = small_file, - schema = c("cdr3_aa", "v_call"), - output_folder = outdir - ) - - checkmate::expect_r6(imdata, classes = "ImmunData") - - # Check that the function didn't crash and files were saved - expect_true(file.exists(file.path(outdir, imd_files()$metadata))) - expect_true(file.exists(file.path(outdir, imd_files()$annotations))) - - # You can do further checks on the resulting annotation columns, etc. -}) - -test_that("read_repertoires() errors when both barcode_col and count_col are set", { - # Provide minimal data but pass both arguments to see if it triggers the expected error - sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") - - expect_error( - read_repertoires( - path = sample_file, - schema = c("cdr3_aa", "v_call"), - barcode_col = "barcode", - count_col = "count_col" # Not actually in the file, but we want the code path tested - ), - "Undefined case" - ) -}) - -test_that("read_repertoires() excludes specified columns", { - sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") - exclude_cols <- c("sequence", "fwr1", "cdr1") - - imdata <- read_repertoires( - path = sample_file, - schema = c("cdr3_aa", "v_call"), # columns that do exist - preprocess = list( - exclude_columns = make_exclude_columns(cols = exclude_cols) - ), - output_folder = file.path(tempdir(), "test-exclude") - ) - - # If `annotations` is an R6 active binding (and not a function call), use: - ann_cols <- colnames(imdata$annotations) - - # Check that the excluded columns are not present - for (col in exclude_cols) { - expect_false( - col %in% ann_cols, - info = paste("Column", col, "should have been excluded but is still present.") - ) - } -}) - -test_that("read_repertoires() correctly renames columns (v_call -> v_gene)", { - sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") - - # We assume the file actually contains 'v_call' - # We'll rename 'v_call' to 'v_gene' - rename_map <- c("v_gene" = "v_call") # new_col = old_col - - imdata <- read_repertoires( - path = sample_file, - schema = c("cdr3_aa", "v_gene"), # We still rely on the old col name for grouping - rename_columns = rename_map, - output_folder = file.path(tempdir(), "test-rename") - ) - - # Access the annotation columns - ann_cols <- colnames(imdata$annotations) - - expect_true( - "v_gene" %in% ann_cols, - info = "Renamed column 'v_call' -> 'v_gene' should appear in the annotation." - ) - expect_false( - "v_call" %in% ann_cols, - info = "Original column 'v_call' should be removed after rename." - ) -}) - -test_that("read_repertoires() excludes columns AND renames simultaneously", { - sample_file <- system.file("extdata/tsv", "sample_1k_2k.tsv", package = "immundata") - - # Suppose the data has columns "j_call" and we want to rename it to "j_gene" - rename_map <- c("j_gene" = "j_call") - exclude_cols <- c("cdr2", "fwr2") # must exist in sample_1k_2k.tsv for the test to pass - - imdata <- read_repertoires( - path = sample_file, - schema = c("cdr3_aa", "v_call", "j_gene"), - preprocess = list( - exclude_columns = make_exclude_columns(cols = exclude_cols) - ), - rename_columns = rename_map, - output_folder = file.path(tempdir(), "test-exclude-rename") - ) - - ann_cols <- colnames(imdata$annotations) - - # Check exclusion - for (col in exclude_cols) { - expect_false( - col %in% ann_cols, - info = paste("Column", col, "should have been excluded.") - ) - } - - # Check rename - expect_true( - "j_gene" %in% ann_cols, - info = "Renamed column 'j_call' -> 'j_gene' should appear." - ) - expect_false( - "j_call" %in% ann_cols, - info = "Original column 'j_call' should be gone." - ) -}) - -test_that("read_repertoires() fails if missing columns in the receptor schema", { - # Provide a sample input file known to have certain columns (like "cdr3_aa" and "v_call"). - sample_file <- system.file("extdata/tsv", "sample_0_1k.tsv", package = "immundata") - - # We intentionally add a column ("some_missing_col") that doesn't exist in the file - bad_schema <- c("cdr3_aa", "v_call", "some_missing_col") - - expect_error( - read_repertoires( - path = sample_file, - schema = bad_schema - ), - "Not all columns in the receptor schema present in the data" - ) -}) diff --git a/tests/testthat/test-mutate-immundata.R b/tests/testthat/test-mutate-immundata.R new file mode 100644 index 0000000..282bf8f --- /dev/null +++ b/tests/testthat/test-mutate-immundata.R @@ -0,0 +1,452 @@ +make_mutate_test_idata <- function() { + ImmunData$new( + schema = c("cdr3_aa", "v_call"), + annotations = make_basic_test_annotations(), + provenance = list( + home_path = tempdir(), + current_path = tempdir(), + snapshot_id = "mutate-test-snapshot", + lineage = list(list(event = "fixture")) + ) + ) +} + +make_grouped_mutate_test_idata <- function() { + annotations <- tibble::tibble( + imd_receptor_id = c(1L, 1L, 2L, 3L, 4L), + imd_barcode = paste0("bc", seq_len(5L)), + imd_chain_id = seq_len(5L), + imd_n_chains = 1L, + cdr3_aa = c("AAA", "AAA", "BBB", "CCC", "DDD"), + group = c("A", "A", "A", "B", "B"), + batch = c("x", "x", "y", "x", "x"), + value = c(1, 3, 5, 10, 14) + ) |> + duckplyr::as_duckdb_tibble(prudence = "stingy") + + ImmunData$new( + schema = "cdr3_aa", + annotations = annotations + ) +} + +test_that("mutate_immundata adds derived annotation columns without changing input", { + idata <- make_mutate_test_idata() + + out <- mutate_immundata( + idata, + cdr3_len = nchar(cdr3_aa), + receptor_label = paste(v_call, cdr3_aa, sep = ":") + ) + + expect_s3_class(out, "ImmunData") + expect_false("cdr3_len" %in% names(idata$annotations)) + + ann <- out$annotations |> + collect() |> + arrange(imd_receptor_id) + + expect_equal(ann$cdr3_len, nchar(ann$cdr3_aa)) + expect_equal(ann$receptor_label, paste(ann$v_call, ann$cdr3_aa, sep = ":")) +}) + +test_that("dplyr mutate method and mutate_immundata produce equivalent annotations", { + idata <- make_mutate_test_idata() + + direct <- mutate_immundata(idata, cdr3_len = nchar(cdr3_aa)) + s3 <- dplyr::mutate(idata, cdr3_len = nchar(cdr3_aa)) + + expect_equal( + direct$annotations |> collect() |> arrange(imd_receptor_id), + s3$annotations |> collect() |> arrange(imd_receptor_id) + ) +}) + +test_that("grouped mutate forwards .by without creating a .by column", { + idata <- make_grouped_mutate_test_idata() + + out <- idata |> + mutate( + centered = value - mean(value, na.rm = TRUE), + above_mean = value > mean(value, na.rm = TRUE), + .by = group + ) + + expect_false(".by" %in% names(out$annotations)) + expect_s3_class(out$annotations, "prudent_duckplyr_df") + + ann <- out |> + collect() |> + arrange(imd_chain_id) + + expect_equal(ann$centered, c(-2, 0, 2, -2, 2)) + expect_equal(ann$above_mean, c(FALSE, FALSE, TRUE, FALSE, TRUE)) +}) + +test_that("grouped mutate falls back once for independent group summaries", { + idata <- make_grouped_mutate_test_idata() + + expect_error( + idata$annotations |> + mutate(group_n_receptors = n_distinct(imd_receptor_id), .by = group), + "not supported in window functions", + fixed = TRUE + ) + + out <- idata |> + mutate( + group_n_rows = n(), + group_n_receptors = n_distinct(imd_receptor_id), + group_max = max(value), + .by = group + ) + + expect_s3_class(out$annotations, "prudent_duckplyr_df") + expect_error( + nrow(out$annotations), + "Materialization is disabled", + fixed = TRUE + ) + + ann <- out |> + collect() |> + arrange(imd_chain_id) + + expect_equal(ann$group_n_rows, c(3, 3, 3, 2, 2)) + expect_equal(ann$group_n_receptors, c(2, 2, 2, 2, 2)) + expect_equal(ann$group_max, c(5, 5, 5, 14, 14)) +}) + +test_that("group summary fallback supports multiple and missing group values", { + idata <- make_grouped_mutate_test_idata() + + multiple <- idata |> + mutate( + group_n_receptors = n_distinct(imd_receptor_id), + .by = c(group, batch) + ) |> + collect() |> + arrange(imd_chain_id) + + expect_equal(multiple$group_n_receptors, c(1, 1, 1, 2, 2)) + + missing_groups <- ImmunData$new( + schema = "cdr3_aa", + annotations = tibble::tibble( + imd_receptor_id = c(1L, 2L, 3L, 3L), + imd_barcode = paste0("bc", seq_len(4L)), + imd_chain_id = seq_len(4L), + imd_n_chains = 1L, + cdr3_aa = c("AAA", "BBB", "CCC", "CCC"), + group = c("A", "A", NA, NA) + ) |> + duckplyr::as_duckdb_tibble(prudence = "stingy") + ) |> + mutate( + group_n_receptors = n_distinct(imd_receptor_id), + .by = group + ) |> + collect() |> + arrange(imd_chain_id) + + expect_equal(missing_groups$group_n_receptors, c(2, 2, 1, 1)) +}) + +test_that("group summary fallback can replace a non-protected column", { + idata <- make_grouped_mutate_test_idata() + + out <- idata |> + mutate(value = n_distinct(imd_receptor_id), .by = group) + + expect_equal(names(out$annotations), names(idata$annotations)) + + ann <- out |> + collect() |> + arrange(imd_chain_id) + + expect_equal(ann$value, rep(2, 5)) +}) + +test_that("group summary fallback does not hide unrelated errors", { + idata <- make_grouped_mutate_test_idata() + + expect_error( + idata |> + mutate(result = no_such_function(value), .by = group), + "Can't translate function `no_such_function()`.", + fixed = TRUE + ) + expect_error( + idata |> + mutate(result = absent + 1, .by = group), + "object 'absent' not found", + fixed = TRUE + ) + expect_error( + idata |> + mutate(result = n(), .by = absent), + "Column `absent` doesn't exist", + fixed = TRUE + ) +}) + +test_that("mixed row and fallback calculations can be split across mutate calls", { + idata <- make_grouped_mutate_test_idata() + + expect_error( + idata |> + mutate( + centered = value - mean(value, na.rm = TRUE), + group_n_receptors = n_distinct(imd_receptor_id), + .by = group + ), + "not supported in window functions", + fixed = TRUE + ) + + out <- idata |> + mutate( + centered = value - mean(value, na.rm = TRUE), + .by = group + ) |> + mutate( + group_n_receptors = n_distinct(imd_receptor_id), + .by = group + ) |> + collect() |> + arrange(imd_chain_id) + + expect_equal(out$centered, c(-2, 0, 2, -2, 2)) + expect_equal(out$group_n_receptors, rep(2, 5)) +}) + +test_that("fallback summary dependencies can be split across mutate calls", { + idata <- make_grouped_mutate_test_idata() + + expect_error( + idata |> + mutate( + group_n_receptors = n_distinct(imd_receptor_id), + twice_group_n_receptors = group_n_receptors * 2, + .by = group + ), + "not supported in window functions", + fixed = TRUE + ) + + out <- idata |> + mutate( + group_n_receptors = n_distinct(imd_receptor_id), + .by = group + ) |> + mutate(twice_group_n_receptors = group_n_receptors * 2) |> + collect() |> + arrange(imd_chain_id) + + expect_equal(out$group_n_receptors, rep(2, 5)) + expect_equal(out$twice_group_n_receptors, rep(4, 5)) +}) + +test_that("mutate_immundata blocks system column writes", { + idata <- make_mutate_test_idata() + + expect_error( + mutate_immundata(idata, imd_receptor_id = 1L), + "system columns" + ) + expect_error( + mutate_immundata(idata, imd_barcode = "x"), + "system columns" + ) + expect_error( + mutate_immundata(idata, imd_chain_id = 1L), + "system columns" + ) +}) + +test_that("mutate_immundata blocks receptor and repertoire schema writes", { + idata <- make_mutate_test_idata() |> + agg_repertoires("sample_id") + + expect_error( + mutate_immundata(idata, cdr3_aa = "changed"), + "schema columns.*cdr3_aa" + ) + expect_error( + mutate_immundata(idata, sample_id = "changed"), + "schema columns.*sample_id" + ) +}) + +test_that("mutate_immundata blocks generated sequence schema writes", { + idata <- make_mutate_test_idata() + collision_idata <- ImmunData$new( + schema = c("cdr3_aa", "imd_sim_exact_1"), + annotations = idata$annotations |> + dplyr::mutate(imd_sim_exact_1 = 0L) + ) + + expect_error( + mutate_immundata( + collision_idata, + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = "AAA", + method = "exact" + ) + ), + "schema columns.*imd_sim_exact_1" + ) +}) + +test_that("mutate_immundata preserves repertoire, strata, and provenance state", { + idata <- make_mutate_test_idata() |> + agg_repertoires("sample_id") |> + agg_strata("sample_id") + + reps_before <- idata$repertoires + strata_before <- idata$strata + prov_before <- get_provenance(idata) + annotation_state_before <- idata$annotations |> + select(imd_receptor_id, imd_repertoire_id, imd_strata_id) |> + collect() |> + arrange(imd_receptor_id) + + out <- mutate_immundata(idata, cohort = "all") + + expect_equal(out$repertoires, reps_before) + expect_equal(out$strata, strata_before) + expect_equal(out$schema_repertoire, idata$schema_repertoire) + expect_equal(out$schema_strata, idata$schema_strata) + expect_equal( + out$annotations |> + select(imd_receptor_id, imd_repertoire_id, imd_strata_id) |> + collect() |> + arrange(imd_receptor_id), + annotation_state_before + ) + expect_equal( + get_provenance(out)[sort(names(get_provenance(out)))], + prov_before[sort(names(prov_before))] + ) +}) + +test_that("mutate_immundata supports repertoire-free and strata-free state", { + annotations_only <- make_mutate_test_idata() + annotations_only_out <- mutate_immundata(annotations_only, cohort = "all") + + expect_null(annotations_only_out$repertoires) + expect_null(annotations_only_out$strata) + expect_null(annotations_only_out$schema_repertoire) + expect_null(annotations_only_out$schema_strata) + + repertoires_only <- annotations_only |> + agg_repertoires("sample_id") + repertoires_only_out <- mutate_immundata(repertoires_only, cohort = "all") + + expect_equal(repertoires_only_out$repertoires, repertoires_only$repertoires) + expect_equal( + repertoires_only_out$schema_repertoire, + repertoires_only$schema_repertoire + ) + expect_null(repertoires_only_out$strata) + expect_null(repertoires_only_out$schema_strata) +}) + +test_that("mutate_immundata supports exact sequence annotations", { + idata <- make_mutate_test_idata() + + out <- mutate_immundata( + idata, + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = c("AAA", "BBB"), + method = "exact" + ) + ) + + ann <- out$annotations |> + collect() |> + arrange(imd_receptor_id) + + expect_true(all(c("imd_sim_exact_1", "imd_sim_exact_2") %in% names(ann))) + expect_equal(ann$imd_sim_exact_1, ann$cdr3_aa == "AAA") + expect_equal(ann$imd_sim_exact_2, ann$cdr3_aa == "BBB") +}) + +test_that("mutate_immundata supports pattern-based sequence annotation names", { + idata <- make_mutate_test_idata() + + out <- mutate_immundata( + idata, + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = "A-A", + method = "exact", + name_type = "pattern" + ) + ) + + expect_true("imd_sim_exact_A_A" %in% names(out$annotations)) +}) + +test_that("mutate_immundata supports regex sequence annotations", { + idata <- make_mutate_test_idata() + + out <- mutate_immundata( + idata, + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = "^AA", + method = "regex" + ) + ) + + ann <- out$annotations |> + collect() |> + arrange(imd_receptor_id) + + expect_equal(ann$imd_sim_regex_1, grepl("^AA", ann$cdr3_aa)) +}) + +test_that("mutate_immundata supports Levenshtein and Hamming distance annotations", { + idata <- make_mutate_test_idata() + + lev <- mutate_immundata( + idata, + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = "AAA", + method = "lev" + ) + )$annotations |> + collect() |> + arrange(imd_receptor_id) + + hamm <- mutate_immundata( + idata, + seq_options = make_seq_options( + query_col = "cdr3_aa", + patterns = "AAA", + method = "hamm" + ) + )$annotations |> + collect() |> + arrange(imd_receptor_id) + + expect_equal(lev$imd_sim_lev_1, c(0, 1, 1, 3)) + expect_equal(hamm$imd_sim_hamm_1, c(0, 1, NA, 3)) +}) + +test_that("mutate_immundata validates seq_options", { + idata <- make_mutate_test_idata() + + expect_error( + mutate_immundata(idata, seq_options = list(patterns = "AAA")), + "Missing fields" + ) + expect_error( + mutate_immundata(idata, seq_options = list(query_col = "cdr3_aa")), + "Missing fields" + ) +}) diff --git a/tests/testthat/test-verbosity.R b/tests/testthat/test-verbosity.R new file mode 100644 index 0000000..6624803 --- /dev/null +++ b/tests/testthat/test-verbosity.R @@ -0,0 +1,92 @@ +test_that("read_repertoires can run quietly", { + input_file <- system.file( + "extdata/tsv", + "sample_0_1k.tsv", + package = "immundata" + ) + output_dir <- create_test_output_dir("quiet_read_") + on.exit(cleanup_output_dir(output_dir), add = TRUE) + + expect_silent( + idata <- read_repertoires( + path = input_file, + schema = c("cdr3_aa", "v_call"), + output_folder = output_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL, + verbose = FALSE + ) + ) + + expect_s3_class(idata$annotations, "duckplyr_df") +}) + +test_that("the package verbosity option is used by default", { + old_options <- options(immundata.verbose = FALSE) + on.exit(options(old_options), add = TRUE) + + dataset <- duckplyr::duckdb_tibble(data.frame( + cdr3_aa = c("CASSA", "CASSB") + )) + + expect_silent( + result <- agg_receptors(dataset, schema = "cdr3_aa") + ) + + expect_s3_class(result, "duckplyr_df") +}) + +test_that("verbose output remains the default behavior", { + dataset <- duckplyr::duckdb_tibble(data.frame( + cdr3_aa = c("CASSA", "CASSB") + )) + + messages <- capture_messages( + agg_receptors(dataset, schema = "cdr3_aa", verbose = TRUE) + ) + + expect_match(paste(messages, collapse = "\n"), "No locus information found") +}) + +test_that("manifest and snapshot I/O can run quietly", { + manifest_path <- system.file( + "extdata/tsv", + "manifest.csv", + package = "immundata" + ) + + expect_silent( + manifest <- read_manifest(manifest_path, verbose = FALSE) + ) + + input_file <- manifest$file[[1]] + root_dir <- create_test_output_dir("quiet_io_root_") + snapshot_dir <- create_test_output_dir("quiet_io_snapshot_") + on.exit(cleanup_output_dir(root_dir), add = TRUE) + on.exit(cleanup_output_dir(snapshot_dir), add = TRUE) + + idata <- read_repertoires( + path = input_file, + schema = c("cdr3_aa", "v_call"), + output_folder = root_dir, + preprocess = NULL, + postprocess = NULL, + rename_columns = NULL, + verbose = FALSE + ) + + expect_silent( + written <- write_immundata( + idata, + output_folder = snapshot_dir, + verbose = FALSE + ) + ) + expect_silent( + loaded <- read_immundata(snapshot_dir, verbose = FALSE) + ) + + expect_s3_class(written$annotations, "duckplyr_df") + expect_s3_class(loaded$annotations, "duckplyr_df") +})