diff --git a/CRAN-SUBMISSION b/CRAN-SUBMISSION index 779d2b1..8ba0b0a 100644 --- a/CRAN-SUBMISSION +++ b/CRAN-SUBMISSION @@ -1,3 +1,3 @@ -Version: 0.0.0.9002 -Date: 2025-04-07 17:20:29 UTC -SHA: a5d940843126c3de1efe60c349dde4c586afba28 +Version: 0.0.3 +Date: 2025-09-04 14:29:33 UTC +SHA: 66d7dcd409a24d5e1a6a5e6e04c7ae8b9257a8a2 diff --git a/DESCRIPTION b/DESCRIPTION index 91ccfed..e3a88c3 100644 --- a/DESCRIPTION +++ b/DESCRIPTION @@ -1,6 +1,6 @@ Package: immundata Title: A Unified Data Layer for Large-Scale Single-Cell, Spatial and Bulk Immunomics -Version: 0.0.4.9000 +Version: 0.0.5 Authors@R: person("Vadim I.", "Nazarov", , "support@immunomind.com", role = c("aut", "cre"), comment = c(ORCID = "0000-0003-3659-2709")) @@ -10,11 +10,11 @@ Description: Provides a unified data layer for single-cell, spatial and bulk but for AIRR data, a.k.a. Adaptive Immune Receptor Repertoire, VDJ-seq, RepSeq, or VDJ sequencing data. License: Apache License (>= 2) -URL: https://immunomind.com/, https://github.com/immunomind/immundata, https://immunomind.github.io/immundata/ +URL: https://immunomind.github.io/docs/, https://github.com/immunomind/immundata BugReports: https://github.com/immunomind/immundata/issues Encoding: UTF-8 Roxygen: list(markdown = TRUE) -RoxygenNote: 7.3.2 +RoxygenNote: 7.3.3 Depends: R (>= 4.1.0), dplyr, @@ -35,6 +35,7 @@ Imports: utils Suggests: rmarkdown, - testthat (>= 3.0.0) + testthat (>= 3.0.0), + Seurat Config/testthat/edition: 3 Config/testthat/parallel: true diff --git a/NAMESPACE b/NAMESPACE index fd78cdb..790a485 100644 --- a/NAMESPACE +++ b/NAMESPACE @@ -12,6 +12,7 @@ export(annotate_barcodes) export(annotate_chains) export(annotate_immundata) export(annotate_receptors) +export(annotate_seurat) export(assert_receptor_schema) export(filter_barcodes) export(filter_immundata) diff --git a/R/core_immundata.R b/R/core_immundata.R index 348e3f3..bcd194f 100644 --- a/R/core_immundata.R +++ b/R/core_immundata.R @@ -101,7 +101,7 @@ ImmunData <- R6Class( private$.annotations }, - #' @field repertoires Get a vector of repertoire names after data aggregation with [agg_repertoires()] + #' @field repertoires Get a table of repertoires and their basic statistics. repertoires = function() { # TODO: cache repertoire table to memory if not very big? if (!is.null(private$.repertoire_table)) { @@ -111,6 +111,18 @@ ImmunData <- R6Class( } else { NULL } + }, + + #' @field metadata Get a table of repertoires without their basic statistics. + metadata = function() { + if (!is.null(private$.repertoire_table)) { + private$.repertoire_table |> + select(c(imd_schema("repertoire"), self$schema_repertoire)) |> + collect() |> + arrange_at(vars(1)) + } else { + NULL + } } ) ) diff --git a/R/io_immundata_write.R b/R/io_immundata_write.R index a02872a..a822b0a 100644 --- a/R/io_immundata_write.R +++ b/R/io_immundata_write.R @@ -34,8 +34,11 @@ #' along with the schema needed to reconstruct/interpret them. #' #' @return -#' Invisibly returns the input `idata` object. Its primary effect is creating -#' `metadata.json` and `annotations.parquet` files in the `output_folder`. +#' Invisibly returns the input `idata` object, saved to disk. +#' In other words, this allows you to create snapshots of the data in the +#' `output_folder`. Mind that by saving the object, you execute all the +#' stored computations, so this operations can take longer than expected. +#' Read more about snapshots on our website in the ["Concept" section](https://immunomind.github.io/docs/concepts/basics/immutability/). #' #' @seealso [read_immundata()] for loading the saved data, [read_repertoires()] #' which uses this function internally, [ImmunData] class definition. @@ -93,5 +96,5 @@ write_immundata <- function(idata, output_folder) { cli::cli_alert_success("ImmunData files saved to [{output_folder}]") - invisible(idata) + invisible(read_immundata(output_folder)) } diff --git a/R/operations_external_annotate_seurat.R b/R/operations_external_annotate_seurat.R new file mode 100644 index 0000000..a380209 --- /dev/null +++ b/R/operations_external_annotate_seurat.R @@ -0,0 +1,66 @@ +#' @title Annotate a Seurat object from ImmunData (by barcode) +#' +#' @description +#' Copy selected columns from `idata$annotations` to Seurat metadata using +#' the cell barcode. This is the simplest way to transfer data from `immundata` +#' to Seurat object, e.g., for plotting data on UMAP. +#' +#' @param idata An [immundata::ImmunData] object. +#' @param sdata A Seurat object (cells are columns; barcodes are `colnames(sdata)`). +#' @param cols Character vector with column names to transfer from `idata$annotations`. +#' Typical choices: `"clonal_prop_bin"` or `"clonal_rank_bin"`. +#' +#' @return The updated Seurat object with new metadata columns. +#' +#' @seealso +#' [ImmunData], [SeuratObject::AddMetaData] +#' +#' @details +#' See functions `annotate_clonality_rank` and `annotate_clonality_prop` in `immunarch` package. +#' +#' +#' @examples +#' \dontrun{ +#' # After annotating receptors: +#' idata <- annotate_clonality_prop(idata) +#' +#' # Transfer the clonality bin to Seurat and plot: +#' sdata <- annotate_seurat(idata, sdata, cols = "clonal_prop_bin") +#' Seurat::DimPlot(sdata, reduction = "umap", group.by = "clonal_prop_bin", shuffle = TRUE) +#' +#' # Alternative: rank bins +#' idata <- annotate_clonality_rank(idata, bins = c(10, 100)) +#' sdata <- annotate_seurat(idata, sdata, cols = "clonal_rank_bin") +#' } +#' +#' @concept Annotation +#' @export +annotate_seurat <- function(idata, + sdata, + cols) { + checkmate::assert_r6(idata, "ImmunData") + checkmate::assert_class(sdata, classes = "Seurat") + checkmate::assert_character(cols, min.len = 1, any.missing = FALSE) + + ann <- idata$annotations + bcsym <- immundata::imd_schema_sym("barcode") + bcname <- immundata::imd_schema("barcode") + + missing_cols <- setdiff(c(bcname, cols), colnames(ann)) + if (length(missing_cols) > 0) { + rlang::abort( + cli::format_inline("Column(s) {cli::col_cyan(missing_cols)} not found in idata$annotations.") + ) + } + + # TODO: mind that we use distinct() here - there could be edge cases + df <- ann |> + dplyr::select(barcode = !!bcsym, dplyr::all_of(cols)) |> + dplyr::distinct(.data$barcode, .keep_all = TRUE) |> + collect() + + meta <- as.data.frame(df[, cols, drop = FALSE]) + rownames(meta) <- df$barcode + + Seurat::AddMetaData(sdata, metadata = meta) +} diff --git a/R/operations_utils.R b/R/operations_utils.R index 29f4738..50ac670 100644 --- a/R/operations_utils.R +++ b/R/operations_utils.R @@ -129,6 +129,7 @@ annotate_tbl_distance <- function(tbl_data, ) # TODO: Optimize it via SQL instead of cycles - if it is even needed... + # TODO: lump together multiple patterns in batches for (i in seq_along(patterns)) { p <- patterns[[i]] col_name_out <- dist_cols[i] @@ -188,6 +189,7 @@ annotate_tbl_distance <- function(tbl_data, # 4) precompute sequence length before (!) any filtering, on data loading, and don't compute it here # TODO: max dist. Left join - compute. Right join - filter + if (is.na(max_dist)) { uniq <- uniq |> as_duckdb_tibble() diff --git a/README.md b/README.md index 61758a0..2975e4a 100644 --- a/README.md +++ b/README.md @@ -28,9 +28,9 @@
- Tutorials + Tutorials | - API reference + API reference | Ecosystem | @@ -1056,7 +1056,7 @@ ggplot2::ggplot(data = clonal_space_homeo) + geom_col(aes(x = Tissue, y = occupi ## 🧩 Use Cases > [!TIP] -> Tutorial on `immundata` + `immunarch` is available [on the ecosystem website](https://immunomind.github.io/docs/tutorials/single-cell/). +> Tutorial on `immundata` + `immunarch` is available [on the ecosystem website](https://immunomind.github.io/docs/tutorials/single_cell/). > > Read the previous section about the analysis. > diff --git a/cran-comments.md b/cran-comments.md index 04ab8e0..31fb022 100644 --- a/cran-comments.md +++ b/cran-comments.md @@ -1,12 +1,2 @@ ## R CMD check results -0 errors | 0 warnings | 1 note - -* This is another try to submit `immundata`. -I fixed all the notes and warnings related to the package. There are some notes left -pointing to the potential misspellings of the terms in the DESCRIPTION, but those are -correct. - -The CRAN check also points out that there is no `read_parquet_duckdb`, but it passes tests on my machine (I know, I know...), -and I added the reference to the package, and the resultant documentation links work. I hope this is fine. -Thank you! diff --git a/man/ImmunData.Rd b/man/ImmunData.Rd index f84aa8c..bde3378 100644 --- a/man/ImmunData.Rd +++ b/man/ImmunData.Rd @@ -33,7 +33,9 @@ grouped into repertoires. This may include sample-level metadata (e.g., \code{sa \item{\code{annotations}}{Accessor for the annotation-level table (\code{.annotations}).} -\item{\code{repertoires}}{Get a vector of repertoire names after data aggregation with \code{\link[=agg_repertoires]{agg_repertoires()}}} +\item{\code{repertoires}}{Get a table of repertoires and their basic statistics.} + +\item{\code{metadata}}{Get a table of repertoires without their basic statistics.} } \if{html}{\out{}} } diff --git a/man/annotate_seurat.Rd b/man/annotate_seurat.Rd new file mode 100644 index 0000000..917c732 --- /dev/null +++ b/man/annotate_seurat.Rd @@ -0,0 +1,46 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/operations_external_annotate_seurat.R +\name{annotate_seurat} +\alias{annotate_seurat} +\title{Annotate a Seurat object from ImmunData (by barcode)} +\usage{ +annotate_seurat(idata, sdata, cols) +} +\arguments{ +\item{idata}{An \link{ImmunData} object.} + +\item{sdata}{A Seurat object (cells are columns; barcodes are \code{colnames(sdata)}).} + +\item{cols}{Character vector with column names to transfer from \code{idata$annotations}. +Typical choices: \code{"clonal_prop_bin"} or \code{"clonal_rank_bin"}.} +} +\value{ +The updated Seurat object with new metadata columns. +} +\description{ +Copy selected columns from \code{idata$annotations} to Seurat metadata using +the cell barcode. This is the simplest way to transfer data from \code{immundata} +to Seurat object, e.g., for plotting data on UMAP. +} +\details{ +See functions \code{annotate_clonality_rank} and \code{annotate_clonality_prop} in \code{immunarch} package. +} +\examples{ +\dontrun{ +# After annotating receptors: +idata <- annotate_clonality_prop(idata) + +# Transfer the clonality bin to Seurat and plot: +sdata <- annotate_seurat(idata, sdata, cols = "clonal_prop_bin") +Seurat::DimPlot(sdata, reduction = "umap", group.by = "clonal_prop_bin", shuffle = TRUE) + +# Alternative: rank bins +idata <- annotate_clonality_rank(idata, bins = c(10, 100)) +sdata <- annotate_seurat(idata, sdata, cols = "clonal_rank_bin") +} + +} +\seealso{ +\link{ImmunData}, \link[SeuratObject:AddMetaData]{SeuratObject::AddMetaData} +} +\concept{Annotation} diff --git a/man/immundata-package.Rd b/man/immundata-package.Rd index 3df38fb..ca01045 100644 --- a/man/immundata-package.Rd +++ b/man/immundata-package.Rd @@ -6,14 +6,13 @@ \alias{immundata-package} \title{immundata: A Unified Data Layer for Large-Scale Single-Cell, Spatial and Bulk Immunomics} \description{ -Provides a unified data layer for single-cell, spatial and bulk T-cell and B-cell immune receptor repertoire data. Think AnnData or SeuratObject, but for AIRR data. +Provides a unified data layer for single-cell, spatial and bulk T-cell and B-cell immune receptor repertoire data. Think AnnData or SeuratObject, but for AIRR data, a.k.a. Adaptive Immune Receptor Repertoire, VDJ-seq, RepSeq, or VDJ sequencing data. } \seealso{ Useful links: \itemize{ - \item \url{https://immunomind.com/} + \item \url{https://immunomind.github.io/docs/} \item \url{https://github.com/immunomind/immundata} - \item \url{https://immunomind.github.io/immundata/} \item Report bugs at \url{https://github.com/immunomind/immundata/issues} } diff --git a/man/write_immundata.Rd b/man/write_immundata.Rd index f86e0d1..dbf7b08 100644 --- a/man/write_immundata.Rd +++ b/man/write_immundata.Rd @@ -16,8 +16,11 @@ will be written. If the directory does not exist, it will be created recursively.} } \value{ -Invisibly returns the input \code{idata} object. Its primary effect is creating -\code{metadata.json} and \code{annotations.parquet} files in the \code{output_folder}. +Invisibly returns the input \code{idata} object, saved to disk. +In other words, this allows you to create snapshots of the data in the +\code{output_folder}. Mind that by saving the object, you execute all the +stored computations, so this operations can take longer than expected. +Read more about snapshots on our website in the \href{https://immunomind.github.io/docs/concepts/basics/immutability/}{"Concept" section}. } \description{ Serializes the essential components of an \code{ImmunData} object to disk for