From c193132002044cce1ab97b5711d9a07c6329526a Mon Sep 17 00:00:00 2001 From: Daena Rys Date: Mon, 5 Jan 2026 13:31:49 +0200 Subject: [PATCH 1/3] Add rank support to getTop --- R/AllGenerics.R | 5 +++-- R/summaries.R | 22 +++++++++++++++++----- man/summary.Rd | 10 ++++++---- tests/testthat/test-8summaries.R | 8 ++++++++ 4 files changed, 34 insertions(+), 11 deletions(-) diff --git a/R/AllGenerics.R b/R/AllGenerics.R index fba73d8a3..85d010fa1 100644 --- a/R/AllGenerics.R +++ b/R/AllGenerics.R @@ -270,8 +270,9 @@ setGeneric("getUnique", signature = c("x"), function(x, ...) #' @export setGeneric("getTop", signature = "x", function( - x, top= 5L, method = c("mean", "sum", "median"), - assay.type = assay_name, assay_name = "counts", na.rm = TRUE, ...) + x, top= 5L, method = c("mean", "sum", "median", "prevalence"), + assay.type = assay_name, assay_name = "counts", rank = NULL, + na.rm = TRUE, ...) standardGeneric("getTop")) #' @rdname taxonomy-methods diff --git a/R/summaries.R b/R/summaries.R index 8eb54df68..c010a06a6 100644 --- a/R/summaries.R +++ b/R/summaries.R @@ -11,6 +11,11 @@ #' @param method \code{Character scalar}. Specify the method to determine top #' taxa. Either sum, mean, median or prevalence. (Default: \code{"mean"}) #' +#' @param rank \code{Character scalar}. Defines a taxonomic rank. When +#' provided, input is first aggregated with \code{agglomerateByRank()} at the +#' given rank before selecting top features. Must be one of +#' \code{taxonomyRanks(x)}. (Default: \code{NULL}). +#' #' @param ... Additional arguments passed, e.g., to getPrevalence: #' \itemize{ #' \item \code{sort}: \code{Logical scalar}. Specify @@ -106,18 +111,25 @@ NULL setMethod("getTop", signature = c(x = "SummarizedExperiment"), function( x, top = 5L, method = c("mean", "sum", "median", "prevalence"), - assay.type = assay_name, assay_name = "counts", + assay.type = assay_name, assay_name = "counts", rank = NULL, na.rm = TRUE, ...){ + dots <- list(...) # input check method <- match.arg(method, c("mean","sum","median","prevalence")) + if (!is.null(rank)) { + .check_taxonomic_rank(rank, x) + x <- agglomerateByRank(x, rank = rank) + } # check max taxa .check_max_taxa(x, top, assay.type) # check assay .check_assay_present(assay.type, x) # if(method == "prevalence"){ - taxs <- getPrevalence( - assay(x, assay.type), sort = TRUE, include.lowest = TRUE, ...) + taxs <- do.call( + getPrevalence, + c(list(assay(x, assay.type), sort = TRUE, include.lowest = TRUE), + dots)) # If there are taxa with prevalence of 0, remove them taxs <- taxs[ taxs > 0 ] } else { @@ -125,13 +137,13 @@ setMethod("getTop", signature = c(x = "SummarizedExperiment"), method, mean = rowMeans2(assay(x, assay.type), na.rm = na.rm), sum = rowSums2(assay(x, assay.type), na.rm = na.rm), - median = rowMedians(assay(x, assay.type)), na.rm = na.rm) + median = rowMedians(assay(x, assay.type), na.rm = na.rm)) names(taxs) <- rownames(assay(x)) taxs <- sort(taxs,decreasing = TRUE) } names <- head(names(taxs), n = top) # Remove NAs and sort if specified - names <- .remove_NAs_and_sort(names, ... ) + names <- do.call(.remove_NAs_and_sort, c(list(names), dots)) return(names) } ) diff --git a/man/summary.Rd b/man/summary.Rd index 74857c894..2aee34577 100644 --- a/man/summary.Rd +++ b/man/summary.Rd @@ -18,9 +18,10 @@ getUnique(x, ...) getTop( x, top = 5L, - method = c("mean", "sum", "median"), + method = c("mean", "sum", "median", "prevalence"), assay.type = assay_name, assay_name = "counts", + rank = NULL, na.rm = TRUE, ... ) @@ -31,6 +32,7 @@ getTop( method = c("mean", "sum", "median", "prevalence"), assay.type = assay_name, assay_name = "counts", + rank = NULL, na.rm = TRUE, ... ) @@ -65,12 +67,12 @@ assay used in calculation. (Default: \code{"counts"})} \item{assay_name}{Deprecated. Use \code{assay.type} instead.} -\item{na.rm}{\code{Logical scalar}. Should NA values be omitted? -(Default: \code{TRUE})} - \item{rank}{\code{Character scalar}. Defines a taxonomic rank. Must be a value of the output of \code{taxonomyRanks()}. (Default: \code{NULl})} +\item{na.rm}{\code{Logical scalar}. Should NA values be omitted? +(Default: \code{TRUE})} + \item{object}{A \code{\link[SummarizedExperiment:SummarizedExperiment-class]{SummarizedExperiment}} object.} diff --git a/tests/testthat/test-8summaries.R b/tests/testthat/test-8summaries.R index 86428ec8e..cb60c01f3 100644 --- a/tests/testthat/test-8summaries.R +++ b/tests/testthat/test-8summaries.R @@ -60,4 +60,12 @@ test_that("summaries", { # Test with multiple equal dominant taxa in one sample assay(GlobalPatterns)[1, 1] <- max(assay(GlobalPatterns)[, 1]) expect_warning(summarizeDominance(GlobalPatterns, complete = FALSE)) + + # getTop with rank aggregation matches explicit agglomeration + agg <- agglomerateByRank(GlobalPatterns, rank = "Genus") + expect_identical( + getTop(GlobalPatterns, rank = "Genus", method = "mean", top = 5, + assay.type = "counts"), + getTop(agg, method = "mean", top = 5, assay.type = "counts") + ) }) From 4fe1b2e1c417f38bc09094a3abb21645175baac9 Mon Sep 17 00:00:00 2001 From: Tuomas Borman Date: Thu, 24 Sep 2026 08:12:12 +0300 Subject: [PATCH 2/3] up --- R/AllGenerics.R | 3 +- R/summaries.R | 50 +++++++++++++------------------- tests/testthat/test-8summaries.R | 18 ++++++------ 3 files changed, 30 insertions(+), 41 deletions(-) diff --git a/R/AllGenerics.R b/R/AllGenerics.R index feedccca9..4fff3bf2a 100644 --- a/R/AllGenerics.R +++ b/R/AllGenerics.R @@ -274,8 +274,7 @@ setGeneric("getUnique", signature = c("x"), function(x, ...) setGeneric("getTop", signature = "x", function( x, top= 5L, method = c("mean", "sum", "median", "prevalence"), - assay.type = assay_name, assay_name = "counts", rank = NULL, - na.rm = TRUE, ...) + assay.type = assay_name, assay_name = "counts", na.rm = TRUE, ...) standardGeneric("getTop")) #' @rdname taxonomy-methods diff --git a/R/summaries.R b/R/summaries.R index c010a06a6..e9f0ec5ce 100644 --- a/R/summaries.R +++ b/R/summaries.R @@ -4,18 +4,13 @@ #' functions are available. #' #' @inheritParams getPrevalence -#' +#' #' @param top \code{Numeric scalar}. Determines how many top taxa to return. #' Default is to return top five taxa. (Default: \code{5}) #' #' @param method \code{Character scalar}. Specify the method to determine top #' taxa. Either sum, mean, median or prevalence. (Default: \code{"mean"}) #' -#' @param rank \code{Character scalar}. Defines a taxonomic rank. When -#' provided, input is first aggregated with \code{agglomerateByRank()} at the -#' given rank before selecting top features. Must be one of -#' \code{taxonomyRanks(x)}. (Default: \code{NULL}). -#' #' @param ... Additional arguments passed, e.g., to getPrevalence: #' \itemize{ #' \item \code{sort}: \code{Logical scalar}. Specify @@ -28,7 +23,7 @@ #' \code{getUnique}, and \code{getTop}. #' (Default: \code{FALSE}) #' } -#' +#' #' @details #' The \code{getTop} extracts the most \code{top} abundant \dQuote{FeatureID}s #' in a @@ -54,7 +49,7 @@ #' top = 5, #' assay.type = "counts") #' top_taxa -#' +#' #' # Use 'detection' to select detection threshold when using prevalence method #' top_taxa <- getTop(GlobalPatterns, #' method = "prevalence", @@ -62,7 +57,7 @@ #' assay_name = "counts", #' detection = 100) #' top_taxa -#' +#' #' # Top taxa os specific rank #' getTop(agglomerateByRank(GlobalPatterns, #' rank = "Genus", @@ -111,25 +106,20 @@ NULL setMethod("getTop", signature = c(x = "SummarizedExperiment"), function( x, top = 5L, method = c("mean", "sum", "median", "prevalence"), - assay.type = assay_name, assay_name = "counts", rank = NULL, + assay.type = assay_name, assay_name = "counts", na.rm = TRUE, ...){ - dots <- list(...) # input check method <- match.arg(method, c("mean","sum","median","prevalence")) - if (!is.null(rank)) { - .check_taxonomic_rank(rank, x) - x <- agglomerateByRank(x, rank = rank) - } + # Optionally agglomerate data + x <- .merge_features(x, ...) # check max taxa .check_max_taxa(x, top, assay.type) # check assay .check_assay_present(assay.type, x) # if(method == "prevalence"){ - taxs <- do.call( - getPrevalence, - c(list(assay(x, assay.type), sort = TRUE, include.lowest = TRUE), - dots)) + taxs <- getPrevalence( + assay(x, assay.type), sort = TRUE, include.lowest = TRUE, ...) # If there are taxa with prevalence of 0, remove them taxs <- taxs[ taxs > 0 ] } else { @@ -143,7 +133,7 @@ setMethod("getTop", signature = c(x = "SummarizedExperiment"), } names <- head(names(taxs), n = top) # Remove NAs and sort if specified - names <- do.call(.remove_NAs_and_sort, c(list(names), dots)) + names <- .remove_NAs_and_sort(names, ... ) return(names) } ) @@ -156,7 +146,7 @@ setMethod("getTop", signature = c(x = "SummarizedExperiment"), #' @return #' The \code{getUnique} returns a vector of unique taxa present at a #' particular rank -#' +#' #' @export NULL @@ -176,8 +166,8 @@ setMethod("getUnique", signature = c(x = "SummarizedExperiment"), #' #' @param group With group, it is possible to group the observations in an #' overview. Must be one of the column names of \code{colData}. -#' -#' @param name \code{Character scalar}. A name for the column of the +#' +#' @param name \code{Character scalar}. A name for the column of the #' \code{colData} where results will be stored. #' (Default: \code{"dominant_taxa"}) #' @@ -195,7 +185,7 @@ setMethod("getUnique", signature = c(x = "SummarizedExperiment"), #' The \code{summarizeDominance} returns an overview in a tibble. It contains #' dominant taxa in a column named \code{*name*} and its abundance in the data #' set. -#' +#' #' @export NULL @@ -255,7 +245,7 @@ setMethod("summarizeDominance", signature = c(x = "SummarizedExperiment"), .tally_col_data <- function(data, group, colname, digits = NULL, ...){ # Convert data to data.frame data <- as.data.frame(data) - + # # If there are multiple dominant taxa in one sample, the column is a list. # # Convert it so that there are multiple rows for sample and each row # contains one dominant taxa. @@ -267,18 +257,18 @@ setMethod("summarizeDominance", signature = c(x = "SummarizedExperiment"), # Add dominant taxa data[[colname]] <- dominant_taxa } - + # Creates a tibble that contains number of times that a column of "name" # is present in samples and relative portion of samples where they # present. - + # digits check if(!is.null(digits)){ if(!is.numeric(digits)) { stop("'digits' must be numeric", call. = FALSE) - } + } } - + if (is.null(group)) { colname <- sym(colname) data <- data %>% @@ -296,7 +286,7 @@ setMethod("summarizeDominance", signature = c(x = "SummarizedExperiment"), ) %>% arrange(desc(n)) if(!is.null(digits)) { - tallied_data["rel_freq"] <- round(tallied_data["rel_freq"], digits) + tallied_data["rel_freq"] <- round(tallied_data["rel_freq"], digits) } return(tallied_data) } diff --git a/tests/testthat/test-8summaries.R b/tests/testthat/test-8summaries.R index cb60c01f3..c9b35681c 100644 --- a/tests/testthat/test-8summaries.R +++ b/tests/testthat/test-8summaries.R @@ -1,11 +1,11 @@ context("summary") test_that("summary", { - + data(GlobalPatterns, package="mia") sumdf <- summary(GlobalPatterns, assay.type="counts") samples.sum <- sumdf$samples - + # check samples colnames.samples <- c("total_counts", "min_counts", "max_counts", "median_counts", @@ -31,11 +31,11 @@ test_that("summary", { context("getUnique") test_that("getUnique", { - + data(GlobalPatterns, package="mia") exp.phy <- c("Crenarchaeota","Euryarchaeota", "Actinobacteria","Spirochaetes","MVP-15") - + expect_equal(getUnique(GlobalPatterns, "Phylum")[1:5], exp.phy) }) @@ -43,20 +43,20 @@ test_that("getUnique", { context("summaries") test_that("summaries", { - + data(GlobalPatterns, package="mia") - expect_equal( getTop(GlobalPatterns, + expect_equal( getTop(GlobalPatterns, method = "mean", top = 5, - assay.type = "counts"), - getTop(GlobalPatterns, + assay.type = "counts"), + getTop(GlobalPatterns, method = "mean", top = 5, assay.type = "counts") ) expect_equal( summarizeDominance(GlobalPatterns), summarizeDominance(GlobalPatterns)) - + # Test with multiple equal dominant taxa in one sample assay(GlobalPatterns)[1, 1] <- max(assay(GlobalPatterns)[, 1]) expect_warning(summarizeDominance(GlobalPatterns, complete = FALSE)) From c2660655450dad225a55ecd375e22a038d5bb168 Mon Sep 17 00:00:00 2001 From: Tuomas Borman Date: Thu, 24 Sep 2026 08:15:58 +0300 Subject: [PATCH 3/3] update documents --- DESCRIPTION | 2 +- R/summaries.R | 38 +++++++++++++++++++++----------------- man/summary.Rd | 48 +++++++++++++++++++++++++----------------------- 3 files changed, 47 insertions(+), 41 deletions(-) diff --git a/DESCRIPTION b/DESCRIPTION index 789b630b3..0261ef90b 100644 --- a/DESCRIPTION +++ b/DESCRIPTION @@ -1,6 +1,6 @@ Package: mia Type: Package -Version: 1.21.8 +Version: 1.21.9 Authors@R: c(person(given = "Tuomas", family = "Borman", role = c("aut", "cre"), email = "tuomas.v.borman@utu.fi", diff --git a/R/summaries.R b/R/summaries.R index e9f0ec5ce..5e46e22e0 100644 --- a/R/summaries.R +++ b/R/summaries.R @@ -44,24 +44,26 @@ #' #' @examples #' data(GlobalPatterns) -#' top_taxa <- getTop(GlobalPatterns, -#' method = "mean", -#' top = 5, -#' assay.type = "counts") +#' top_taxa <- getTop( +#' GlobalPatterns, +#' method = "mean", +#' top = 5, +#' assay.type = "counts" +#' ) #' top_taxa #' #' # Use 'detection' to select detection threshold when using prevalence method -#' top_taxa <- getTop(GlobalPatterns, -#' method = "prevalence", -#' top = 5, -#' assay_name = "counts", -#' detection = 100) +#' top_taxa <- getTop( +#' GlobalPatterns, +#' method = "prevalence", +#' top = 5, +#' assay.type = "counts", +#' detection = 100 +#' ) #' top_taxa #' -#' # Top taxa os specific rank -#' getTop(agglomerateByRank(GlobalPatterns, -#' rank = "Genus", -#' na.rm = TRUE)) +#' # Top taxa in specific rank +#' getTop(GlobalPatterns, rank = "Genus", na.rm = TRUE) #' #' # Gets the overview of dominant taxa #' dominant_taxa <- summarizeDominance(GlobalPatterns, @@ -70,10 +72,12 @@ #' #' # With group, it is possible to group observations based on specified groups #' # Gets the overview of dominant taxa -#' dominant_taxa <- summarizeDominance(GlobalPatterns, -#' rank = "Genus", -#' group = "SampleType", -#' na.rm = TRUE) +#' dominant_taxa <- summarizeDominance( +#' GlobalPatterns, +#' rank = "Genus", +#' group = "SampleType", +#' na.rm = TRUE +#' ) #' #' dominant_taxa #' diff --git a/man/summary.Rd b/man/summary.Rd index 2aee34577..7ebf546fd 100644 --- a/man/summary.Rd +++ b/man/summary.Rd @@ -21,7 +21,6 @@ getTop( method = c("mean", "sum", "median", "prevalence"), assay.type = assay_name, assay_name = "counts", - rank = NULL, na.rm = TRUE, ... ) @@ -32,7 +31,6 @@ getTop( method = c("mean", "sum", "median", "prevalence"), assay.type = assay_name, assay_name = "counts", - rank = NULL, na.rm = TRUE, ... ) @@ -67,12 +65,12 @@ assay used in calculation. (Default: \code{"counts"})} \item{assay_name}{Deprecated. Use \code{assay.type} instead.} -\item{rank}{\code{Character scalar}. Defines a taxonomic rank. Must be a -value of the output of \code{taxonomyRanks()}. (Default: \code{NULl})} - \item{na.rm}{\code{Logical scalar}. Should NA values be omitted? (Default: \code{TRUE})} +\item{rank}{\code{Character scalar}. Defines a taxonomic rank. Must be a +value of the output of \code{taxonomyRanks()}. (Default: \code{NULl})} + \item{object}{A \code{\link[SummarizedExperiment:SummarizedExperiment-class]{SummarizedExperiment}} object.} @@ -114,24 +112,26 @@ object. } \examples{ data(GlobalPatterns) -top_taxa <- getTop(GlobalPatterns, - method = "mean", - top = 5, - assay.type = "counts") +top_taxa <- getTop( + GlobalPatterns, + method = "mean", + top = 5, + assay.type = "counts" +) top_taxa # Use 'detection' to select detection threshold when using prevalence method -top_taxa <- getTop(GlobalPatterns, - method = "prevalence", - top = 5, - assay_name = "counts", - detection = 100) +top_taxa <- getTop( + GlobalPatterns, + method = "prevalence", + top = 5, + assay.type = "counts", + detection = 100 +) top_taxa - -# Top taxa os specific rank -getTop(agglomerateByRank(GlobalPatterns, - rank = "Genus", - na.rm = TRUE)) + +# Top taxa in specific rank +getTop(GlobalPatterns, rank = "Genus", na.rm = TRUE) # Gets the overview of dominant taxa dominant_taxa <- summarizeDominance(GlobalPatterns, @@ -140,10 +140,12 @@ dominant_taxa # With group, it is possible to group observations based on specified groups # Gets the overview of dominant taxa -dominant_taxa <- summarizeDominance(GlobalPatterns, - rank = "Genus", - group = "SampleType", - na.rm = TRUE) +dominant_taxa <- summarizeDominance( + GlobalPatterns, + rank = "Genus", + group = "SampleType", + na.rm = TRUE +) dominant_taxa