diff --git a/NAMESPACE b/NAMESPACE index 4d71feb..783ba21 100644 --- a/NAMESPACE +++ b/NAMESPACE @@ -2,8 +2,10 @@ S3method(print,topicHierarchy) export(annotateProteinInfoFromIndra) +export(backend_capabilities) export(bootstrapTopicModels) export(compareTopicModels) +export(convert_ids) export(cytoscapeNetwork) export(cytoscapeNetworkOutput) export(decomposeSubnetworkByTopic) @@ -12,9 +14,23 @@ export(deleteEdgeFromNetwork) export(exportNetworkToHTML) export(filterSubnetworkByContext) export(getSubnetworkFromIndra) +export(get_entity_properties) +export(get_network) +export(indra_backend) +export(prepare_entities) export(previewNetworkInBrowser) export(renderCytoscapeNetwork) +export(select_entities) +export(subnetwork_query) export(validate_network) +exportClasses(IndraBackend) +exportClasses(NetworkBackend) +exportClasses(NetworkQuery) +exportClasses(SubnetworkQuery) +exportMethods(backend_capabilities) +exportMethods(convert_ids) +exportMethods(get_entity_properties) +exportMethods(get_network) importFrom(grDevices, colorRamp, rgb diff --git a/NEWS.md b/NEWS.md index b82e8a2..2deb698 100644 --- a/NEWS.md +++ b/NEWS.md @@ -2,6 +2,35 @@ ## New features +* A new API for building networks from MSstats results, which separates +the steps that `annotateProteinInfoFromIndra()` and +`getSubnetworkFromIndra()` combine, and works for proteins, PTM sites, and +metabolites: + * `prepare_entities()` builds an entity table with one row per analyte, + its entity type (`"protein"`, `"ptm_site"`, `"metabolite"`, ...), + identifier system, organism, and statistics. It copies `log2FC`, + `log10FC`, or `logFC` to `logFC`, parses PTM sites, and stops when the + input has several comparisons in `Label` unless `label` names one. + * `indra_backend()` creates the INDRA backend, and + `backend_capabilities()` lists what a backend supports. + * `convert_ids()` grounds the entity table (CoGEx for UniProt IDs and + mnemonics, Gilda for gene symbols and chemical names), and + `get_entity_properties()` adds `is_transcription_factor`, `is_kinase`, + and `is_phosphatase`. + * `select_entities()` flags the rows that pass the cutoffs, and drops + none, so that nodes in the input are recognized even when they fail + the cutoffs. + * `get_network(backend, entities, query = subnetwork_query())` queries + the backend and returns `nodes` and `edges` that meet the contract. It + takes `interaction_types`, `min_evidence`, `min_confidence` (new: drops + edges below it, and edges with no score), `evidence_sources`, and + `include_entities`. It prints the question it asks as a message, e.g. + "INDRA subnetwork: how are 42 selected proteins connected to each + other, with no other nodes added?". + * `?network_queries` describes the questions a query can ask, with a + glossary. `subnetwork_query()` is the first; more are planned. + * The S4 classes `NetworkBackend`, `IndraBackend`, `NetworkQuery`, and + `SubnetworkQuery` are exported, so other packages can add backends. * New function `validate_network()` checks a `list(nodes, edges)` network against the edge and node contract (v1.0): required columns and types, the statement-type and entity-type vocabularies, value ranges, `NA` statistics @@ -92,29 +121,18 @@ release and be removed in the one after. * `getSubnetworkFromIndra()` now runs through an internal INDRA backend object and the S4 generic `get_network()`, the first step toward supporting -network databases other than INDRA. Its output is unchanged. The new -functions are not exported yet. +network databases other than INDRA. * The error for a non-character `sources_filter` in `getSubnetworkFromIndra()` now reads "evidence_sources must be a character vector", the name of the argument in the new API. -* Added internal functions for the entity table that the new API takes as -input: `prepare_entities()` (one row per analyte, with its entity type, -identifier system, and organism; copies `log2FC`, `log10FC`, or `logFC` to -`logFC`; stops when the input has several comparisons in `Label` and -`label` doesn't name one), `parse_ptm_sites()`, `build_grounding_table()`, and -`select_entities()` (flags rows that pass the cutoffs and drops none). -Nothing calls them yet. -* `annotateProteinInfoFromIndra()` now runs through two internal generics -of the new API: `convert_ids()`, which grounds an entity table through the -INDRA backend (CoGEx for UniProt IDs and mnemonics, Gilda for gene symbols -and chemical names), and `get_entity_properties()`, which adds the -`is_transcription_factor`, `is_kinase`, and `is_phosphatase` columns. Its -output is unchanged. The INDRA backend now also holds the Gilda URL, and +* `annotateProteinInfoFromIndra()` now runs through `convert_ids()` and +`get_entity_properties()`. Its output is unchanged. The INDRA backend now also holds the Gilda URL, and the organism of the entity table is passed to Gilda in place of a hard-coded human taxon ID. * `getSubnetworkFromIndra()` builds an entity table from `input`, flags the rows to query with `select_entities()`, and passes all rows to `get_network()`, which matches the nodes INDRA returns against every row. +Like `get_network()`, it now prints the question it asks as a message. # MSstatsBioNet 0.99.0 diff --git a/R/AllClasses.R b/R/AllClasses.R index 669d397..ee73033 100644 --- a/R/AllClasses.R +++ b/R/AllClasses.R @@ -1,23 +1,27 @@ #' Network backend classes #' -#' A backend is a source of prior-knowledge networks. \code{get_network()} -#' dispatches on the backend and the query, so each (backend, query) pair has -#' its own method. +#' A backend is a source of prior-knowledge networks, such as INDRA. +#' \code{\link{get_network}()} dispatches on the backend and the query, so +#' each (backend, query) pair has its own method. \code{NetworkBackend} is +#' virtual: create a backend with a constructor such as +#' \code{\link{indra_backend}()}. Other packages can add a backend by +#' extending \code{NetworkBackend} and writing methods for the generics. #' -#' Internal until the entity model is added (Phase 3 of the API refactor). -#' @importFrom methods setClass setValidity -#' @keywords internal -#' @noRd -setClass("NetworkBackend", representation("VIRTUAL")) - -#' INDRA backend +#' \code{IndraBackend} queries INDRA CoGEx for networks and grounds names +#' with Gilda, INDRA's grounding service. #' -#' Queries INDRA CoGEx, and grounds names with Gilda. The Network Search URL -#' is added with its first query (Phase 7 of the API refactor). #' @slot cogex_url base URL of INDRA CoGEx #' @slot grounding_url base URL of Gilda -#' @keywords internal -#' @noRd +#' +#' @seealso \code{\link{indra_backend}()}, \code{\link{backend_capabilities}()} +#' @name NetworkBackend-class +#' @aliases NetworkBackend-class IndraBackend-class +#' @importFrom methods setClass setValidity +#' @exportClass NetworkBackend IndraBackend +NULL + +setClass("NetworkBackend", representation("VIRTUAL")) + setClass("IndraBackend", contains = "NetworkBackend", representation(cogex_url = "character", grounding_url = "character")) @@ -34,12 +38,17 @@ setValidity("IndraBackend", function(object) { #' Network query classes #' -#' A query says which question \code{get_network()} asks of the backend. -#' @keywords internal -#' @noRd +#' A query says which question \code{\link{get_network}()} asks of the +#' backend. \code{NetworkQuery} is virtual: create a query with a +#' constructor such as \code{\link{subnetwork_query}()}. See +#' \code{\link{network_queries}} for the questions each query answers. +#' +#' @seealso \code{\link{network_queries}} +#' @name NetworkQuery-class +#' @aliases NetworkQuery-class SubnetworkQuery-class +#' @exportClass NetworkQuery SubnetworkQuery +NULL + setClass("NetworkQuery", representation("VIRTUAL")) -#' Subnetwork query: the edges among the selected entities, adding no nodes -#' @keywords internal -#' @noRd setClass("SubnetworkQuery", contains = "NetworkQuery") diff --git a/R/AllGenerics.R b/R/AllGenerics.R index 8109556..440e7bb 100644 --- a/R/AllGenerics.R +++ b/R/AllGenerics.R @@ -1,53 +1,90 @@ #' Get a network from a backend #' -#' Dispatches on \code{backend} and \code{query}. A missing \code{query} -#' runs \code{subnetwork_query()}. +#' Sends the selected entities to a backend and returns the network that +#' answers the query, as \code{nodes} and \code{edges} tables. #' -#' Only the \code{included_in_query} rows of \code{entities} are queried. -#' Every node the backend returns is matched against all rows, so a node -#' in the data gets \code{measured = TRUE} and its statistics, and a node -#' outside it gets \code{measured = FALSE} and \code{NA} statistics. +#' Only the rows of \code{entities} with \code{included_in_query = TRUE} +#' are sent to the backend. Every node the backend returns is matched +#' against all rows, so a node in the input gets \code{measured = TRUE} and +#' its statistics, and a node not in the input gets \code{measured = FALSE} +#' and \code{NA} statistics. #' -#' Internal until the end of Phase 3 of the API refactor. +#' \code{get_network()} prints the question it asks as a message, with the +#' number of entities, so the query in a saved script or log is readable +#' without the documentation. #' -#' @param backend a \code{NetworkBackend}, e.g. from \code{indra_backend()} -#' @param entities entity table from \code{prepare_entities()}, grounded by -#' \code{convert_ids()} and flagged by \code{select_entities()} -#' @param query a \code{NetworkQuery}, e.g. from \code{subnetwork_query()} +#' Confidence values are comparable within one backend, not across +#' backends. For INDRA, \code{confidence} is the INDRA belief score. +#' +#' @param backend a \code{NetworkBackend}, e.g. from +#' \code{\link{indra_backend}()} +#' @param entities entity table from \code{\link{prepare_entities}()}, +#' grounded by \code{\link{convert_ids}()} and flagged by +#' \code{\link{select_entities}()} +#' @param query a \code{NetworkQuery} saying which question to ask, e.g. +#' \code{\link{subnetwork_query}()} (the default). See +#' \code{\link{network_queries}}. #' @param interaction_types values of \code{edges$interaction} to keep #' (INDRA statement types, e.g. \code{"Activation"}). \code{NULL} keeps all. #' @param min_evidence minimum evidence count per edge +#' @param min_confidence minimum \code{confidence} per edge, in [0, 1]. +#' Edges with no confidence score (\code{NA}) are dropped too, and a message +#' says how many. \code{NULL} applies no cutoff. #' @param evidence_sources keeps edges with evidence from at least one of #' these sources, e.g. \code{c("reach")}. \code{NULL} keeps all. -#' @param include_entities \code{"namespace:identifier"} strings to add to -#' the query, e.g. \code{"HGNC:1234"} +#' @param include_entities \code{"namespace:identifier"} groundings to add to +#' the query, e.g. \code{"HGNC:1234"}. Use this for entities outside the +#' input; to keep entities of the input that fail the cutoffs, use +#' \code{select_entities(force_include = )}. #' @param ... passed to methods -#' @return list of \code{nodes} and \code{edges} that meets the contract -#' checked by \code{validate_network()} +#' @return list of \code{nodes} and \code{edges} data.frames that meets the +#' contract checked by \code{\link{validate_network}()} +#' @seealso \code{\link{network_queries}}, \code{\link{backend_capabilities}()} #' @importFrom methods setGeneric setMethod -#' @keywords internal -#' @noRd +#' @export +#' @examples +#' input <- data.table::fread(system.file( +#' "extdata/groupComparisonModel.csv", +#' package = "MSstatsBioNet" +#' )) +#' entities <- prepare_entities(input, entity_type = "protein", +#' id_type = "uniprot") +#' \donttest{ +#' indra <- indra_backend() +#' entities <- convert_ids(indra, entities) +#' entities <- select_entities(entities, pvalue_cutoff = 0.05) +#' network <- get_network(indra, entities, subnetwork_query(), +#' interaction_types = "Complex") +#' head(network$nodes) +#' head(network$edges) +#' } setGeneric("get_network", function(backend, entities, query = subnetwork_query(), interaction_types = NULL, min_evidence = 1, - evidence_sources = NULL, include_entities = NULL, ...) + min_confidence = NULL, evidence_sources = NULL, + include_entities = NULL, ...) standardGeneric("get_network"), signature = c("backend", "query")) # S4 dispatches a missing argument as class "missing", not on its default. # The shared arguments are generic formals, so they don't reach `...` and # must be passed on by name. +#' @rdname get_network +#' @export setMethod("get_network", signature("NetworkBackend", "missing"), function(backend, entities, query, interaction_types = NULL, - min_evidence = 1, evidence_sources = NULL, - include_entities = NULL, ...) { + min_evidence = 1, min_confidence = NULL, + evidence_sources = NULL, include_entities = NULL, ...) { get_network(backend, entities, subnetwork_query(), interaction_types = interaction_types, min_evidence = min_evidence, + min_confidence = min_confidence, evidence_sources = evidence_sources, include_entities = include_entities, ...) }) +#' @rdname get_network +#' @export setMethod("get_network", signature("NetworkBackend", "NetworkQuery"), function(backend, entities, query, ...) { stop(class(backend), " does not support ", class(query), ".", @@ -59,20 +96,36 @@ setMethod("get_network", signature("NetworkBackend", "NetworkQuery"), #' Fills in \code{namespace}, \code{entity_id}, and \code{entity_name} from #' each row's \code{id} (\code{parent_id} for \code{ptm_site} rows), read #' as the identifier system in \code{id_type}. Rows that don't ground are -#' left \code{NA}. +#' left \code{NA}. \code{backend_capabilities(backend)$id_conversions} lists the +#' (entity type, identifier system) pairs a backend can convert. +#' +#' Ground the full results table, not only the significant rows, so that +#' \code{\link{get_network}()} can recognize every node that is in the +#' input. #' -#' Internal until the end of Phase 3 of the API refactor. +#' For INDRA, UniProt IDs and mnemonics are mapped through CoGEx, and gene +#' symbols and chemical names are grounded with Gilda. When an identifier +#' is a protein group (\code{"P1;P2"}) or a name with several candidates, +#' the groundings are \code{";"}-joined and positionally aligned. #' -#' @param backend a \code{NetworkBackend}, e.g. from \code{indra_backend()} -#' @param entities entity table from \code{prepare_entities()} +#' @param backend a \code{NetworkBackend}, e.g. from +#' \code{\link{indra_backend}()} +#' @param entities entity table from \code{\link{prepare_entities}()} #' @param ... passed to methods #' @return \code{entities} with the grounding columns filled in -#' @keywords internal -#' @noRd +#' @seealso \code{\link{get_entity_properties}()} +#' @export +#' @examples +#' df <- data.frame(Protein = c("P04637", "Q00610")) +#' entities <- prepare_entities(df, entity_type = "protein", +#' id_type = "uniprot") +#' convert_ids(indra_backend(), entities) setGeneric("convert_ids", function(backend, entities, ...) standardGeneric("convert_ids"), signature = "backend") +#' @rdname convert_ids +#' @export setMethod("convert_ids", "NetworkBackend", function(backend, entities, ...) { stop(class(backend), " does not support convert_ids().", @@ -84,24 +137,75 @@ setMethod("convert_ids", "NetworkBackend", #' Adds one column per property, e.g. \code{is_kinase}. A property is #' \code{NA} for rows whose \code{entity_type} it doesn't apply to, and for #' rows the backend has no answer for. +#' \code{backend_capabilities(backend)$entity_properties} lists the properties a +#' backend supports. #' -#' Internal until the end of Phase 3 of the API refactor. +#' For INDRA, the properties are \code{is_transcription_factor}, +#' \code{is_kinase}, and \code{is_phosphatase}, looked up by gene symbol. +#' Only rows with a single HGNC grounding get values. A PTM site gets the +#' properties of its parent protein. #' -#' @param backend a \code{NetworkBackend}, e.g. from \code{indra_backend()} -#' @param entities entity table, grounded by \code{convert_ids()} +#' @param backend a \code{NetworkBackend}, e.g. from +#' \code{\link{indra_backend}()} +#' @param entities entity table, grounded by \code{\link{convert_ids}()} #' @param properties the properties to add. \code{NULL} adds every #' property the backend supports. #' @param ... passed to methods #' @return \code{entities} with one column per property -#' @keywords internal -#' @noRd +#' @export +#' @examples +#' df <- data.frame(Protein = c("P04637", "Q00610")) +#' indra <- indra_backend() +#' entities <- prepare_entities(df, entity_type = "protein", +#' id_type = "uniprot") +#' entities <- convert_ids(indra, entities) +#' get_entity_properties(indra, entities, properties = "is_kinase") setGeneric("get_entity_properties", function(backend, entities, properties = NULL, ...) standardGeneric("get_entity_properties"), signature = "backend") +#' @rdname get_entity_properties +#' @export setMethod("get_entity_properties", "NetworkBackend", function(backend, entities, properties = NULL, ...) { stop(class(backend), " does not support get_entity_properties().", call. = FALSE) }) + +#' What a backend supports +#' +#' Lists the queries, identifier conversions, entity properties, and +#' limits of a backend, so a choice can be checked before a query is sent. +#' +#' @param backend a \code{NetworkBackend}, e.g. from +#' \code{\link{indra_backend}()} +#' @return named list: +#' \describe{ +#' \item{query_types}{the \code{query_type} values of the queries the +#' backend answers, e.g. \code{"subnetwork"} for +#' \code{\link{subnetwork_query}()}} +#' \item{id_conversions}{for each \code{entity_type}, the \code{id_type} +#' values \code{\link{convert_ids}()} can convert} +#' \item{entity_properties}{the properties +#' \code{\link{get_entity_properties}()} can add} +#' \item{interaction_types}{the \code{edges$interaction} values the +#' backend returns} +#' \item{max_nodes}{for each query type, the largest number of +#' groundings one query can take} +#' } +#' @seealso \code{\link{network_queries}} +#' @export +#' @examples +#' backend_capabilities(indra_backend()) +setGeneric("backend_capabilities", + function(backend) standardGeneric("backend_capabilities"), + signature = "backend") + +#' @rdname backend_capabilities +#' @export +setMethod("backend_capabilities", "NetworkBackend", + function(backend) { + stop(class(backend), " does not describe its backend_capabilities().", + call. = FALSE) + }) diff --git a/R/backend-indra.R b/R/backend-indra.R index ac18d73..83e2dd8 100644 --- a/R/backend-indra.R +++ b/R/backend-indra.R @@ -10,35 +10,51 @@ GILDA_API_URL <- "https://grounding.indra.bio" #' Create an INDRA backend #' -#' Internal until the end of Phase 3 of the API refactor. +#' INDRA is a knowledge graph of mechanisms (activations, phosphorylations, +#' complexes, ...) assembled from the literature and curated databases. The +#' backend queries INDRA CoGEx for networks and grounds gene symbols and +#' chemical names with Gilda, INDRA's grounding service. +#' \code{backend_capabilities(indra_backend())} lists what it supports. +#' +#' This function includes third-party software components that are licensed +#' under the BSD 2-Clause License. Include the third-party licensing +#' agreements if redistributing this package or results based on it. See +#' the LICENSE file for details. +#' #' @param cogex_url base URL of INDRA CoGEx #' @param grounding_url base URL of Gilda -#' @return an \code{IndraBackend} object +#' @return an \code{IndraBackend} object, to pass to +#' \code{\link{convert_ids}()}, \code{\link{get_entity_properties}()}, and +#' \code{\link{get_network}()} +#' @seealso \code{\link{NetworkBackend-class}} #' @importFrom methods new -#' @keywords internal -#' @noRd +#' @export +#' @examples +#' indra <- indra_backend() +#' backend_capabilities(indra)$query_types indra_backend <- function(cogex_url = INDRA_API_URL, grounding_url = GILDA_API_URL) { new("IndraBackend", cogex_url = cogex_url, grounding_url = grounding_url) } -#' INDRA subnetwork query -#' -#' Sends the groundings of the \code{included_in_query} rows to CoGEx -#' \code{indra_subnetwork_relations}, filters the statements, and normalizes -#' them to the edge contract. The source and target of each statement are -#' matched back to the rows of \code{entities} by grounding, so nodes carry -#' their statistics. Backend nodes that match no row (from -#' \code{include_entities}) become latent nodes. -#' @keywords internal -#' @noRd +# INDRA subnetwork query. Sends the groundings of the included_in_query rows +# to CoGEx indra_subnetwork_relations, filters the statements, and +# normalizes them to the edge contract. The source and target of each +# statement are matched back to the rows of entities by grounding, so nodes +# carry their statistics. Backend nodes that match no row (from +# include_entities) become latent nodes. +#' @rdname get_network +#' @export setMethod("get_network", signature("IndraBackend", "SubnetworkQuery"), function(backend, entities, query, interaction_types = NULL, - min_evidence = 1, evidence_sources = NULL, - include_entities = NULL, ...) { + min_evidence = 1, min_confidence = NULL, + evidence_sources = NULL, include_entities = NULL, ...) { groundings <- .get_groundings_to_query(entities) .validateIndraSubnetworkInput(groundings, evidence_sources, include_entities) + .check_min_confidence(min_confidence) + message(.describe_subnetwork_question(entities, groundings, + include_entities)) statements <- .callIndraCogexApi(groundings$namespace, groundings$entity_id, include_entities, backend@cogex_url) @@ -46,6 +62,7 @@ setMethod("get_network", signature("IndraBackend", "SubnetworkQuery"), min_evidence, evidence_sources) grounding_lookup <- .build_grounding_lookup(entities) edges <- .constructEdgesDataFrame(statements, grounding_lookup) + edges <- .filter_by_min_confidence(edges, min_confidence) edges <- .filterEdgesDataFrame(edges) nodes <- .build_network_nodes( grounding_lookup, edges, @@ -55,6 +72,80 @@ setMethod("get_network", signature("IndraBackend", "SubnetworkQuery"), network }) +#' Largest number of groundings one INDRA query takes, by query type +#' +#' CoGEx \code{indra_subnetwork_relations} takes fewer than 400, so at most +#' 399. +#' @keywords internal +#' @noRd +INDRA_MAX_NODES <- c(subnetwork = 399) + +#' @rdname backend_capabilities +#' @export +setMethod("backend_capabilities", "IndraBackend", + function(backend) { + list(query_types = "subnetwork", + id_conversions = INDRA_ID_CONVERSIONS, + entity_properties = names(INDRA_ENTITY_PROPERTIES), + interaction_types = INTERACTION_TYPES, + max_nodes = INDRA_MAX_NODES) + }) + +#' The question a subnetwork query asks, with the entity counts +#' +#' Printed by \code{get_network()}, e.g. "INDRA subnetwork: how are 42 +#' selected proteins connected to each other, with no other nodes added?". +#' @param entities entity table +#' @param groundings groundings of the rows to query, from +#' \code{.get_groundings_to_query()} +#' @param include_entities groundings added to the query +#' @return a single string +#' @keywords internal +#' @noRd +.describe_subnetwork_question <- function(entities, groundings, + include_entities) { + queried_types <- entities$entity_type[entities$id %in% groundings$id] + selected <- .describe_entity_count(queried_types, "selected") + if (length(include_entities) > 0) { + selected <- paste0(selected, " and ", length(include_entities), + " added ", + if (length(include_entities) == 1) "entity" + else "entities") + } + paste0("INDRA subnetwork: how are ", selected, " connected to each ", + "other, with no other nodes added?") +} + +#' Names of entity types in messages, singular and plural +#' @keywords internal +#' @noRd +ENTITY_TYPE_NAMES <- list( + protein = c("protein", "proteins"), + gene = c("gene", "genes"), + transcript = c("transcript", "transcripts"), + ptm_site = c("PTM site", "PTM sites"), + metabolite = c("metabolite", "metabolites"), + lipid = c("lipid", "lipids"), + drug = c("drug", "drugs"), + complex = c("complex", "complexes"), + family = c("family", "families"), + other = c("entity", "entities") +) + +#' Count entities by type in words, e.g. "42 selected proteins" +#' @param entity_types entity type of each entity +#' @param adjective word before the type, e.g. "selected" +#' @return a single string. Several types are counted as "entities". +#' @keywords internal +#' @noRd +.describe_entity_count <- function(entity_types, adjective) { + types <- unique(entity_types) + names <- if (length(types) == 1) ENTITY_TYPE_NAMES[[types]] else + ENTITY_TYPE_NAMES$other + noun <- if (length(entity_types) == 1) names[1] else names[2] + paste(format(length(entity_types), big.mark = ","), adjective, noun) +} + #' Groundings of the entities to query #' #' Rows with \code{included_in_query} but no grounding can't be sent to a @@ -89,7 +180,7 @@ setMethod("get_network", signature("IndraBackend", "SubnetworkQuery"), unique_groundings <- unique(paste(groundings$namespace, groundings$entity_id, sep = ":")) num_proteins <- length(unique_groundings) + length(include_entities) - if (num_proteins >= 400) { + if (num_proteins > INDRA_MAX_NODES[["subnetwork"]]) { stop("Invalid Input Error: INDRA query must contain less than 400 proteins. Consider lowering your p-value cutoff") } if (nrow(groundings) == 0) { @@ -436,18 +527,15 @@ INDRA_ENTITY_PROPERTIES <- list( entity_types = c("protein", "ptm_site")) ) -#' INDRA identifier conversion -#' -#' Groups rows by \code{id_type} and makes one batch of calls per group: -#' \code{uniprot} through CoGEx's UniProt-to-HGNC mapping, -#' \code{uniprot_mnemonic} through CoGEx's mnemonic-to-UniProt mapping -#' first, \code{hgnc_symbol} through Gilda restricted to HGNC and the -#' row's organism, and \code{chemical_name} through Gilda with no namespace -#' restriction. Each \code{";"}-separated member of an identifier (a protein -#' group) is grounded on its own, and the groundings are pooled onto the -#' row. -#' @keywords internal -#' @noRd +# INDRA identifier conversion. Groups rows by id_type and makes one batch of +# calls per group: uniprot through CoGEx's UniProt-to-HGNC mapping, +# uniprot_mnemonic through CoGEx's mnemonic-to-UniProt mapping first, +# hgnc_symbol through Gilda restricted to HGNC and the row's organism, and +# chemical_name through Gilda with no namespace restriction. Each +# ";"-separated member of an identifier (a protein group) is grounded on its +# own, and the groundings are pooled onto the row. +#' @rdname convert_ids +#' @export setMethod("convert_ids", "IndraBackend", function(backend, entities, ...) { .validate_entities(entities) @@ -473,14 +561,12 @@ setMethod("convert_ids", "IndraBackend", entities }) -#' INDRA entity properties -#' -#' Supports the properties in \code{INDRA_ENTITY_PROPERTIES}. They are -#' looked up by gene symbol, so only rows with a single HGNC grounding get -#' values; rows with several groundings (a protein group, or an ambiguous -#' name) are \code{NA}. -#' @keywords internal -#' @noRd +# INDRA entity properties. Supports the properties in +# INDRA_ENTITY_PROPERTIES. They are looked up by gene symbol, so only rows +# with a single HGNC grounding get values; rows with several groundings (a +# protein group, or an ambiguous name) are NA. +#' @rdname get_entity_properties +#' @export setMethod("get_entity_properties", "IndraBackend", function(backend, entities, properties = NULL, ...) { .validate_entities(entities) diff --git a/R/entities.R b/R/entities.R index a01ed69..1f630d6 100644 --- a/R/entities.R +++ b/R/entities.R @@ -1,7 +1,6 @@ # The entity table: one row per analyte in the MSstats results. Built by # prepare_entities(), grounded by convert_ids(), flagged by -# select_entities(), and read by get_network(). Phase 3 of the API -# refactor; internal until the end of Phase 3. +# select_entities(), and read by get_network(). #' Input identifier systems for entities$id_type #' @@ -44,21 +43,33 @@ REQUIRED_ENTITY_COLUMNS <- c( #' Prepare an entity table from MSstats results #' -#' Builds the table that \code{convert_ids()} grounds, -#' \code{select_entities()} flags, and \code{get_network()} queries. It has -#' one row per analyte, and keeps every analyte, not only the significant -#' ones, so that nodes returned by a backend can be marked as measured. +#' Builds the table that \code{\link{convert_ids}()} grounds, +#' \code{\link{select_entities}()} flags, and \code{\link{get_network}()} +#' queries. It has one row per analyte, and keeps every analyte, not only +#' the significant ones, so that nodes returned by a backend can be +#' recognized as being in the input. +#' +#' For PTM sites (\code{entity_type = "ptm_site"}), the site is parsed +#' from the end of each identifier, e.g. \code{"P00533_S1039_S1042"} gives +#' parent \code{"P00533"} and site \code{"S1039_S1042"}. A +#' \code{GlobalProtein} column, as MSstatsPTM writes, overrides the parsed +#' parent. #' #' @param df output of \code{groupComparison()}'s \code{ComparisonResult} #' table, or any table with one row per analyte. #' @param id_column name of the column holding the analyte identifiers. #' @param entity_type one of the entity types (\code{"protein"}, -#' \code{"ptm_site"}, \code{"metabolite"}, ...), or the name of a column of -#' \code{df} holding one per row. +#' \code{"gene"}, \code{"transcript"}, \code{"ptm_site"}, +#' \code{"metabolite"}, \code{"lipid"}, \code{"drug"}, \code{"complex"}, +#' \code{"family"}, \code{"other"}), or the name of a column of \code{df} +#' holding one per row. #' @param id_type the identifier system of \code{id_column} #' (\code{"uniprot"}, \code{"uniprot_mnemonic"}, \code{"hgnc_symbol"}, -#' \code{"chemical_name"}, ...), or the name of a column of \code{df} -#' holding one per row. For PTM sites, the identifier system of the parent +#' \code{"ensembl_protein"}, \code{"ensembl_gene"}, \code{"entrez"}, +#' \code{"chemical_name"}, \code{"inchikey"}, \code{"hmdb"}, +#' \code{"chebi"}, \code{"chembl"}), or the name of a column of \code{df} +#' holding one per row. Which of them a backend can convert is listed by +#' \code{backend_capabilities(backend)$id_conversions}. For PTM sites, the identifier system of the parent #' protein. \code{"chemical_name"} is a metabolite, lipid, or drug name, #' common or IUPAC (e.g. \code{"glucose"}), grounded by text matching. #' @param organism NCBI taxon ID, as a string. @@ -73,9 +84,22 @@ REQUIRED_ENTITY_COLUMNS <- c( #' (all \code{NA} until \code{convert_ids()}), \code{included_in_query} #' (\code{TRUE} until \code{select_entities()}), \code{site} and #' \code{parent_id} (for \code{ptm_site} rows), \code{organism}, and -#' \code{logFC} and \code{adj.pvalue} when \code{df} has them. -#' @keywords internal -#' @noRd +#' \code{logFC} and \code{adj.pvalue} when \code{df} has them. Other +#' columns of \code{df} are not copied; join them back on \code{id}. +#' @export +#' @examples +#' input <- data.table::fread(system.file( +#' "extdata/groupComparisonModel.csv", +#' package = "MSstatsBioNet" +#' )) +#' entities <- prepare_entities(input, entity_type = "protein", +#' id_type = "uniprot") +#' head(entities) +#' +#' # MSstatsPTM results: the parent protein and site are parsed from the id +#' ptm <- data.frame(Protein = c("P00533_S1039_S1042", "P04637_S15"), +#' log2FC = c(1.2, -0.8), adj.pvalue = c(0.01, 0.2)) +#' prepare_entities(ptm, entity_type = "ptm_site", id_type = "uniprot") prepare_entities <- function(df, id_column = "Protein", entity_type, id_type, organism = "9606", label = NULL, logfc_column = NULL) { @@ -248,12 +272,13 @@ build_grounding_table <- function(entities, namespaces = NULL) { #' Flag the entities to query #' -#' Sets \code{included_in_query} from the statistical cutoffs. Rows that -#' fail are kept, so that \code{get_network()} can still mark them as -#' measured when a backend returns them. Rows with a missing +#' Sets \code{included_in_query} from the statistical cutoffs. Only these +#' rows are sent to the backend by \code{\link{get_network}()}. Rows that +#' fail are kept, so that \code{get_network()} can still recognize them as +#' being in the input when a backend returns them. Rows with a missing #' \code{adj.pvalue} are not selected. #' -#' @param entities entity table from \code{prepare_entities()} +#' @param entities entity table from \code{\link{prepare_entities}()} #' @param pvalue_cutoff keep rows with \code{adj.pvalue} below this. #' \code{NULL} applies no cutoff. #' @param logfc_cutoff keep rows with \code{abs(logFC)} above this, on the @@ -272,8 +297,17 @@ build_grounding_table <- function(entities, namespaces = NULL) { #' @return \code{entities} with \code{included_in_query} set, and #' \code{user_added} (\code{TRUE} for rows selected only through #' \code{force_include}). -#' @keywords internal -#' @noRd +#' @export +#' @examples +#' input <- data.table::fread(system.file( +#' "extdata/groupComparisonModel.csv", +#' package = "MSstatsBioNet" +#' )) +#' entities <- prepare_entities(input, entity_type = "protein", +#' id_type = "uniprot") +#' entities <- select_entities(entities, pvalue_cutoff = 0.01, +#' direction = "up") +#' table(entities$included_in_query) select_entities <- function(entities, pvalue_cutoff = NULL, logfc_cutoff = NULL, direction = c("both", "up", "down"), diff --git a/R/filter-edges.R b/R/filter-edges.R new file mode 100644 index 0000000..3b40590 --- /dev/null +++ b/R/filter-edges.R @@ -0,0 +1,38 @@ +# Edge filters shared by every backend's get_network() method. + +#' Stop unless min_confidence is NULL or a single number in [0, 1] +#' @param min_confidence the argument of get_network() +#' @keywords internal +#' @noRd +.check_min_confidence <- function(min_confidence) { + if (is.null(min_confidence)) { + return(invisible(NULL)) + } + if (!is.numeric(min_confidence) || length(min_confidence) != 1 || + is.na(min_confidence) || min_confidence < 0 || min_confidence > 1) { + stop("min_confidence must be a single number between 0 and 1.", + call. = FALSE) + } + invisible(NULL) +} + +#' Drop edges below a confidence cutoff +#' +#' Edges with no confidence score (\code{NA}) are dropped too, since they +#' can't be shown to pass, and a message says how many. +#' @param edges edges data.frame with a \code{confidence} column +#' @param min_confidence cutoff, or \code{NULL} to keep every edge +#' @return \code{edges}, filtered +#' @keywords internal +#' @noRd +.filter_by_min_confidence <- function(edges, min_confidence) { + if (is.null(min_confidence)) { + return(edges) + } + no_score <- is.na(edges$confidence) + if (any(no_score)) { + message("Dropping ", sum(no_score), " edge(s) with no confidence ", + "score (NA confidence), because min_confidence is set.") + } + edges[!no_score & edges$confidence >= min_confidence, , drop = FALSE] +} diff --git a/R/queries.R b/R/queries.R index 7e9a9c7..c7c7779 100644 --- a/R/queries.R +++ b/R/queries.R @@ -1,14 +1,69 @@ -#' Subnetwork query +#' Questions you can ask of a network backend #' -#' Asks which edges connect the selected entities directly. It adds no -#' nodes, other than entities passed as \code{include_entities} to -#' \code{get_network()}. Edges get \code{query_type = "subnetwork"}. +#' Each query constructor asks one question of a backend. Pass the query to +#' \code{\link{get_network}()}. Every query returns the same \code{nodes} +#' and \code{edges} tables (see \code{\link{validate_network}()}), so +#' visualization and filtering work the same whichever question produced +#' the network. The \code{query_type} column of \code{edges} names the +#' query, without \code{_query}. #' -#' Internal until the entity model is added (Phase 3 of the API refactor). -#' @return a \code{SubnetworkQuery} object +#' \tabular{llll}{ +#' \strong{Constructor} \tab \strong{Question} \tab +#' \strong{Nodes added} \tab \strong{\code{query_type}} \cr +#' \code{\link{subnetwork_query}()} \tab How are my selected entities +#' connected to each other, with no other nodes added? \tab none, other +#' than \code{include_entities} \tab \code{"subnetwork"} \cr +#' } +#' +#' More queries (shared regulators, regulator enrichment, paths, and +#' others) are planned. \code{backend_capabilities(backend)$query_types} lists the +#' ones a backend supports. +#' +#' @section Glossary: +#' \describe{ +#' \item{entity}{One row of the entity table from +#' \code{\link{prepare_entities}()}: one analyte of the input, such as a +#' protein, a PTM site, or a metabolite.} +#' \item{selected}{An entity with \code{included_in_query = TRUE}, set by +#' \code{\link{select_entities}()}: it passed the cutoffs, or was forced +#' in. Only selected entities are sent to the backend.} +#' \item{in the input (\code{measured})}{A node that matches a row of the +#' entity table, whether or not it was selected. It carries that row's +#' statistics.} +#' \item{latent}{A node that matches no row of the entity table, such as +#' an entity added through \code{include_entities}. It has +#' \code{measured = FALSE} and \code{NA} statistics. It may still have +#' been measured in another experiment, or filtered out before the +#' entity table was built.} +#' \item{grounding}{A \code{"namespace:identifier"} pair that names an +#' entity in a backend, such as \code{"HGNC:11998"} (TP53).} +#' } +#' +#' @seealso \code{\link{get_network}()}, \code{\link{backend_capabilities}()} +#' @name network_queries +#' @aliases network_queries +NULL + +#' How are my selected entities connected to each other? +#' +#' Asks which edges connect the selected entities directly, with no other +#' nodes added. Only entities passed to \code{get_network()} as +#' \code{include_entities} are added. +#' +#' Uses the rows of \code{entities} with \code{included_in_query = TRUE}. +#' Each node the backend returns is matched against all rows, so that it +#' carries its statistics. Edges get \code{query_type = "subnetwork"}. +#' For INDRA, the query goes to CoGEx \code{indra_subnetwork_relations}, and +#' takes fewer than 400 groundings +#' (\code{backend_capabilities(indra_backend())$max_nodes}). +#' +#' @return a \code{SubnetworkQuery} object, to pass to +#' \code{\link{get_network}()} +#' @seealso \code{\link{network_queries}} for the other questions #' @importFrom methods new -#' @keywords internal -#' @noRd +#' @export +#' @examples +#' subnetwork_query() subnetwork_query <- function() { new("SubnetworkQuery") } diff --git a/man/NetworkBackend-class.Rd b/man/NetworkBackend-class.Rd new file mode 100644 index 0000000..cf12a56 --- /dev/null +++ b/man/NetworkBackend-class.Rd @@ -0,0 +1,29 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/AllClasses.R +\name{NetworkBackend-class} +\alias{NetworkBackend-class} +\alias{IndraBackend-class} +\title{Network backend classes} +\description{ +A backend is a source of prior-knowledge networks, such as INDRA. +\code{\link{get_network}()} dispatches on the backend and the query, so +each (backend, query) pair has its own method. \code{NetworkBackend} is +virtual: create a backend with a constructor such as +\code{\link{indra_backend}()}. Other packages can add a backend by +extending \code{NetworkBackend} and writing methods for the generics. +} +\details{ +\code{IndraBackend} queries INDRA CoGEx for networks and grounds names +with Gilda, INDRA's grounding service. +} +\section{Slots}{ + +\describe{ +\item{\code{cogex_url}}{base URL of INDRA CoGEx} + +\item{\code{grounding_url}}{base URL of Gilda} +}} + +\seealso{ +\code{\link{indra_backend}()}, \code{\link{backend_capabilities}()} +} diff --git a/man/NetworkQuery-class.Rd b/man/NetworkQuery-class.Rd new file mode 100644 index 0000000..e51561c --- /dev/null +++ b/man/NetworkQuery-class.Rd @@ -0,0 +1,15 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/AllClasses.R +\name{NetworkQuery-class} +\alias{NetworkQuery-class} +\alias{SubnetworkQuery-class} +\title{Network query classes} +\description{ +A query says which question \code{\link{get_network}()} asks of the +backend. \code{NetworkQuery} is virtual: create a query with a +constructor such as \code{\link{subnetwork_query}()}. See +\code{\link{network_queries}} for the questions each query answers. +} +\seealso{ +\code{\link{network_queries}} +} diff --git a/man/backend_capabilities.Rd b/man/backend_capabilities.Rd new file mode 100644 index 0000000..7a4525d --- /dev/null +++ b/man/backend_capabilities.Rd @@ -0,0 +1,44 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/AllGenerics.R, R/backend-indra.R +\name{backend_capabilities} +\alias{backend_capabilities} +\alias{backend_capabilities,NetworkBackend-method} +\alias{backend_capabilities,IndraBackend-method} +\title{What a backend supports} +\usage{ +backend_capabilities(backend) + +\S4method{backend_capabilities}{NetworkBackend}(backend) + +\S4method{backend_capabilities}{IndraBackend}(backend) +} +\arguments{ +\item{backend}{a \code{NetworkBackend}, e.g. from +\code{\link{indra_backend}()}} +} +\value{ +named list: +\describe{ + \item{query_types}{the \code{query_type} values of the queries the + backend answers, e.g. \code{"subnetwork"} for + \code{\link{subnetwork_query}()}} + \item{id_conversions}{for each \code{entity_type}, the \code{id_type} + values \code{\link{convert_ids}()} can convert} + \item{entity_properties}{the properties + \code{\link{get_entity_properties}()} can add} + \item{interaction_types}{the \code{edges$interaction} values the + backend returns} + \item{max_nodes}{for each query type, the largest number of + groundings one query can take} +} +} +\description{ +Lists the queries, identifier conversions, entity properties, and +limits of a backend, so a choice can be checked before a query is sent. +} +\examples{ +backend_capabilities(indra_backend()) +} +\seealso{ +\code{\link{network_queries}} +} diff --git a/man/convert_ids.Rd b/man/convert_ids.Rd new file mode 100644 index 0000000..2b017e2 --- /dev/null +++ b/man/convert_ids.Rd @@ -0,0 +1,51 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/AllGenerics.R, R/backend-indra.R +\name{convert_ids} +\alias{convert_ids} +\alias{convert_ids,NetworkBackend-method} +\alias{convert_ids,IndraBackend-method} +\title{Ground entities in a backend's namespaces} +\usage{ +convert_ids(backend, entities, ...) + +\S4method{convert_ids}{NetworkBackend}(backend, entities, ...) + +\S4method{convert_ids}{IndraBackend}(backend, entities, ...) +} +\arguments{ +\item{backend}{a \code{NetworkBackend}, e.g. from +\code{\link{indra_backend}()}} + +\item{entities}{entity table from \code{\link{prepare_entities}()}} + +\item{...}{passed to methods} +} +\value{ +\code{entities} with the grounding columns filled in +} +\description{ +Fills in \code{namespace}, \code{entity_id}, and \code{entity_name} from +each row's \code{id} (\code{parent_id} for \code{ptm_site} rows), read +as the identifier system in \code{id_type}. Rows that don't ground are +left \code{NA}. \code{backend_capabilities(backend)$id_conversions} lists the +(entity type, identifier system) pairs a backend can convert. +} +\details{ +Ground the full results table, not only the significant rows, so that +\code{\link{get_network}()} can recognize every node that is in the +input. + +For INDRA, UniProt IDs and mnemonics are mapped through CoGEx, and gene +symbols and chemical names are grounded with Gilda. When an identifier +is a protein group (\code{"P1;P2"}) or a name with several candidates, +the groundings are \code{";"}-joined and positionally aligned. +} +\examples{ +df <- data.frame(Protein = c("P04637", "Q00610")) +entities <- prepare_entities(df, entity_type = "protein", + id_type = "uniprot") +convert_ids(indra_backend(), entities) +} +\seealso{ +\code{\link{get_entity_properties}()} +} diff --git a/man/get_entity_properties.Rd b/man/get_entity_properties.Rd new file mode 100644 index 0000000..5e2802f --- /dev/null +++ b/man/get_entity_properties.Rd @@ -0,0 +1,49 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/AllGenerics.R, R/backend-indra.R +\name{get_entity_properties} +\alias{get_entity_properties} +\alias{get_entity_properties,NetworkBackend-method} +\alias{get_entity_properties,IndraBackend-method} +\title{Add a backend's properties of each entity} +\usage{ +get_entity_properties(backend, entities, properties = NULL, ...) + +\S4method{get_entity_properties}{NetworkBackend}(backend, entities, properties = NULL, ...) + +\S4method{get_entity_properties}{IndraBackend}(backend, entities, properties = NULL, ...) +} +\arguments{ +\item{backend}{a \code{NetworkBackend}, e.g. from +\code{\link{indra_backend}()}} + +\item{entities}{entity table, grounded by \code{\link{convert_ids}()}} + +\item{properties}{the properties to add. \code{NULL} adds every +property the backend supports.} + +\item{...}{passed to methods} +} +\value{ +\code{entities} with one column per property +} +\description{ +Adds one column per property, e.g. \code{is_kinase}. A property is +\code{NA} for rows whose \code{entity_type} it doesn't apply to, and for +rows the backend has no answer for. +\code{backend_capabilities(backend)$entity_properties} lists the properties a +backend supports. +} +\details{ +For INDRA, the properties are \code{is_transcription_factor}, +\code{is_kinase}, and \code{is_phosphatase}, looked up by gene symbol. +Only rows with a single HGNC grounding get values. A PTM site gets the +properties of its parent protein. +} +\examples{ +df <- data.frame(Protein = c("P04637", "Q00610")) +indra <- indra_backend() +entities <- prepare_entities(df, entity_type = "protein", + id_type = "uniprot") +entities <- convert_ids(indra, entities) +get_entity_properties(indra, entities, properties = "is_kinase") +} diff --git a/man/get_network.Rd b/man/get_network.Rd new file mode 100644 index 0000000..9417c8d --- /dev/null +++ b/man/get_network.Rd @@ -0,0 +1,130 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/AllGenerics.R, R/backend-indra.R +\name{get_network} +\alias{get_network} +\alias{get_network,NetworkBackend,missing-method} +\alias{get_network,NetworkBackend,NetworkQuery-method} +\alias{get_network,IndraBackend,SubnetworkQuery-method} +\title{Get a network from a backend} +\usage{ +get_network( + backend, + entities, + query = subnetwork_query(), + interaction_types = NULL, + min_evidence = 1, + min_confidence = NULL, + evidence_sources = NULL, + include_entities = NULL, + ... +) + +\S4method{get_network}{NetworkBackend,missing}( + backend, + entities, + query = subnetwork_query(), + interaction_types = NULL, + min_evidence = 1, + min_confidence = NULL, + evidence_sources = NULL, + include_entities = NULL, + ... +) + +\S4method{get_network}{NetworkBackend,NetworkQuery}( + backend, + entities, + query = subnetwork_query(), + interaction_types = NULL, + min_evidence = 1, + min_confidence = NULL, + evidence_sources = NULL, + include_entities = NULL, + ... +) + +\S4method{get_network}{IndraBackend,SubnetworkQuery}( + backend, + entities, + query = subnetwork_query(), + interaction_types = NULL, + min_evidence = 1, + min_confidence = NULL, + evidence_sources = NULL, + include_entities = NULL, + ... +) +} +\arguments{ +\item{backend}{a \code{NetworkBackend}, e.g. from +\code{\link{indra_backend}()}} + +\item{entities}{entity table from \code{\link{prepare_entities}()}, +grounded by \code{\link{convert_ids}()} and flagged by +\code{\link{select_entities}()}} + +\item{query}{a \code{NetworkQuery} saying which question to ask, e.g. +\code{\link{subnetwork_query}()} (the default). See +\code{\link{network_queries}}.} + +\item{interaction_types}{values of \code{edges$interaction} to keep +(INDRA statement types, e.g. \code{"Activation"}). \code{NULL} keeps all.} + +\item{min_evidence}{minimum evidence count per edge} + +\item{min_confidence}{minimum \code{confidence} per edge, in [0, 1]. +Edges with no confidence score (\code{NA}) are dropped too, and a message +says how many. \code{NULL} applies no cutoff.} + +\item{evidence_sources}{keeps edges with evidence from at least one of +these sources, e.g. \code{c("reach")}. \code{NULL} keeps all.} + +\item{include_entities}{\code{"namespace:identifier"} groundings to add to +the query, e.g. \code{"HGNC:1234"}. Use this for entities outside the +input; to keep entities of the input that fail the cutoffs, use +\code{select_entities(force_include = )}.} + +\item{...}{passed to methods} +} +\value{ +list of \code{nodes} and \code{edges} data.frames that meets the +contract checked by \code{\link{validate_network}()} +} +\description{ +Sends the selected entities to a backend and returns the network that +answers the query, as \code{nodes} and \code{edges} tables. +} +\details{ +Only the rows of \code{entities} with \code{included_in_query = TRUE} +are sent to the backend. Every node the backend returns is matched +against all rows, so a node in the input gets \code{measured = TRUE} and +its statistics, and a node not in the input gets \code{measured = FALSE} +and \code{NA} statistics. + +\code{get_network()} prints the question it asks as a message, with the +number of entities, so the query in a saved script or log is readable +without the documentation. + +Confidence values are comparable within one backend, not across +backends. For INDRA, \code{confidence} is the INDRA belief score. +} +\examples{ +input <- data.table::fread(system.file( + "extdata/groupComparisonModel.csv", + package = "MSstatsBioNet" +)) +entities <- prepare_entities(input, entity_type = "protein", + id_type = "uniprot") +\donttest{ +indra <- indra_backend() +entities <- convert_ids(indra, entities) +entities <- select_entities(entities, pvalue_cutoff = 0.05) +network <- get_network(indra, entities, subnetwork_query(), + interaction_types = "Complex") +head(network$nodes) +head(network$edges) +} +} +\seealso{ +\code{\link{network_queries}}, \code{\link{backend_capabilities}()} +} diff --git a/man/indra_backend.Rd b/man/indra_backend.Rd new file mode 100644 index 0000000..a849ca1 --- /dev/null +++ b/man/indra_backend.Rd @@ -0,0 +1,38 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/backend-indra.R +\name{indra_backend} +\alias{indra_backend} +\title{Create an INDRA backend} +\usage{ +indra_backend(cogex_url = INDRA_API_URL, grounding_url = GILDA_API_URL) +} +\arguments{ +\item{cogex_url}{base URL of INDRA CoGEx} + +\item{grounding_url}{base URL of Gilda} +} +\value{ +an \code{IndraBackend} object, to pass to +\code{\link{convert_ids}()}, \code{\link{get_entity_properties}()}, and +\code{\link{get_network}()} +} +\description{ +INDRA is a knowledge graph of mechanisms (activations, phosphorylations, +complexes, ...) assembled from the literature and curated databases. The +backend queries INDRA CoGEx for networks and grounds gene symbols and +chemical names with Gilda, INDRA's grounding service. +\code{backend_capabilities(indra_backend())} lists what it supports. +} +\details{ +This function includes third-party software components that are licensed +under the BSD 2-Clause License. Include the third-party licensing +agreements if redistributing this package or results based on it. See +the LICENSE file for details. +} +\examples{ +indra <- indra_backend() +backend_capabilities(indra)$query_types +} +\seealso{ +\code{\link{NetworkBackend-class}} +} diff --git a/man/network_queries.Rd b/man/network_queries.Rd new file mode 100644 index 0000000..625bb50 --- /dev/null +++ b/man/network_queries.Rd @@ -0,0 +1,51 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/queries.R +\name{network_queries} +\alias{network_queries} +\title{Questions you can ask of a network backend} +\description{ +Each query constructor asks one question of a backend. Pass the query to +\code{\link{get_network}()}. Every query returns the same \code{nodes} +and \code{edges} tables (see \code{\link{validate_network}()}), so +visualization and filtering work the same whichever question produced +the network. The \code{query_type} column of \code{edges} names the +query, without \code{_query}. +} +\details{ +\tabular{llll}{ + \strong{Constructor} \tab \strong{Question} \tab + \strong{Nodes added} \tab \strong{\code{query_type}} \cr + \code{\link{subnetwork_query}()} \tab How are my selected entities + connected to each other, with no other nodes added? \tab none, other + than \code{include_entities} \tab \code{"subnetwork"} \cr +} + +More queries (shared regulators, regulator enrichment, paths, and +others) are planned. \code{backend_capabilities(backend)$query_types} lists the +ones a backend supports. +} +\section{Glossary}{ + +\describe{ + \item{entity}{One row of the entity table from + \code{\link{prepare_entities}()}: one analyte of the input, such as a + protein, a PTM site, or a metabolite.} + \item{selected}{An entity with \code{included_in_query = TRUE}, set by + \code{\link{select_entities}()}: it passed the cutoffs, or was forced + in. Only selected entities are sent to the backend.} + \item{in the input (\code{measured})}{A node that matches a row of the + entity table, whether or not it was selected. It carries that row's + statistics.} + \item{latent}{A node that matches no row of the entity table, such as + an entity added through \code{include_entities}. It has + \code{measured = FALSE} and \code{NA} statistics. It may still have + been measured in another experiment, or filtered out before the + entity table was built.} + \item{grounding}{A \code{"namespace:identifier"} pair that names an + entity in a backend, such as \code{"HGNC:11998"} (TP53).} +} +} + +\seealso{ +\code{\link{get_network}()}, \code{\link{backend_capabilities}()} +} diff --git a/man/prepare_entities.Rd b/man/prepare_entities.Rd new file mode 100644 index 0000000..36a6161 --- /dev/null +++ b/man/prepare_entities.Rd @@ -0,0 +1,85 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/entities.R +\name{prepare_entities} +\alias{prepare_entities} +\title{Prepare an entity table from MSstats results} +\usage{ +prepare_entities( + df, + id_column = "Protein", + entity_type, + id_type, + organism = "9606", + label = NULL, + logfc_column = NULL +) +} +\arguments{ +\item{df}{output of \code{groupComparison()}'s \code{ComparisonResult} +table, or any table with one row per analyte.} + +\item{id_column}{name of the column holding the analyte identifiers.} + +\item{entity_type}{one of the entity types (\code{"protein"}, +\code{"gene"}, \code{"transcript"}, \code{"ptm_site"}, +\code{"metabolite"}, \code{"lipid"}, \code{"drug"}, \code{"complex"}, +\code{"family"}, \code{"other"}), or the name of a column of \code{df} +holding one per row.} + +\item{id_type}{the identifier system of \code{id_column} +(\code{"uniprot"}, \code{"uniprot_mnemonic"}, \code{"hgnc_symbol"}, +\code{"ensembl_protein"}, \code{"ensembl_gene"}, \code{"entrez"}, +\code{"chemical_name"}, \code{"inchikey"}, \code{"hmdb"}, +\code{"chebi"}, \code{"chembl"}), or the name of a column of \code{df} +holding one per row. Which of them a backend can convert is listed by +\code{backend_capabilities(backend)$id_conversions}. For PTM sites, the identifier system of the parent +protein. \code{"chemical_name"} is a metabolite, lipid, or drug name, +common or IUPAC (e.g. \code{"glucose"}), grounded by text matching.} + +\item{organism}{NCBI taxon ID, as a string.} + +\item{label}{the comparison to keep, when \code{df} has a \code{Label} +column with more than one value.} + +\item{logfc_column}{the fold-change column to copy to \code{logFC}. +\code{NULL} uses whichever of \code{log2FC}, \code{log10FC}, or +\code{logFC} \code{df} has. Values are copied unchanged, in the log base +of the input.} +} +\value{ +data.frame with columns \code{id}, \code{entity_type}, +\code{id_type}, \code{namespace}, \code{entity_id}, \code{entity_name} +(all \code{NA} until \code{convert_ids()}), \code{included_in_query} +(\code{TRUE} until \code{select_entities()}), \code{site} and +\code{parent_id} (for \code{ptm_site} rows), \code{organism}, and +\code{logFC} and \code{adj.pvalue} when \code{df} has them. Other +columns of \code{df} are not copied; join them back on \code{id}. +} +\description{ +Builds the table that \code{\link{convert_ids}()} grounds, +\code{\link{select_entities}()} flags, and \code{\link{get_network}()} +queries. It has one row per analyte, and keeps every analyte, not only +the significant ones, so that nodes returned by a backend can be +recognized as being in the input. +} +\details{ +For PTM sites (\code{entity_type = "ptm_site"}), the site is parsed +from the end of each identifier, e.g. \code{"P00533_S1039_S1042"} gives +parent \code{"P00533"} and site \code{"S1039_S1042"}. A +\code{GlobalProtein} column, as MSstatsPTM writes, overrides the parsed +parent. +} +\examples{ +input <- data.table::fread(system.file( + "extdata/groupComparisonModel.csv", + package = "MSstatsBioNet" +)) +entities <- prepare_entities(input, entity_type = "protein", + id_type = "uniprot") +head(entities) + +# MSstatsPTM results: the parent protein and site are parsed from the id +ptm <- data.frame(Protein = c("P00533_S1039_S1042", "P04637_S15"), + log2FC = c(1.2, -0.8), adj.pvalue = c(0.01, 0.2)) +prepare_entities(ptm, entity_type = "ptm_site", id_type = "uniprot") +} diff --git a/man/select_entities.Rd b/man/select_entities.Rd new file mode 100644 index 0000000..96ca4be --- /dev/null +++ b/man/select_entities.Rd @@ -0,0 +1,61 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/entities.R +\name{select_entities} +\alias{select_entities} +\title{Flag the entities to query} +\usage{ +select_entities( + entities, + pvalue_cutoff = NULL, + logfc_cutoff = NULL, + direction = c("both", "up", "down"), + include_infinite_fc = FALSE, + force_include = NULL +) +} +\arguments{ +\item{entities}{entity table from \code{\link{prepare_entities}()}} + +\item{pvalue_cutoff}{keep rows with \code{adj.pvalue} below this. +\code{NULL} applies no cutoff.} + +\item{logfc_cutoff}{keep rows with \code{abs(logFC)} above this, on the +log scale of the input. +\code{NULL} applies no cutoff.} + +\item{direction}{\code{"both"}, \code{"up"} (\code{logFC > 0}), or +\code{"down"} (\code{logFC < 0}).} + +\item{include_infinite_fc}{whether rows with infinite \code{logFC} +(detected in one condition only) are selected regardless of +\code{pvalue_cutoff} and \code{logfc_cutoff}. \code{direction} still +applies.} + +\item{force_include}{values of \code{id}, or \code{"namespace:identifier"} +groundings (e.g. \code{"HGNC:1234"}), selected regardless of the +cutoffs. Entities outside the table are added with +\code{get_network(include_entities = )} instead.} +} +\value{ +\code{entities} with \code{included_in_query} set, and +\code{user_added} (\code{TRUE} for rows selected only through +\code{force_include}). +} +\description{ +Sets \code{included_in_query} from the statistical cutoffs. Only these +rows are sent to the backend by \code{\link{get_network}()}. Rows that +fail are kept, so that \code{get_network()} can still recognize them as +being in the input when a backend returns them. Rows with a missing +\code{adj.pvalue} are not selected. +} +\examples{ +input <- data.table::fread(system.file( + "extdata/groupComparisonModel.csv", + package = "MSstatsBioNet" +)) +entities <- prepare_entities(input, entity_type = "protein", + id_type = "uniprot") +entities <- select_entities(entities, pvalue_cutoff = 0.01, + direction = "up") +table(entities$included_in_query) +} diff --git a/man/subnetwork_query.Rd b/man/subnetwork_query.Rd new file mode 100644 index 0000000..62b341c --- /dev/null +++ b/man/subnetwork_query.Rd @@ -0,0 +1,31 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/queries.R +\name{subnetwork_query} +\alias{subnetwork_query} +\title{How are my selected entities connected to each other?} +\usage{ +subnetwork_query() +} +\value{ +a \code{SubnetworkQuery} object, to pass to +\code{\link{get_network}()} +} +\description{ +Asks which edges connect the selected entities directly, with no other +nodes added. Only entities passed to \code{get_network()} as +\code{include_entities} are added. +} +\details{ +Uses the rows of \code{entities} with \code{included_in_query = TRUE}. +Each node the backend returns is matched against all rows, so that it +carries its statistics. Edges get \code{query_type = "subnetwork"}. +For INDRA, the query goes to CoGEx \code{indra_subnetwork_relations}, and +takes fewer than 400 groundings +(\code{backend_capabilities(indra_backend())$max_nodes}). +} +\examples{ +subnetwork_query() +} +\seealso{ +\code{\link{network_queries}} for the other questions +} diff --git a/tests/testthat/test-backend-indra.R b/tests/testthat/test-backend-indra.R index 6e08ce2..a31c1cb 100644 --- a/tests/testthat/test-backend-indra.R +++ b/tests/testthat/test-backend-indra.R @@ -416,3 +416,114 @@ test_that("annotateProteinInfoFromIndra() grounds through convert_ids() and get_ "EntityId", "EntityName", "IsTranscriptionFactor", "IsKinase", "IsPhosphatase")) }) + +# ----- Exported API (Phase 3d of the API refactor) ----- + +test_that("the new API is exported", { + exports <- getNamespaceExports("MSstatsBioNet") + expect_true(all(c("prepare_entities", "select_entities", "indra_backend", + "convert_ids", "get_entity_properties", + "subnetwork_query", "get_network", + "backend_capabilities") %in% exports)) + # Helpers stay internal until a caller needs them + expect_false(any(c("build_grounding_table", "parse_ptm_sites") %in% exports)) +}) + +test_that("backend_capabilities() describes the INDRA backend", { + capabilities <- backend_capabilities(indra_backend()) + expect_named(capabilities, c("query_types", "id_conversions", + "entity_properties", "interaction_types", + "max_nodes")) + expect_equal(capabilities$query_types, "subnetwork") + expect_equal(capabilities$id_conversions$protein, + c("uniprot", "uniprot_mnemonic", "hgnc_symbol")) + expect_setequal(capabilities$entity_properties, + c("is_transcription_factor", "is_kinase", "is_phosphatase")) + expect_true(all(c("Activation", "Complex") %in% + capabilities$interaction_types)) + expect_equal(capabilities$max_nodes[["subnetwork"]], 399) +}) + +test_that("the subnetwork query accepts max_nodes groundings and no more", { + max_nodes <- backend_capabilities(indra_backend())$max_nodes[["subnetwork"]] + .groundings <- function(n) { + data.frame(namespace = "HGNC", entity_id = as.character(seq_len(n))) + } + expect_silent(MSstatsBioNet:::.validateIndraSubnetworkInput( + .groundings(max_nodes), evidence_sources = NULL, + include_entities = NULL)) + expect_error(MSstatsBioNet:::.validateIndraSubnetworkInput( + .groundings(max_nodes + 1), evidence_sources = NULL, + include_entities = NULL), "less than 400 proteins") +}) + +test_that("backend_capabilities() errors for a backend without a method", { + where <- environment() + setClass("BackendWithoutCapabilities", contains = "NetworkBackend", + where = where) + on.exit(removeClass("BackendWithoutCapabilities", where = where), + add = TRUE) + expect_error(backend_capabilities(new("BackendWithoutCapabilities")), + "BackendWithoutCapabilities does not describe its backend_capabilities") +}) + +test_that("get_network() drops edges below min_confidence", { + .mock_indra_response() + input <- .selected_input() + all_edges <- suppressMessages(get_network(indra_backend(), input))$edges + cutoff <- stats::median(all_edges$confidence) + network <- suppressMessages(get_network(indra_backend(), input, + min_confidence = cutoff)) + expect_true(all(network$edges$confidence >= cutoff)) + expect_equal(nrow(network$edges), sum(all_edges$confidence >= cutoff)) + expect_true(all(network$nodes$id %in% + c(network$edges$source, network$edges$target))) +}) + +test_that("min_confidence drops edges with no confidence score, with a message", { + edges <- data.frame(confidence = c(0.9, NA, 0.2, NA)) + expect_message(kept <- .filter_by_min_confidence(edges, 0.5), + "Dropping 2 edge\\(s\\) with no confidence score") + expect_equal(kept$confidence, 0.9) + expect_identical(.filter_by_min_confidence(edges, NULL), edges) +}) + +test_that("get_network() checks min_confidence before calling INDRA", { + local_mocked_bindings( + .callIndraCogexApi = function(ns, ids, fio, cogex_url) { + stop("INDRA should not be called") + } + ) + input <- .selected_input() + for (bad in list(-0.1, 1.5, "0.5", c(0.1, 0.2), NA_real_)) { + expect_error(suppressMessages(get_network(indra_backend(), input, + min_confidence = bad)), + "min_confidence must be a single number between 0 and 1") + } +}) + +test_that("get_network() prints the question it asks, with counts", { + .mock_indra_response() + input <- .selected_input() + n_selected <- sum(input$included_in_query & !is.na(input$entity_id)) + expect_message( + get_network(indra_backend(), input), + paste0("INDRA subnetwork: how are ", n_selected, " selected proteins ", + "connected to each other, with no other nodes added\\?")) + expect_message( + get_network(indra_backend(), input, include_entities = "HGNC:1097"), + "selected proteins and 1 added entity connected") +}) + +test_that(".describe_entity_count() names the entity types", { + expect_equal(.describe_entity_count(rep("protein", 42), "selected"), + "42 selected proteins") + expect_equal(.describe_entity_count("ptm_site", "selected"), + "1 selected PTM site") + expect_equal(.describe_entity_count(c("family", "family"), "selected"), + "2 selected families") + expect_equal(.describe_entity_count(c("protein", "metabolite"), "selected"), + "2 selected entities") + expect_equal(.describe_entity_count(rep("protein", 1200), "selected"), + "1,200 selected proteins") +})