Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions NAMESPACE
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,9 @@ export(decomposeSubnetworkIntoHierarchicalTopics)
export(deleteEdgeFromNetwork)
export(exportNetworkToHTML)
export(filterSubnetworkByContext)
export(filter_by_curation)
export(getSubnetworkFromIndra)
export(get_curations)
export(get_entity_properties)
export(get_evidence)
export(get_network)
Expand All @@ -30,6 +32,7 @@ exportClasses(NetworkQuery)
exportClasses(SubnetworkQuery)
exportMethods(backend_capabilities)
exportMethods(convert_ids)
exportMethods(get_curations)
exportMethods(get_entity_properties)
exportMethods(get_evidence)
exportMethods(get_network)
Expand Down
14 changes: 14 additions & 0 deletions NEWS.md
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,13 @@ metabolites:
glossary. `subnetwork_query()` is the first; more are planned.
* `get_evidence(backend, edges)` returns the evidence sentences and
PubMed IDs behind each edge, looked up by `statement_id`.
* `get_curations(backend, edges)` counts, for each statement, the
evidence curated as incorrect, and `filter_by_curation(network,
min_evidence = 1)` subtracts it from `evidence_count`, drops the edges
left below `min_evidence` (and edges left with no evidence), and drops
the nodes left without edges. Like `get_evidence()`, it uses the
backend named in each edge's `backend_database` unless `backend` is
given. `indra_backend(curation_url = )` sets the INDRA database URL.
* The S4 classes `NetworkBackend`, `IndraBackend`, `NetworkQuery`, and
`SubnetworkQuery` are exported, so other packages can add backends.
* New function `validate_network()` checks a `list(nodes, edges)` network
Expand Down Expand Up @@ -166,6 +173,13 @@ Like `get_network()`, it now prints the question it asks as a message.
* The context and topic functions get edge evidence through `get_evidence()`.
The INDRA evidence lookup now uses the backend's `cogex_url` in place of a
hard-coded URL. Its output is unchanged.
* `getSubnetworkFromIndra(filter_by_curation = TRUE)` runs
`filter_by_curation()`. It looks up each statement hash once, where it used
to look up every edge, uses the backend's `curation_url`, prints how many
edges it drops, and also drops edges left with no evidence when
`evidence_count_cutoff` is below 1. The curation lookup reads evidence
hashes as character, so two hashes can no longer round to one number and
be counted once.

# MSstatsBioNet 0.99.0

Expand Down
10 changes: 7 additions & 3 deletions R/AllClasses.R
Original file line number Diff line number Diff line change
Expand Up @@ -8,10 +8,13 @@
#' extending \code{NetworkBackend} and writing methods for the generics.
#'
#' \code{IndraBackend} queries INDRA CoGEx for networks and grounds names
#' with Gilda, INDRA's grounding service.
#' with Gilda, INDRA's grounding service. Curations come from the INDRA
#' database.
#'
#' @slot cogex_url base URL of INDRA CoGEx
#' @slot grounding_url base URL of Gilda
#' @slot curation_url base URL of the INDRA database, which holds the
#' curations
#'
#' @seealso \code{\link{indra_backend}()}, \code{\link{backend_capabilities}()}
#' @name NetworkBackend-class
Expand All @@ -24,10 +27,11 @@ setClass("NetworkBackend", representation("VIRTUAL"))

setClass("IndraBackend", contains = "NetworkBackend",
representation(cogex_url = "character",
grounding_url = "character"))
grounding_url = "character",
curation_url = "character"))

setValidity("IndraBackend", function(object) {
for (slot_name in c("cogex_url", "grounding_url")) {
for (slot_name in c("cogex_url", "grounding_url", "curation_url")) {
url <- methods::slot(object, slot_name)
if (length(url) != 1 || is.na(url) || !nzchar(url)) {
return(paste(slot_name, "must be a single non-empty string"))
Expand Down
47 changes: 47 additions & 0 deletions R/AllGenerics.R
Original file line number Diff line number Diff line change
Expand Up @@ -228,6 +228,53 @@ setMethod("get_evidence", "NetworkBackend",
call. = FALSE)
})

#' Get the curations of edges from a backend
#'
#' Curators mark single pieces of evidence for a statement as correct or
#' incorrect. \code{get_curations()} counts, for each statement, the
#' evidence curated as incorrect. INDRA curations come from the INDRA
#' database, with one request per statement. Most users call
#' \code{\link{filter_by_curation}()}, which subtracts these counts from
#' \code{evidence_count}.
#'
#' @inheritParams get_evidence
#' @param edges the \code{edges} of a network from
#' \code{\link{get_network}()}. Needs the column \code{statement_id}.
#' @return data.frame with one row per unique \code{statement_id} and the
#' columns \code{statement_id} (character) and \code{incorrect_count}
#' (integer). A failed request warns and counts as 0. A method for another
#' backend may leave out statements with no curations, which
#' \code{\link{filter_by_curation}()} counts as 0, but must return
#' \code{statement_id} as character: \code{filter_by_curation()} errors
#' otherwise, since a numeric hash can lose precision and match no edge.
#' @seealso \code{\link{filter_by_curation}()}
#' @export
#' @examples
#' \donttest{
#' input <- data.table::fread(system.file(
#' "extdata/groupComparisonModel.csv",
#' package = "MSstatsBioNet"
#' ))
#' indra <- indra_backend()
#' entities <- prepare_entities(input, entity_type = "protein",
#' id_type = "uniprot")
#' entities <- convert_ids(indra, entities)
#' entities <- select_entities(entities, pvalue_cutoff = 0.05)
#' network <- get_network(indra, entities, interaction_types = "Complex")
#' get_curations(indra, head(network$edges, 2))
#' }
setGeneric("get_curations",
function(backend, edges, ...) standardGeneric("get_curations"),
signature = "backend")

#' @rdname get_curations
#' @export
setMethod("get_curations", "NetworkBackend",
function(backend, edges, ...) {
stop(class(backend), " does not support get_curations().",
call. = FALSE)
})

#' What a backend supports
#'
#' Lists the queries, identifier conversions, entity properties, and
Expand Down
44 changes: 44 additions & 0 deletions R/backend-indra-db.R
Original file line number Diff line number Diff line change
@@ -0,0 +1,44 @@
# HTTP calls to the INDRA database (db.indra.bio), a different service from
# CoGEx.

#' Count the evidence of a statement curated as incorrect
#'
#' Curations tagged anything other than \code{"correct"} count as
#' incorrect, once per evidence (\code{source_hash}, read as character so
#' that no two hashes round to the same number). A failed request
#' warns and counts as 0.
#'
#' @param statement_id INDRA statement hash
#' @param curation_url base URL of the INDRA database
#' @return number of evidences curated as incorrect
#' @keywords internal
#' @noRd
#' @importFrom httr GET status_code content
#' @importFrom jsonlite fromJSON
.get_incorrect_curation_count <- function(statement_id,
curation_url = INDRA_DB_URL) {
statement_id <- as.character(statement_id)
url <- file.path(curation_url, "curation/list", statement_id)

tryCatch({
response <- GET(url)
if (status_code(response) == 200) {
# source_hash values are above 2^53, so read them as character:
# as doubles, two different hashes can round to one value
curations <- fromJSON(content(response, "text", encoding = "UTF-8"),
bigint_as_char = TRUE)
if (length(curations) == 0) {
return(0)
}
incorrect_curations <- curations[curations$tag != "correct", ]
length(unique(incorrect_curations$source_hash))
} else {
warning(paste("API request failed for hash", statement_id,
"with status code", status_code(response)))
0
}
}, error = function(e) {
warning(paste("Error processing hash", statement_id, ":", e$message))
0
})
}
43 changes: 38 additions & 5 deletions R/backend-indra.R
Original file line number Diff line number Diff line change
Expand Up @@ -8,12 +8,18 @@ INDRA_API_URL <- "https://discovery.indra.bio"
#' @noRd
GILDA_API_URL <- "https://grounding.indra.bio"

#' Base URL of the INDRA database, which holds the curations
#' @keywords internal
#' @noRd
INDRA_DB_URL <- "https://db.indra.bio"

#' Create an INDRA backend
#'
#' INDRA is a knowledge graph of mechanisms (activations, phosphorylations,
#' complexes, ...) assembled from the literature and curated databases. The
#' backend queries INDRA CoGEx for networks and grounds gene symbols and
#' chemical names with Gilda, INDRA's grounding service.
#' chemical names with Gilda, INDRA's grounding service. Curations, which
#' mark evidence as correct or incorrect, come from the INDRA database.
#' \code{backend_capabilities(indra_backend())} lists what it supports.
#'
#' This function includes third-party software components that are licensed
Expand All @@ -23,18 +29,23 @@ GILDA_API_URL <- "https://grounding.indra.bio"
#'
#' @param cogex_url base URL of INDRA CoGEx
#' @param grounding_url base URL of Gilda
#' @param curation_url base URL of the INDRA database, which holds the
#' curations
#' @return an \code{IndraBackend} object, to pass to
#' \code{\link{convert_ids}()}, \code{\link{get_entity_properties}()}, and
#' \code{\link{get_network}()}
#' \code{\link{convert_ids}()}, \code{\link{get_entity_properties}()},
#' \code{\link{get_network}()}, \code{\link{get_evidence}()}, and
#' \code{\link{get_curations}()}
#' @seealso \code{\link{NetworkBackend-class}}
#' @importFrom methods new
#' @export
#' @examples
#' indra <- indra_backend()
#' backend_capabilities(indra)$query_types
indra_backend <- function(cogex_url = INDRA_API_URL,
grounding_url = GILDA_API_URL) {
new("IndraBackend", cogex_url = cogex_url, grounding_url = grounding_url)
grounding_url = GILDA_API_URL,
curation_url = INDRA_DB_URL) {
new("IndraBackend", cogex_url = cogex_url, grounding_url = grounding_url,
curation_url = curation_url)
}

# INDRA subnetwork query. Sends the groundings of the included_in_query rows
Expand Down Expand Up @@ -108,6 +119,28 @@ setMethod("get_evidence", "IndraBackend",
.build_indra_evidence_table(edges, evidence_by_statement)
})

# INDRA curations. Asks the INDRA database for the curations of each unique
# statement hash, one request per hash, and counts the evidence curated as
# anything other than correct.
#' @rdname get_curations
#' @export
setMethod("get_curations", "IndraBackend",
function(backend, edges, ...) {
if (!"statement_id" %in% names(edges)) {
stop("Missing required columns: statement_id")
}
statement_ids <- unique(as.character(edges$statement_id))
incorrect_counts <- integer(length(statement_ids))
for (i in seq_along(statement_ids)) {
if (i > 1) Sys.sleep(0.1)
incorrect_counts[i] <- as.integer(.get_incorrect_curation_count(
statement_ids[i], backend@curation_url))
}
data.frame(statement_id = statement_ids,
incorrect_count = incorrect_counts,
stringsAsFactors = FALSE)
})

#' Edge columns that get_evidence() copies to its output
#' @keywords internal
#' @noRd
Expand Down
62 changes: 62 additions & 0 deletions R/backend-registry.R
Original file line number Diff line number Diff line change
Expand Up @@ -85,3 +85,65 @@ BACKEND_CONSTRUCTORS <- list(
}
do.call(rbind, evidence)
}

#' Count the incorrect evidence of each edge, from the backend of each edge
#'
#' Calls \code{get_curations()} once per backend, see
#' \code{.split_edges_by_backend()}, and matches the counts back to the
#' edges within each backend, since two backends can share a
#' \code{statement_id}. Statements a backend returns no count for count 0,
#' after \code{.check_curation_table()} has ruled out a \code{statement_id}
#' that can't match, such as a number.
#'
#' @param edges edges data.frame
#' @param backend \code{NULL} or a \code{NetworkBackend}
#' @return integer vector, one count per row of \code{edges}
#' @keywords internal
#' @noRd
.count_incorrect_evidence <- function(edges, backend = NULL) {
incorrect_counts <- integer(nrow(edges))
edges$.edge_row <- seq_len(nrow(edges))
for (group in .split_edges_by_backend(edges, backend)) {
curations <- get_curations(group$backend, group$edges)
.check_curation_table(curations, group$backend)
counts <- curations$incorrect_count[match(
as.character(group$edges$statement_id),
as.character(curations$statement_id))]
counts[is.na(counts)] <- 0L
incorrect_counts[group$edges$.edge_row] <- as.integer(counts)
}
incorrect_counts
}

#' Check the table a get_curations() method returned
#'
#' A numeric \code{statement_id} matches no edge (INDRA hashes above 2^53
#' lose precision as numbers), so every edge would silently count 0
#' incorrect evidence. Errors naming the backend.
#'
#' @param curations output of \code{get_curations()}
#' @param backend the backend that returned it
#' @return \code{NULL}, invisibly; errors otherwise
#' @keywords internal
#' @noRd
.check_curation_table <- function(curations, backend) {
problem <- if (!is.data.frame(curations) ||
!all(c("statement_id", "incorrect_count") %in%
names(curations))) {
"a data.frame with columns statement_id and incorrect_count"
} else if (!is.character(curations$statement_id) ||
anyNA(curations$statement_id)) {
"a character statement_id with no NA"
} else if (!is.numeric(curations$incorrect_count) ||
anyNA(curations$incorrect_count) ||
any(curations$incorrect_count < 0) ||
any(curations$incorrect_count !=
round(curations$incorrect_count))) {
"an incorrect_count of non-negative whole numbers"
}
if (!is.null(problem)) {
stop("get_curations() for ", class(backend), " must return ",
problem, ".", call. = FALSE)
}
invisible(NULL)
}
11 changes: 9 additions & 2 deletions R/getSubnetworkFromIndra.R
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,9 @@
#' network, regardless if those ids are in the input data. Should be formatted
#' as "namespace:identifier", e.g. "HGNC:1234" or "CHEBI:4911".
#' @param filter_by_curation logical, whether to filter out statements that
#' have been curated as incorrect in INDRA. Default is FALSE.
#' have been curated as incorrect in INDRA. Default is FALSE. Runs
#' \code{\link{filter_by_curation}()} with
#' \code{min_evidence = evidence_count_cutoff}.
#' @param filter_by_ptm_site logical, whether to filter edges based on whether the
#' site information from INDRA matches with the PTM site in the input. Default is FALSE.
#' Only applicable for differential PTM abundance results.
Expand Down Expand Up @@ -136,7 +138,12 @@ getSubnetworkFromIndra <- function(input,
edges = edges)
}
subnetwork = .filterByPtmSite(subnetwork$nodes, subnetwork$edges, filter_by_ptm_site)
subnetwork = .filterByCuration(subnetwork$nodes, subnetwork$edges, evidence_count_cutoff, filter_by_curation)
if (filter_by_curation) {
# the call finds the exported function, not this logical argument
subnetwork <- filter_by_curation(subnetwork,
min_evidence = evidence_count_cutoff,
backend = indra_backend())
}
validate_network(subnetwork)
warning(
"NOTICE: This function includes third-party software components
Expand Down
Loading
Loading