Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions NAMESPACE
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,7 @@ export(exportNetworkToHTML)
export(filterSubnetworkByContext)
export(getSubnetworkFromIndra)
export(get_entity_properties)
export(get_evidence)
export(get_network)
export(indra_backend)
export(prepare_entities)
Expand All @@ -30,6 +31,7 @@ exportClasses(SubnetworkQuery)
exportMethods(backend_capabilities)
exportMethods(convert_ids)
exportMethods(get_entity_properties)
exportMethods(get_evidence)
exportMethods(get_network)
importFrom(grDevices,
colorRamp,
Expand Down
12 changes: 12 additions & 0 deletions NEWS.md
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,8 @@ metabolites:
other, with no other nodes added?".
* `?network_queries` describes the questions a query can ask, with a
glossary. `subnetwork_query()` is the first; more are planned.
* `get_evidence(backend, edges)` returns the evidence sentences and
PubMed IDs behind each edge, looked up by `statement_id`.
* The S4 classes `NetworkBackend`, `IndraBackend`, `NetworkQuery`, and
`SubnetworkQuery` are exported, so other packages can add backends.
* New function `validate_network()` checks a `list(nodes, edges)` network
Expand All @@ -45,6 +47,13 @@ fill. `entity_type` sets the node shape: hexagon for metabolites and lipids,
diamond for drugs, octagon for complexes, barrel for families. The legend
shows only the node styles, shapes, and edge types present, and covers
every statement type of the contract.
* `filterSubnetworkByContext()`, `decomposeSubnetworkByTopic()`,
`decomposeSubnetworkIntoHierarchicalTopics()`, `compareTopicModels()`, and
`bootstrapTopicModels()` take a `backend` argument. By default each edge's
evidence comes from the backend named in its `backend_database` column, so a
network subset or rebuilt with `list(nodes =, edges =)` still works. Pass
`backend` to use one built with non-default settings, such as
`indra_backend(cogex_url = )`.

## Breaking changes

Expand Down Expand Up @@ -154,6 +163,9 @@ hard-coded human taxon ID.
the rows to query with `select_entities()`, and passes all rows to
`get_network()`, which matches the nodes INDRA returns against every row.
Like `get_network()`, it now prints the question it asks as a message.
* The context and topic functions get edge evidence through `get_evidence()`.
The INDRA evidence lookup now uses the backend's `cogex_url` in place of a
hard-coded URL. Its output is unchanged.

# MSstatsBioNet 0.99.0

Expand Down
55 changes: 55 additions & 0 deletions R/AllGenerics.R
Original file line number Diff line number Diff line change
Expand Up @@ -173,6 +173,61 @@ setMethod("get_entity_properties", "NetworkBackend",
call. = FALSE)
})

#' Get the evidence behind network edges
#'
#' Looks up the evidence sentences and their PubMed IDs for each edge,
#' using the edge's \code{statement_id}. Edges that share a
#' \code{statement_id} get the same evidence.
#'
#' For INDRA, the evidence comes from CoGEx. Evidence without text is left
#' out, and \code{pmid} is \code{""} for evidence from a source with no
#' PubMed ID.
#'
#' \code{\link{filterSubnetworkByContext}()} and the topic functions call
#' \code{get_evidence()} with the backend named in each edge's
#' \code{backend_database}, unless their \code{backend} argument is given.
#'
#' @param backend a \code{NetworkBackend}, e.g. from
#' \code{\link{indra_backend}()}
#' @param edges the \code{edges} of a network from
#' \code{\link{get_network}()}. Needs the columns \code{source},
#' \code{target}, \code{interaction}, \code{site}, \code{evidence_url}, and
#' \code{statement_id}.
#' @param ... passed to methods
#' @return data.frame with one row per (edge, evidence sentence) pair and
#' the columns \code{source}, \code{target}, \code{interaction},
#' \code{site}, \code{evidence_url}, \code{statement_id}, \code{text}, and
#' \code{pmid}. It has no rows, with a warning, when no edge has evidence
#' text.
#' @seealso \code{\link{filterSubnetworkByContext}()}
#' @export
#' @examples
#' \donttest{
#' input <- data.table::fread(system.file(
#' "extdata/groupComparisonModel.csv",
#' package = "MSstatsBioNet"
#' ))
#' indra <- indra_backend()
#' entities <- prepare_entities(input, entity_type = "protein",
#' id_type = "uniprot")
#' entities <- convert_ids(indra, entities)
#' entities <- select_entities(entities, pvalue_cutoff = 0.05)
#' network <- get_network(indra, entities, interaction_types = "Complex")
#' evidence <- get_evidence(indra, head(network$edges, 2))
#' head(evidence[, c("source", "target", "pmid", "text")])
#' }
setGeneric("get_evidence",
function(backend, edges, ...) standardGeneric("get_evidence"),
signature = "backend")

#' @rdname get_evidence
#' @export
setMethod("get_evidence", "NetworkBackend",
function(backend, edges, ...) {
stop(class(backend), " does not support get_evidence().",
call. = FALSE)
})

#' What a backend supports
#'
#' Lists the queries, identifier conversions, entity properties, and
Expand Down
69 changes: 69 additions & 0 deletions R/backend-indra-cogex.R
Original file line number Diff line number Diff line change
Expand Up @@ -480,3 +480,72 @@
})
return(res)
}


#' Query INDRA CoGEx for the evidence of statements
#'
#' Hashes that are missing from the response (unknown hashes, or a
#' batch whose request failed) are absent from the returned list.
#'
#' @param stmt_hashes Character vector of statement hash strings
#' @param cogex_url base URL of INDRA CoGEx
#' @param batch_size Number of hashes to request per API call
#' @param sleep Seconds to pause between API calls
#' @return Named list: statement_id -> list of evidence objects. Empty list when
#' no evidence could be retrieved.
#' @keywords internal
#' @noRd
#' @importFrom httr POST status_code content content_type_json
.query_indra_evidence <- function(stmt_hashes, cogex_url = INDRA_API_URL,
batch_size = 100, sleep = 1) {
url <- file.path(cogex_url, "api/get_evidences_for_stmt_hashes")

stmt_hashes <- unique(as.character(stmt_hashes))
if (length(stmt_hashes) == 0) return(list())

batches <- split(stmt_hashes, ceiling(seq_along(stmt_hashes) / batch_size))
n_batches <- length(batches)
results <- list()

cat(sprintf("Fetching evidence for %d hashes in %d batch(es) of up to %d...\n",
length(stmt_hashes), n_batches, batch_size))

for (i in seq_along(batches)) {
batch <- batches[[i]]

parsed <- tryCatch({
response <- POST(
url,
body = list(stmt_hashes = I(batch)),
encode = "json",
content_type_json()
)

if (status_code(response) != 200) {
warning(sprintf("API returned status %d for stmt_hashes: %s",
status_code(response),
paste(batch, collapse = ", ")))
NULL
} else {
content(response, as = "parsed")
}
}, error = function(e) {
warning(sprintf("Error querying stmt_hashes %s: %s",
paste(batch, collapse = ", "), e$message))
return(NULL)
})

cat(sprintf("Progress: %d/%d batches (%.1f%%)\n",
i, n_batches, (i / n_batches) * 100))

if (!is.null(parsed)) {
matched <- intersect(names(parsed), batch)
results[matched] <- parsed[matched]
}

if (i < n_batches) Sys.sleep(sleep)
}

cat("Done fetching evidence!\n")
results
}
96 changes: 96 additions & 0 deletions R/backend-indra.R
Original file line number Diff line number Diff line change
Expand Up @@ -91,6 +91,102 @@ setMethod("backend_capabilities", "IndraBackend",
max_nodes = INDRA_MAX_NODES)
})

# INDRA evidence. Sends the unique statement hashes to CoGEx
# get_evidences_for_stmt_hashes and copies each evidence sentence onto every
# edge row with that hash. Network Search edges carry the same hashes, so
# this covers every INDRA query type.
#' @rdname get_evidence
#' @export
setMethod("get_evidence", "IndraBackend",
function(backend, edges, ...) {
.check_evidence_edge_columns(edges)
statement_ids <- unique(edges$statement_id)
cat(sprintf("Processing %d unique statement hashes...\n",
length(statement_ids)))
evidence_by_statement <- .query_indra_evidence(statement_ids,
backend@cogex_url)
.build_indra_evidence_table(edges, evidence_by_statement)
})

#' Edge columns that get_evidence() copies to its output
#' @keywords internal
#' @noRd
EVIDENCE_EDGE_COLUMNS <- c("source", "target", "interaction", "site",
"evidence_url", "statement_id")

#' Check that edges have the columns get_evidence() needs
#' @param edges edges data.frame
#' @return \code{NULL}, invisibly; errors naming the missing columns
#' @keywords internal
#' @noRd
.check_evidence_edge_columns <- function(edges) {
missing_cols <- setdiff(EVIDENCE_EDGE_COLUMNS, names(edges))
if (length(missing_cols) > 0) {
stop(sprintf("Missing required columns: %s",
paste(missing_cols, collapse = ", ")))
}
invisible(NULL)
}

#' Build the evidence table from the CoGEx evidence of each statement
#' @param edges edges data.frame with \code{EVIDENCE_EDGE_COLUMNS}
#' @param evidence_by_statement named list from \code{.query_indra_evidence()}
#' @return data.frame with \code{EVIDENCE_EDGE_COLUMNS}, \code{text}, and
#' \code{pmid}; one row per (edge, evidence with text) pair
#' @keywords internal
#' @noRd
.build_indra_evidence_table <- function(edges, evidence_by_statement) {
results_list <- list()
result_count <- 0

for (statement_id in unique(edges$statement_id)) {
evidence_list <- evidence_by_statement[[as.character(statement_id)]]
if (is.null(evidence_list) || length(evidence_list) == 0) next

matching_indices <- which(edges$statement_id == statement_id)

for (evidence in evidence_list) {
if (!is.null(evidence[["text"]]) && nchar(evidence[["text"]]) > 0) {
for (idx in matching_indices) {
result_count <- result_count + 1
results_list[[result_count]] <- data.frame(
source = edges$source[idx],
target = edges$target[idx],
interaction = edges$interaction[idx],
site = edges$site[idx],
evidence_url = edges$evidence_url[idx],
statement_id = edges$statement_id[idx],
text = evidence[["text"]],
pmid = if (is.null(evidence[["pmid"]])) "" else evidence[["pmid"]],
stringsAsFactors = FALSE
)
}
}
}
}

if (result_count == 0) {
warning("No evidence text found for any statement hash")
return(.build_empty_evidence_table())
}

results_df <- do.call(rbind, results_list)
cat(sprintf("\nComplete! Found %d evidence text entries.\n", nrow(results_df)))
results_df
}

#' Evidence table with no rows
#' @return data.frame with the get_evidence() columns and no rows
#' @keywords internal
#' @noRd
.build_empty_evidence_table <- function() {
data.frame(
source = character(), target = character(), interaction = character(),
site = character(), evidence_url = character(), statement_id = character(),
text = character(), pmid = character(), stringsAsFactors = FALSE
)
}

#' The question a subnetwork query asks, with the entity counts
#'
#' Printed by \code{get_network()}, e.g. "INDRA subnetwork: how are 42
Expand Down
87 changes: 87 additions & 0 deletions R/backend-registry.R
Original file line number Diff line number Diff line change
@@ -0,0 +1,87 @@
#' Default backend for each backend_database value
#'
#' Used when a function that takes a finished network is called with
#' \code{backend = NULL}: each edge goes to the backend named in its
#' \code{backend_database}, built with default settings.
#' @keywords internal
#' @noRd
BACKEND_CONSTRUCTORS <- list(
INDRA = function() indra_backend()
)

#' Check a backend argument
#' @param backend \code{NULL} or a \code{NetworkBackend}
#' @return \code{NULL}, invisibly; errors otherwise
#' @keywords internal
#' @noRd
.check_backend_argument <- function(backend) {
if (!is.null(backend) && !methods::is(backend, "NetworkBackend")) {
stop("`backend` must be NULL or a NetworkBackend, e.g. from ",
"indra_backend().", call. = FALSE)
}
invisible(NULL)
}

#' Split edges by the backend that answers for them
#'
#' With \code{backend = NULL}, the edges are split by
#' \code{backend_database} and each group gets that value's default backend
#' from \code{BACKEND_CONSTRUCTORS}. A given \code{backend} takes all edges.
#'
#' @param edges edges data.frame
#' @param backend \code{NULL} or a \code{NetworkBackend}
#' @return list of \code{list(backend =, edges =)}, one per backend, in
#' order of first appearance in \code{edges}
#' @keywords internal
#' @noRd
.split_edges_by_backend <- function(edges, backend = NULL) {
.check_backend_argument(backend)
if (!is.null(backend)) {
return(list(list(backend = backend, edges = edges)))
}
if (!"backend_database" %in% names(edges)) {
stop("`edges` has no `backend_database` column, so the backend ",
"can't be chosen. Pass `backend`, e.g. ",
"`backend = indra_backend()`.", call. = FALSE)
}
databases <- as.character(edges$backend_database)
if (anyNA(databases) || any(!nzchar(databases))) {
stop("`backend_database` is missing for some edges. Pass ",
"`backend`, e.g. `backend = indra_backend()`.", call. = FALSE)
}
unknown <- setdiff(unique(databases), names(BACKEND_CONSTRUCTORS))
if (length(unknown) > 0) {
stop("No default backend for backend_database ",
paste0("\"", unknown, "\"", collapse = ", "),
". Pass `backend` to choose one.", call. = FALSE)
}
lapply(unique(databases), function(database) {
list(backend = BACKEND_CONSTRUCTORS[[database]](),
edges = edges[databases == database, , drop = FALSE])
})
}

#' Get the evidence of edges from the backend of each edge
#'
#' Calls \code{get_evidence()} once per backend, see
#' \code{.split_edges_by_backend()}, and stacks the results.
#'
#' @param edges edges data.frame
#' @param backend \code{NULL} or a \code{NetworkBackend}
#' @return evidence data.frame, as from \code{get_evidence()}
#' @keywords internal
#' @noRd
.fetch_evidence <- function(edges, backend = NULL) {
groups <- .split_edges_by_backend(edges, backend)
if (length(groups) == 0) {
warning("No evidence text found for any statement hash")
return(.build_empty_evidence_table())
}
evidence <- lapply(groups, function(group) {
get_evidence(group$backend, group$edges)
})
if (length(evidence) == 1) {
return(evidence[[1]])
}
do.call(rbind, evidence)
}
Loading
Loading