From dde9886dcfa351daa068b6d56f53c0a539c5a08c Mon Sep 17 00:00:00 2001 From: David Schoch Date: Tue, 29 Sep 2026 21:25:21 +0200 Subject: [PATCH 1/2] feat: `graph_from_data_frame(vertex_ids = TRUE)` reads numeric vertex IDs, so `as_data_frame(what = "both")` round-trips unnamed graphs Closes #223. Co-Authored-By: Claude Opus 5.5 (1M context) --- R/conversion.R | 113 ++++++++++++++++++++++++---- tests/testthat/_snaps/conversion.md | 47 ++++++++++++ tests/testthat/test-conversion.R | 91 ++++++++++++++++++++++ tools/migrations/conversion.R | 3 +- 4 files changed, 237 insertions(+), 17 deletions(-) diff --git a/R/conversion.R b/R/conversion.R index 53ced795d5d..c4fd9b90839 100644 --- a/R/conversion.R +++ b/R/conversion.R @@ -1920,6 +1920,17 @@ graph.data.frame <- function(d, directed = TRUE, vertices = NULL) { #' symbolic edge list given in `d` is checked to contain only vertex names #' listed in `vertices`. #' +#' If `vertex_ids` is `TRUE`, the first two columns of `d` are interpreted as +#' numeric vertex IDs instead of symbolic vertex names. +#' In this case, the rows of `vertices` correspond to the vertices in the order +#' of their IDs, all columns of `vertices` are added as vertex attributes, and +#' no \sQuote{`name`} attribute is created unless `vertices` has a `name` +#' column. +#' If `vertices` is `NULL`, the number of vertices is the largest vertex ID in +#' `d`. +#' This is the inverse of `as_data_frame(what = "both")` for graphs without +#' vertex names. +#' #' Typically, the data frames are exported from some spreadsheet software like #' Excel and are imported into R via [read.table()], #' [read.delim()] or [read.csv()]. @@ -1955,6 +1966,9 @@ graph.data.frame <- function(d, directed = TRUE, vertices = NULL) { #' @param vertices A data frame with vertex metadata, or `NULL`. See #' details below. Since version 0.7 this argument is coerced to a data frame #' with `as.data.frame`, if not `NULL`. +#' @param vertex_ids Logical, whether the first two columns of `d` contain +#' numeric vertex IDs rather than symbolic vertex names. +#' See details below. #' @return An igraph graph object for `graph_from_data_frame()`, and either a #' data frame or a list of two data frames named `edges` and #' `vertices` for `as.data.frame`. @@ -2001,16 +2015,32 @@ graph.data.frame <- function(d, directed = TRUE, vertices = NULL) { #' as_data_frame(g, what = "vertices") #' as_data_frame(g, what = "edges") #' +#' ## Round trip for a graph without vertex names, +#' ## including the isolated vertex 5 +#' g2 <- make_graph(c(1, 2, 2, 3, 3, 4, 4, 1), n = 5, directed = FALSE) +#' V(g2)$color <- c("red", "green", "blue", "red", "green") +#' df <- as_data_frame(g2, what = "both") +#' g3 <- graph_from_data_frame( +#' df$edges, +#' directed = FALSE, +#' vertices = df$vertices, +#' vertex_ids = TRUE +#' ) +#' identical_graphs(g2, g3) +#' #' @export graph_from_data_frame <- function( d, directed = TRUE, ..., - vertices = NULL + vertices = NULL, + vertex_ids = FALSE ) { # BEGIN GENERATED ARG_HANDLE: graph_from_data_frame, do not edit, see tools/generate-migrations.R # fmt: skip if (...length() > 0L) { + .arg_ambiguous <- base::intersect(base::names(base::substitute(...())), base::c("v", "ve", "ver", "vert")) + if (base::length(.arg_ambiguous) > 0L) cli::cli_abort("Argument {.arg {(.arg_ambiguous[[1L]])}} matches multiple arguments of {.fn graph_from_data_frame}.") # Pre-3.0.0 signature: graph_from_data_frame(d, directed, vertices) .old_signature <- function(vertices, ...) { if (...length() > 0L) { @@ -2043,6 +2073,8 @@ graph_from_data_frame <- function( } # END GENERATED ARG_HANDLE + check_bool(vertex_ids) + d <- as.data.frame(d) if (!is.null(vertices)) { vertices <- as.data.frame(vertices) @@ -2055,28 +2087,36 @@ graph_from_data_frame <- function( ## Handle if some elements are 'NA' (first two columns are interpreted as from/to) ensure_no_na(d[, 1:2], "edge data frame") - if (!is.null(vertices) && anyNA(vertices[, 1])) { - cli::cli_warn( - "In {.code vertices[,1]}, {.code NA} elements were replaced with string {.str NA}." - ) - vertices[, 1][is.na(vertices[, 1])] <- "NA" + if (vertex_ids) { + return(graph_from_data_frame_ids(d, directed, vertices)) } names <- unique(c(as.character(d[, 1]), as.character(d[, 2]))) if (!is.null(vertices)) { names2 <- names - vertices <- as.data.frame(vertices) if (ncol(vertices) < 1) { - cli::cli_abort("{.arg vertices} contains no rows") + cli::cli_abort(c( + "{.arg vertices} contains no columns.", + i = "Use {.code vertex_ids = TRUE} if {.arg d} contains numeric vertex IDs." + )) + } + if (anyNA(vertices[, 1])) { + cli::cli_warn( + "In {.code vertices[,1]}, {.code NA} elements were replaced with string {.str NA}." + ) + vertices[, 1][is.na(vertices[, 1])] <- "NA" } names <- as.character(vertices[, 1]) if (anyDuplicated(names) > 0) { cli::cli_abort("{.arg vertices} contains duplicated vertex names") } if (!all(names2 %in% names)) { - cli::cli_abort( - "Some vertex names in {.arg d} are not listed in {.arg vertices}" - ) + cli::cli_abort(c( + "Some vertex names in {.arg d} are not listed in {.arg vertices}", + i = if (looks_like_vertex_ids(d, nrow(vertices))) { + "Use {.code vertex_ids = TRUE} if {.arg d} contains numeric vertex IDs." + } + )) } } @@ -2103,17 +2143,58 @@ graph_from_data_frame <- function( edges <- rbind(match(from, names), match(to, names)) # edge attributes + attrs <- edge_attrs_from_data_frame(d) + + # add the edges + g <- add_edges(g, edges, attr = attrs) + g +} + +# `vertex_ids = TRUE`: the first two columns of `d` are vertex IDs, +# and every column of `vertices` is a vertex attribute, so no `name` is created. +# This is the inverse of `as_data_frame(what = "both")` for unnamed graphs. +graph_from_data_frame_ids <- function( + d, + directed, + vertices, + call = caller_env() +) { + ids <- c(d[, 1], d[, 2]) + if (!is.numeric(ids) || any(ids < 1) || any(ids != round(ids))) { + cli::cli_abort( + "The first two columns of {.arg d} must contain positive whole numbers if {.code vertex_ids = TRUE}.", + call = call + ) + } + + n <- if (is.null(vertices)) max(ids, 0) else nrow(vertices) + if (any(ids > n)) { + cli::cli_abort( + "Some vertex IDs in {.arg d} are larger than the number of rows in {.arg vertices} ({n}).", + call = call + ) + } + + g <- make_empty_graph(n = 0, directed = directed) + g <- add_vertices(g, n, attr = as.list(vertices)) + + edges <- rbind(d[, 1], d[, 2]) + add_edges(g, edges, attr = edge_attrs_from_data_frame(d)) +} + +edge_attrs_from_data_frame <- function(d) { attrs <- list() if (ncol(d) > 2) { for (i in 3:ncol(d)) { - newval <- d[, i] - attrs[[names(d)[i]]] <- newval + attrs[[names(d)[i]]] <- d[, i] } } + attrs +} - # add the edges - g <- add_edges(g, edges, attr = attrs) - g +looks_like_vertex_ids <- function(d, n) { + ids <- c(d[, 1], d[, 2]) + is.numeric(ids) && all(ids >= 1 & ids <= n & ids == round(ids)) } #' Constructor specifications for `graph_()`, `make_()` and `sample_()` diff --git a/tests/testthat/_snaps/conversion.md b/tests/testthat/_snaps/conversion.md index b3d8924221b..8030b279477 100644 --- a/tests/testthat/_snaps/conversion.md +++ b/tests/testthat/_snaps/conversion.md @@ -152,6 +152,53 @@ 2 2 3 2 b B c C 3 1 3 1 a A c C +# graph_from_data_frame(vertex_ids = TRUE) errors + + Code + graph_from_data_frame(data.frame(from = "a", to = "b"), vertex_ids = TRUE) + Condition + Error in `graph_from_data_frame()`: + ! The first two columns of `d` must contain positive whole numbers if `vertex_ids = TRUE`. + Code + graph_from_data_frame(data.frame(from = 1.5, to = 2), vertex_ids = TRUE) + Condition + Error in `graph_from_data_frame()`: + ! The first two columns of `d` must contain positive whole numbers if `vertex_ids = TRUE`. + Code + graph_from_data_frame(data.frame(from = 1, to = 4), vertices = data.frame(a = 1: + 3), vertex_ids = TRUE) + Condition + Error in `graph_from_data_frame()`: + ! Some vertex IDs in `d` are larger than the number of rows in `vertices` (3). + Code + graph_from_data_frame(data.frame(from = 1, to = 2), vertex_ids = NA) + Condition + Error in `graph_from_data_frame()`: + ! `vertex_ids` must be `TRUE` or `FALSE`, not `NA`. + +# graph_from_data_frame() hints at `vertex_ids` for numeric edges + + Code + graph_from_data_frame(data.frame(from = 2, to = 3), vertices = data.frame(a = letters[ + 1:4])) + Condition + Error in `graph_from_data_frame()`: + ! Some vertex names in `d` are not listed in `vertices` + i Use `vertex_ids = TRUE` if `d` contains numeric vertex IDs. + Code + graph_from_data_frame(data.frame(from = 1, to = 2), vertices = data.frame( + row.names = 1:2)) + Condition + Error in `graph_from_data_frame()`: + ! `vertices` contains no columns. + i Use `vertex_ids = TRUE` if `d` contains numeric vertex IDs. + Code + graph_from_data_frame(data.frame(from = "x", to = "y"), vertices = data.frame( + name = c("a", "b"))) + Condition + Error in `graph_from_data_frame()`: + ! Some vertex names in `d` are not listed in `vertices` + # graph_from_edgelist errors for NAs Code diff --git a/tests/testthat/test-conversion.R b/tests/testthat/test-conversion.R index b5014209aba..a0713c3c1db 100644 --- a/tests/testthat/test-conversion.R +++ b/tests/testthat/test-conversion.R @@ -851,6 +851,97 @@ test_that("graph_from_data_frame works on matrices", { expect_equal(as.data.frame(el), el2, ignore_attr = TRUE) }) +test_that("graph_from_data_frame(vertex_ids = TRUE) round-trips unnamed graphs (#223)", { + g <- make_graph(~ A, B - -C, C - -D) + V(g)$a <- letters[1:4] + E(g)$w <- c(1.5, 2.5) + g <- delete_vertex_attr(g, "name") + + df <- as_data_frame(g, what = "both") + r <- graph_from_data_frame( + df$edges, + directed = FALSE, + vertices = df$vertices, + vertex_ids = TRUE + ) + + expect_identical_graphs(g, r) + expect_false(is_named(r)) + expect_identical(V(r)$a, letters[1:4]) + expect_identical(E(r)$w, c(1.5, 2.5)) + expect_identical(as_data_frame(r, what = "both"), df) +}) + +test_that("graph_from_data_frame(vertex_ids = TRUE) works without attributes", { + # Graph without any attributes and with an isolated vertex + g <- make_empty_graph(6) + g <- add_edges(g, c(1, 2, 2, 3, 3, 1, 4, 5)) + + df <- as_data_frame(g, what = "both") + r <- graph_from_data_frame( + df$edges, + vertices = df$vertices, + vertex_ids = TRUE + ) + expect_identical_graphs(g, r) + expect_equal(vcount(r), 6) +}) + +test_that("graph_from_data_frame(vertex_ids = TRUE) infers the vertex count", { + d <- data.frame(from = c(2, 3), to = c(3, 5)) + g <- graph_from_data_frame(d, vertex_ids = TRUE) + expect_equal(vcount(g), 5) + expect_false(is_named(g)) + expect_equal(as_edgelist(g), cbind(c(2, 3), c(3, 5))) +}) + +test_that("graph_from_data_frame(vertex_ids = TRUE) keeps a `name` column", { + d <- data.frame(from = 1, to = 2) + v <- data.frame(name = c("x", "y"), size = 1:2) + g <- graph_from_data_frame(d, vertices = v, vertex_ids = TRUE) + expect_identical(V(g)$name, c("x", "y")) + expect_identical(V(g)$size, 1:2) +}) + +test_that("graph_from_data_frame(vertex_ids = TRUE) errors", { + expect_snapshot(error = TRUE, { + graph_from_data_frame( + data.frame(from = "a", to = "b"), + vertex_ids = TRUE + ) + graph_from_data_frame( + data.frame(from = 1.5, to = 2), + vertex_ids = TRUE + ) + graph_from_data_frame( + data.frame(from = 1, to = 4), + vertices = data.frame(a = 1:3), + vertex_ids = TRUE + ) + graph_from_data_frame( + data.frame(from = 1, to = 2), + vertex_ids = NA + ) + }) +}) + +test_that("graph_from_data_frame() hints at `vertex_ids` for numeric edges", { + expect_snapshot(error = TRUE, { + graph_from_data_frame( + data.frame(from = 2, to = 3), + vertices = data.frame(a = letters[1:4]) + ) + graph_from_data_frame( + data.frame(from = 1, to = 2), + vertices = data.frame(row.names = 1:2) + ) + graph_from_data_frame( + data.frame(from = "x", to = "y"), + vertices = data.frame(name = c("a", "b")) + ) + }) +}) + test_that("edge names work", { ## named edges local_igraph_options(print.edge.attributes = TRUE) diff --git a/tools/migrations/conversion.R b/tools/migrations/conversion.R index 05e91040159..328c114e895 100644 --- a/tools/migrations/conversion.R +++ b/tools/migrations/conversion.R @@ -147,7 +147,8 @@ migrations <- list( d, directed = TRUE, ..., - vertices = NULL + vertices = NULL, + vertex_ids = FALSE ) {}, when = "3.0.0" ), From efeafabf2fa11d1666c697385724cb8773ac686e Mon Sep 17 00:00:00 2001 From: schochastics Date: Tue, 29 Sep 2026 19:35:17 +0000 Subject: [PATCH 2/2] chore: Auto-update from GitHub Actions Run: https://github.com/igraph/rigraph/actions/runs/36619162670 --- man/graph_from_data_frame.Rd | 36 +++++++++++++++++++++++++++++++++++- 1 file changed, 35 insertions(+), 1 deletion(-) diff --git a/man/graph_from_data_frame.Rd b/man/graph_from_data_frame.Rd index 3be6d13a4c3..564dfba84c1 100644 --- a/man/graph_from_data_frame.Rd +++ b/man/graph_from_data_frame.Rd @@ -7,7 +7,13 @@ \usage{ as_data_frame(x, what = c("edges", "vertices", "both")) -graph_from_data_frame(d, directed = TRUE, ..., vertices = NULL) +graph_from_data_frame( + d, + directed = TRUE, + ..., + vertices = NULL, + vertex_ids = FALSE +) } \arguments{ \item{x}{An igraph object.} @@ -27,6 +33,10 @@ version 0.7 this argument is coerced to a data frame with \item{vertices}{A data frame with vertex metadata, or \code{NULL}. See details below. Since version 0.7 this argument is coerced to a data frame with \code{as.data.frame}, if not \code{NULL}.} + +\item{vertex_ids}{Logical, whether the first two columns of \code{d} contain +numeric vertex IDs rather than symbolic vertex names. +See details below.} } \value{ An igraph graph object for \code{graph_from_data_frame()}, and either a @@ -54,6 +64,17 @@ additional vertex attributes. If \code{vertices} is not \code{NULL} then the symbolic edge list given in \code{d} is checked to contain only vertex names listed in \code{vertices}. +If \code{vertex_ids} is \code{TRUE}, the first two columns of \code{d} are interpreted as +numeric vertex IDs instead of symbolic vertex names. +In this case, the rows of \code{vertices} correspond to the vertices in the order +of their IDs, all columns of \code{vertices} are added as vertex attributes, and +no \sQuote{\code{name}} attribute is created unless \code{vertices} has a \code{name} +column. +If \code{vertices} is \code{NULL}, the number of vertices is the largest vertex ID in +\code{d}. +This is the inverse of \code{as_data_frame(what = "both")} for graphs without +vertex names. + Typically, the data frames are exported from some spreadsheet software like Excel and are imported into R via \code{\link[=read.table]{read.table()}}, \code{\link[=read.delim]{read.delim()}} or \code{\link[=read.csv]{read.csv()}}. @@ -123,6 +144,19 @@ print(g, e = TRUE, v = TRUE) as_data_frame(g, what = "vertices") as_data_frame(g, what = "edges") +## Round trip for a graph without vertex names, +## including the isolated vertex 5 +g2 <- make_graph(c(1, 2, 2, 3, 3, 4, 4, 1), n = 5, directed = FALSE) +V(g2)$color <- c("red", "green", "blue", "red", "green") +df <- as_data_frame(g2, what = "both") +g3 <- graph_from_data_frame( + df$edges, + directed = FALSE, + vertices = df$vertices, + vertex_ids = TRUE +) +identical_graphs(g2, g3) + } \seealso{ \code{\link[=graph_from_literal]{graph_from_literal()}}