From 067465ab19fd1954477a56de0dee75de3634ef36 Mon Sep 17 00:00:00 2001 From: Tam Nguyen Duc <1218621+tamnd@users.noreply.github.com> Date: Sun, 16 Aug 2026 06:03:19 +0700 Subject: [PATCH] Copy a graph block for block, which is what AS COPY OF asked for A graph created AS COPY OF another one used to be refused when the source held tables, because a props directory names the blocks its columns live in and a copy that walked them wrong would be a copy that read as data. It walks them right now, and the mirror of free_graph_storage is what says which ones there are: a table per table in the catalog, keeping the name and taking a new id, and per id the segments the old id addressed, read whole and written into blocks of their own. The copy is by value and shares nothing with the source. A copy that pointed at the same segments would be a second name for one graph and the first write to either would show up in both. Every segment block is copied byte for byte, so it costs a read and a write per block and no decode, and what a column holds is none of the copy's business: the labels, the validity, the key index, the CSR pair per group, the edge properties and the tombstones a node table has all carry over the same way. The directories are re-encoded, because a directory names the blocks its segments live in and those are now different blocks. The statistics carry over too, since the copy holds the same rows and gathering them again would read every column for numbers the file already has. AS COPY OF now also takes CURRENT_PROPERTY_GRAPH, which is the graph the statement is against, the same thing a USE clause names. That is what a copy of a loaded file means, and it is what the gql-compat case for GG05 uses, since what the working graph is called is the adapter's business. Nothing is published along the way. The table index and the statistics are staged and the checkpoint that stores the catalog makes all three visible together, so a copy that fails halfway leaves the file exactly as it was. A graph asked to be replaced by a copy of itself is refused rather than emptied, because a replacement frees what the old graph held before it writes the new one. optional/gg05/graph-as-copy-of-holds-the-rows passes, so the corpus is at 225 passing, was 224. --- crates/zu-query/src/ast.rs | 5 +- crates/zu-query/src/parser.rs | 36 +++++- crates/zu-zu1/src/catalog.rs | 68 +++++++++++ crates/zu-zu1/src/graph.rs | 207 ++++++++++++++++++++++++++++++++++ crates/zu-zu1/src/props.rs | 17 +++ crates/zu/src/catalog_stmt.rs | 70 ++++++++---- crates/zu/src/session.rs | 83 ++++++++++++-- docs/07-query-engine.md | 4 +- docs/api/model.json | 15 +++ docs/conformance/zu.json | 6 +- docs/gql-conformance.md | 8 +- docs/grammar.ebnf | 7 +- 12 files changed, 481 insertions(+), 45 deletions(-) diff --git a/crates/zu-query/src/ast.rs b/crates/zu-query/src/ast.rs index 8657a862..d645dc6d 100644 --- a/crates/zu-query/src/ast.rs +++ b/crates/zu-query/src/ast.rs @@ -56,8 +56,9 @@ pub enum CatalogStmt { if_not_exists: bool, or_replace: bool, of: GraphTypeRef, - /// GG05: the graph whose contents the new one starts with. - copy_of: Option, + /// GG05: the graph whose contents the new one starts with, + /// which may be the graph the statement is against. + copy_of: Option, }, DropGraph { name: GraphName, diff --git a/crates/zu-query/src/parser.rs b/crates/zu-query/src/parser.rs index 106efefa..73199fcf 100644 --- a/crates/zu-query/src/parser.rs +++ b/crates/zu-query/src/parser.rs @@ -313,7 +313,7 @@ impl Parser<'_> { self.expect_kw("AS")?; self.expect_kw("COPY")?; self.expect_kw("OF")?; - Some(self.expect_name("the graph a copy is taken from")?) + Some(self.parse_graph_ref()?) } else { None }; @@ -777,17 +777,23 @@ impl Parser<'_> { if !self.eat_kw("USE") { return Ok(None); } + Ok(Some(self.parse_graph_ref()?)) + } + + /// The graph a clause names, which is the same thing written in a + /// `USE` clause and after `AS COPY OF`. + fn parse_graph_ref(&mut self) -> Result { if self.eat_kw("CURRENT_PROPERTY_GRAPH") || self.eat_kw("CURRENT_GRAPH") { - return Ok(Some(GraphRef::Current)); + return Ok(GraphRef::Current); } - // `PROPERTY GRAPH` before the name is the long spelling of the - // same clause and says nothing the name does not. + // `PROPERTY GRAPH` before the name is the long spelling and + // says nothing the name does not. if self.eat_kw("PROPERTY") { self.expect_kw("GRAPH")?; } else { self.eat_kw("GRAPH"); } - Ok(Some(GraphRef::Named(self.parse_graph_name()?))) + Ok(GraphRef::Named(self.parse_graph_name()?)) } fn parse_where(&mut self) -> Result> { @@ -2114,7 +2120,25 @@ mod tests { if_not_exists: false, or_replace: true, of: GraphTypeRef::Any, - copy_of: Some("h".into()), + copy_of: Some(GraphRef::Named(GraphName { + schema: None, + name: "h".into(), + })), + } + ); + // The graph the statement is against is a graph to copy like + // any other, and the one a copy of a loaded file means. + assert_eq!( + catalog_stmt("CREATE GRAPH g ANY AS COPY OF CURRENT_PROPERTY_GRAPH"), + CatalogStmt::CreateGraph { + name: GraphName { + schema: None, + name: "g".into(), + }, + if_not_exists: false, + or_replace: false, + of: GraphTypeRef::Any, + copy_of: Some(GraphRef::Current), } ); // GG03 and GG04: a type written where the graph is created. diff --git a/crates/zu-zu1/src/catalog.rs b/crates/zu-zu1/src/catalog.rs index 57393ceb..419fc0fd 100644 --- a/crates/zu-zu1/src/catalog.rs +++ b/crates/zu-zu1/src/catalog.rs @@ -998,6 +998,74 @@ impl Catalog { nodes.chain(rels).collect() } + /// Copies the table definitions of one graph into another, which is + /// the catalog half of `CREATE GRAPH ... AS COPY OF` (GC04). + /// + /// Each copy keeps the name it had, which is what makes the copy a + /// copy: a query written against the source runs against the copy + /// once `USE` names it. A name belongs to a graph rather than to + /// the file, so both graphs holding a `person` is two tables and + /// not a clash. What the copies do not keep is their ids, because + /// an id is what the table index addresses storage by and the copy + /// gets storage of its own. + /// + /// Answers the source id, the copy's id and the kind of each table, + /// which is what [`crate::graph::copy_graph_storage`] needs to copy + /// the blocks those ids stand for. + pub fn copy_graph_tables( + &mut self, + source: u32, + target: u32, + ) -> Result> { + let mut copied = Vec::new(); + // Node tables first, because a rel table names the two node + // tables it runs between and wants the copies' ids, not the + // source's, or its edges would land back in the source graph. + for table in self.nodes.clone() { + if table.graph != source { + continue; + } + let id = self.next_id()?; + self.nodes.push(NodeTable { + id, + graph: target, + ..table.clone() + }); + copied.push((table.id, id, ElementKind::Node)); + } + let nodes: Vec<(u32, u32)> = copied.iter().map(|&(from, to, _)| (from, to)).collect(); + let mapped = |id: u32| { + nodes + .iter() + .find(|&&(from, _)| from == id) + .map(|&(_, to)| to) + }; + for table in self.rels.clone() { + if table.graph != source { + continue; + } + // Both ends are in the graph being copied, which `validate` + // holds every catalog to, so a missing one is a file that + // should never have opened. + let (Some(from), Some(to)) = (mapped(table.from), mapped(table.to)) else { + return Err(corrupt( + "catalog", + format!("rel table '{}' ends outside graph {source}", table.name), + )); + }; + let id = self.next_id()?; + self.rels.push(RelTable { + id, + graph: target, + from, + to, + ..table.clone() + }); + copied.push((table.id, id, ElementKind::Edge)); + } + Ok(copied) + } + /// Drops a graph and every table in it, answering whether there was /// one. The storage those tables held is the caller's to free; this /// is the catalog half of `DROP GRAPH`. diff --git a/crates/zu-zu1/src/graph.rs b/crates/zu-zu1/src/graph.rs index 3d265a77..049adf7d 100644 --- a/crates/zu-zu1/src/graph.rs +++ b/crates/zu-zu1/src/graph.rs @@ -1046,6 +1046,115 @@ pub fn free_graph_storage(db: &mut Zu1File, catalog: &Catalog, graph: u32) -> Re stats.store(db) } +/// Copies everything the tables of one graph hold into blocks of their +/// own, which is the storage half of `CREATE GRAPH ... AS COPY OF` +/// (GC04). +/// +/// `tables` pairs each source table id with the id the catalog gave its +/// copy, which the caller has already put in the new graph. The copy is +/// block for block: every segment block of the source is read whole and +/// written into a freshly allocated one, and the directories are +/// re-encoded only because a directory names the blocks its segments +/// live in and those are now different blocks. The bytes of a column, +/// a CSR array and a key index are the source's byte for byte, so the +/// copy costs a read and a write per block and no decode, and it does +/// not matter to it what the columns hold. +/// +/// Nothing is shared with the source. A copy that pointed at the same +/// segments would be a second name for one graph, and the first write +/// to either would show up in both; `COPY OF` is a graph that starts +/// out equal and goes its own way. +/// +/// Nothing is published here either, as in [`free_graph_storage`]: the +/// table index and the statistics are staged and the checkpoint that +/// stores the catalog makes all three visible at once. +pub fn copy_graph_storage(db: &mut Zu1File, tables: &[(u32, u32, ElementKind)]) -> Result<()> { + let mut index = TableIndex::load(db)?; + let mut stats = crate::stats::Stats::load(db)?; + for &(source, copy, kind) in tables { + if let Some(root) = index.get(source) { + let copied = match kind { + ElementKind::Node => crate::props::copy_props(db, root)?, + ElementKind::Edge => copy_directory(db, root)?, + }; + index.set(copy, copied); + } + // A node table's deleted rows are part of what it holds: a copy + // that left them behind would answer a scan with rows the + // source no longer has. + if kind == ElementKind::Node + && let Some(root) = index.get(source | crate::fold::TOMBSTONE_KEY) + { + let copied = copy_chain(db, root)?; + index.set(copy | crate::fold::TOMBSTONE_KEY, copied); + } + // The copy holds the same rows, so it plans the same way. + // Gathering the statistics again would read every column of it + // for numbers the file already has. + if let Some(rels) = stats.rels.get(&source).cloned() { + stats.rels.insert(copy, rels); + } + if let Some(cols) = stats.cols.get(&source).cloned() { + stats.cols.insert(copy, cols); + } + } + free_chain(db, db.db_header().table_index_root)?; + free_chain(db, db.db_header().stats_root)?; + let index_root = meta::write_chain(db, &index.encode())?; + db.db_header_mut().table_index_root = index_root; + stats.store(db) +} + +/// Copies a group directory and everything it points at, answering the +/// root of the copy. The mirror of [`free_directory`], and it walks the +/// same pointers: a block either free walks past or copy walks past is +/// a block the other one leaks. +fn copy_directory(db: &mut Zu1File, root: BlockPtr) -> Result { + let mut directory = Directory::decode(&meta::read_chain(db, root)?)?; + if directory.props != NULL_BLOCK { + directory.props = crate::props::copy_props(db, directory.props)?; + } + if let Some(keys) = &mut directory.keys { + for seg in [&mut keys.keys, &mut keys.rows] { + seg.blocks = copy_blocks(db, &seg.blocks)?; + } + } + for group in &mut directory.groups { + for seg in [ + &mut group.fwd.offsets, + &mut group.fwd.neighbors, + &mut group.bwd.offsets, + &mut group.bwd.neighbors, + ] { + seg.blocks = copy_blocks(db, &seg.blocks)?; + } + } + meta::write_chain(db, &directory.encode()) +} + +/// Copies a meta chain, answering the root of the copy. The payload is +/// what carries over rather than the blocks, because a chain block +/// holds the pointer to the next one and those are the caller's to +/// hand out. +fn copy_chain(db: &mut Zu1File, root: BlockPtr) -> Result { + let payload = meta::read_chain(db, root)?; + meta::write_chain(db, &payload) +} + +/// Copies a segment's blocks into fresh ones, answering where they +/// landed. A block is read and written whole: what a segment block +/// holds is the encoder's business and none of this function's. +pub(crate) fn copy_blocks(db: &mut Zu1File, blocks: &[BlockPtr]) -> Result> { + let mut out = Vec::with_capacity(blocks.len()); + for &ptr in blocks { + let data = db.read_block(ptr)?; + let copy = db.allocate_block(); + db.write_block(copy, &data)?; + out.push(copy); + } + Ok(out) +} + /// Read access to a bulk-loaded graph, caching the most recently decoded /// group per direction so sequential scans decode each group once. The /// two directions cache independently because a plan often walks both @@ -2311,4 +2420,102 @@ mod tests { let err = bulk_load_keyed(&mut db3, "node", "edge", n, edges, Some(&[1, 2])).unwrap_err(); assert!(format!("{err}").contains("2 keys")); } + + /// The property and CSR blocks a table's storage names, in the + /// order the copy walks them, so two tables' lists line up entry + /// for entry when one is a copy of the other. The meta chain is not + /// in here: a chain block holds the pointers to the segments and to + /// the next chain block, and those are what a copy has of its own. + fn segment_blocks(db: &mut Zu1File, root: BlockPtr, kind: ElementKind) -> Vec { + let mut out = Vec::new(); + match kind { + ElementKind::Node => props_segment_blocks(db, root, &mut out), + ElementKind::Edge => { + let directory = Directory::decode(&meta::read_chain(db, root).unwrap()).unwrap(); + if directory.props != NULL_BLOCK { + props_segment_blocks(db, directory.props, &mut out); + } + if let Some(keys) = &directory.keys { + out.extend(&keys.keys.blocks); + out.extend(&keys.rows.blocks); + } + for group in &directory.groups { + for seg in [ + &group.fwd.offsets, + &group.fwd.neighbors, + &group.bwd.offsets, + &group.bwd.neighbors, + ] { + out.extend(&seg.blocks); + } + } + } + } + out + } + + fn props_segment_blocks(db: &mut Zu1File, root: BlockPtr, out: &mut Vec) { + let directory = + crate::props::PropsDirectory::decode(&meta::read_chain(db, root).unwrap()).unwrap(); + out.extend(directory.labels.iter().flat_map(|m| &m.blocks)); + for col in &directory.columns { + out.extend(&col.meta.blocks); + out.extend(col.validity.iter().flat_map(|m| &m.blocks)); + } + } + + #[test] + fn a_graph_copy_holds_the_same_bytes_in_blocks_of_its_own() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("copy.zu1"); + let mut db = Zu1File::create(&path).unwrap(); + let mut edges = vec![(0, 1), (0, 3), (1, 2), (3, 0)]; + let edges = sorted_edges(&mut edges).to_vec(); + bulk_load_keyed( + &mut db, + "person", + "follows", + 4, + &edges, + Some(&[10, 20, 30, 40]), + ) + .unwrap(); + crate::props::store_props( + &mut db, + "person", + &[("age", crate::props::PropValues::Int(&[31, 32, 33, 34]))], + ) + .unwrap(); + db.checkpoint().unwrap(); + + let mut catalog = Catalog::load(&mut db).unwrap(); + let source = catalog.home_graph_id(); + let target = catalog + .add_graph( + "twin", + crate::catalog::ROOT_SCHEMA, + crate::catalog::GraphTypeOf::Open, + ) + .unwrap(); + let tables = catalog.copy_graph_tables(source, target).unwrap(); + copy_graph_storage(&mut db, &tables).unwrap(); + catalog.store(&mut db).unwrap(); + // The catalog validates on store, so a copy that got its table + // names or its endpoints wrong never gets this far. + assert_eq!(tables.len(), 2); + + let index = TableIndex::load(&mut db).unwrap(); + for (from, to, kind) in tables { + let (from, to) = (index.get(from).unwrap(), index.get(to).unwrap()); + let source = segment_blocks(&mut db, from, kind); + let copy = segment_blocks(&mut db, to, kind); + assert_eq!(source.len(), copy.len(), "{kind:?}"); + assert!(!source.is_empty(), "{kind:?} stores something"); + for (&a, &b) in source.iter().zip(©) { + assert_ne!(a, b, "a copied block is a block of its own"); + assert_eq!(db.read_block(a).unwrap(), db.read_block(b).unwrap()); + } + } + crate::verify(&path).unwrap(); + } } diff --git a/crates/zu-zu1/src/props.rs b/crates/zu-zu1/src/props.rs index 141a5287..b90ca5c6 100644 --- a/crates/zu-zu1/src/props.rs +++ b/crates/zu-zu1/src/props.rs @@ -801,6 +801,23 @@ impl PropsDirectory { } } +/// Copies a props directory and every segment it points at, answering +/// the root of the copy. The mirror of [`free_props`], walking the same +/// pointers in the same order. +pub(crate) fn copy_props(db: &mut Zu1File, root: BlockPtr) -> Result { + let mut directory = PropsDirectory::decode(&meta::read_chain(db, root)?)?; + if let Some(labels) = &mut directory.labels { + labels.blocks = crate::graph::copy_blocks(db, &labels.blocks)?; + } + for col in &mut directory.columns { + col.meta.blocks = crate::graph::copy_blocks(db, &col.meta.blocks)?; + if let Some(validity) = &mut col.validity { + validity.blocks = crate::graph::copy_blocks(db, &validity.blocks)?; + } + } + meta::write_chain(db, &directory.encode()) +} + pub(crate) fn free_props(db: &mut Zu1File, root: BlockPtr) -> Result<()> { free_props_parts(db, root, true) } diff --git a/crates/zu/src/catalog_stmt.rs b/crates/zu/src/catalog_stmt.rs index 273e27a3..62f896a6 100644 --- a/crates/zu/src/catalog_stmt.rs +++ b/crates/zu/src/catalog_stmt.rs @@ -20,7 +20,8 @@ use zu_common::{Result, ZuError}; use zu_query::ast::{ - CatalogStmt, ElementDefKind, ElementTypeDef, Endpoint, GraphName, GraphTypeRef, GraphTypeSource, + CatalogStmt, ElementDefKind, ElementTypeDef, Endpoint, GraphName, GraphRef, GraphTypeRef, + GraphTypeSource, }; use crate::zu1::catalog::{Catalog, ElementKind, ElementType, GraphType, GraphTypeOf, ROOT_SCHEMA}; @@ -146,20 +147,31 @@ pub fn apply(db: &mut Zu1File, stmt: &CatalogStmt) -> Result { // that was there. let id = existing.id; let graph_type = graph_type_of(&mut catalog, &name, of)?; - copy(&catalog, copy_of.as_deref())?; + let source = copy_source(&catalog, copy_of.as_ref())?; + // A replacement frees what the old graph held before it + // writes the new one, so a graph asked to become a copy + // of itself is asked for a copy of what is about to be + // gone. + if source == Some(id) { + return Err(ZuError::InvalidArgument(format!( + "'{name}' cannot be replaced by a copy of itself" + ))); + } // Everything the add below could refuse is asked here, // because after the free there is no graph to leave // standing. catalog.check_graph(&schema, &graph_type)?; graph::free_graph_storage(db, &catalog, id)?; catalog.drop_graph(id); - catalog.add_graph(&name, &schema, graph_type)?; + let target = catalog.add_graph(&name, &schema, graph_type)?; + copy_into(db, &mut catalog, source, target)?; catalog.store(db)?; return Ok(Effect::Created); } let graph_type = graph_type_of(&mut catalog, &name, of)?; - copy(&catalog, copy_of.as_deref())?; - catalog.add_graph(&name, &schema, graph_type)?; + let source = copy_source(&catalog, copy_of.as_ref())?; + let target = catalog.add_graph(&name, &schema, graph_type)?; + copy_into(db, &mut catalog, source, target)?; } CatalogStmt::DropGraph { name, if_exists } => { let (schema, name) = split(name); @@ -232,24 +244,44 @@ fn graph_type_of(catalog: &mut Catalog, name: &str, of: &GraphTypeRef) -> Result } } -/// What `AS COPY OF` copies (GG05). +/// The graph `AS COPY OF` names (GG05), resolved before anything is +/// created so a statement naming a graph the file does not have leaves +/// the file alone. +fn copy_source(catalog: &Catalog, source: Option<&GraphRef>) -> Result> { + match source { + None => Ok(None), + // The graph the statement is against, which is the home graph: + // that is the one a query with no `USE` reads and the one a + // loaded file put its tables in. + Some(GraphRef::Current) => Ok(Some(catalog.home_graph_id())), + Some(GraphRef::Named(name)) => { + let (schema, name) = split(name); + let graph = catalog.graph(&schema, &name).ok_or_else(|| { + ZuError::InvalidArgument(format!("'{name}' is no graph in '{schema}'")) + })?; + Ok(Some(graph.id)) + } + } +} + +/// Fills a created graph with a copy of another one (GG05). /// -/// A graph that holds no tables copies as the empty graph it is. A -/// graph that holds some is a block for block copy of its tables, which -/// is not written yet, and saying so beats creating a graph that is -/// empty where the statement asked for a copy. -fn copy(catalog: &Catalog, source: Option<&str>) -> Result<()> { +/// The copy is by value and not by reference: the new graph gets tables +/// of its own holding the same names and blocks of its own holding the +/// same bytes, so a write to either graph is nothing to the other. A +/// graph that holds no tables copies as the empty graph it is, which +/// falls out of there being nothing to walk. +fn copy_into( + db: &mut Zu1File, + catalog: &mut Catalog, + source: Option, + target: u32, +) -> Result<()> { let Some(source) = source else { return Ok(()); }; - let graph = graph_named(catalog, source)?; - if !catalog.graph_tables(graph.id).is_empty() { - return Err(ZuError::Unsupported { - what: "AS COPY OF a graph that holds tables", - id: graph.id, - }); - } - Ok(()) + let tables = catalog.copy_graph_tables(source, target)?; + graph::copy_graph_storage(db, &tables) } /// The graph a statement names, which is a graph in the root schema diff --git a/crates/zu/src/session.rs b/crates/zu/src/session.rs index fc6bf905..350b9f69 100644 --- a/crates/zu/src/session.rs +++ b/crates/zu/src/session.rs @@ -726,16 +726,24 @@ mod tests { }; assert_eq!(ty.elements.len(), 1); - // GG05: an empty graph copies as the empty graph it is, and a - // graph with tables in it is the copy nobody has written yet. + // GG05: an empty graph copies as the empty graph it is. session .run("CREATE GRAPH copy_of_it ANY AS COPY OF open_one", &[]) .expect("copy of an empty graph"); - let err = session - .run("CREATE GRAPH copy_of_home ANY AS COPY OF home", &[]) - .expect_err("a copy of a graph that holds tables") - .to_string(); - assert!(err.contains("AS COPY OF"), "{err}"); + assert!( + session + .graph + .catalog() + .graph_tables( + session + .graph + .catalog() + .graph("/", "copy_of_it") + .expect("the copy") + .id + ) + .is_empty() + ); let err = session .run("CREATE GRAPH lost ANY AS COPY OF nowhere", &[]) @@ -757,6 +765,67 @@ mod tests { assert!(session.graph.catalog().graph("/", "home").is_some()); } + #[test] + fn a_copy_of_a_graph_holds_the_rows_of_the_one_it_copied() { + let dir = tempfile::tempdir().expect("tempdir"); + let path = dir.path().join("copy.zu1"); + let edges = seeded(&path); + + let mut session = Session::open(&path).expect("open"); + session + .run("CREATE GRAPH twin ANY AS COPY OF home", &[]) + .expect("copy of a graph that holds tables"); + + let pattern = "MATCH (a:person {id: $src})-[:follows]->(b) \ + RETURN b.id AS friend ORDER BY friend"; + let mut want: Vec = edges + .iter() + .filter(|(s, _)| *s == 3) + .map(|(_, d)| i64::from(*d)) + .collect(); + want.sort_unstable(); + let friends = |session: &mut Session, source: &str| -> Vec { + session + .run(source, &[("src", Value::Int(3))]) + .expect("query") + .rows + .iter() + .map(|r| match &r[0] { + Value::Int(i) => *i, + other => panic!("expected int, got {other:?}"), + }) + .collect() + }; + assert_eq!(friends(&mut session, pattern), want); + assert_eq!(friends(&mut session, &format!("USE twin {pattern}")), want); + + // The graph the statement is against is a graph to copy like + // any other, and the one a file loaded from outside has its + // tables in. + session + .run( + "CREATE GRAPH twin_of_here ANY AS COPY OF CURRENT_PROPERTY_GRAPH", + &[], + ) + .expect("copy of the graph the statement is against"); + assert_eq!( + friends(&mut session, &format!("USE twin_of_here {pattern}")), + want + ); + + // The copy holds blocks of its own, which dropping it is what + // proves: a copy that had merely pointed at the source's + // segments would have handed the source's blocks back here and + // the query below would read a graph that is no longer there. + session.run("DROP GRAPH twin", &[]).expect("drop the copy"); + assert_eq!(friends(&mut session, pattern), want); + let err = session + .run(&format!("USE twin {pattern}"), &[]) + .expect_err("the copy is gone") + .to_string(); + assert!(err.contains("which is no graph in"), "{err}"); + } + /// How many blocks the committed free list names, which is what a /// drop that reclaims has to grow. fn free_blocks(db: &mut Zu1File) -> u64 { diff --git a/docs/07-query-engine.md b/docs/07-query-engine.md index b169f9c3..8b638d82 100644 --- a/docs/07-query-engine.md +++ b/docs/07-query-engine.md @@ -115,7 +115,9 @@ Which graph a table belongs to is a field on the table rather than a list on the A graph is created with the open type, with a graph type the file already holds (`CREATE GRAPH g :: social`), or with one written where the graph is created (`CREATE GRAPH g { (:Person {name :: STRING}) }`, which is GG03, and `CREATE GRAPH g LIKE h`, which is GG04 read at the graph). An inline type is kept on the graph and not added to the file's graph types, since nobody wrote a name for it. -`AS COPY OF` (GG05) says what the new graph starts with rather than what it is, so it is read after the type. A copy of an empty graph is a graph with no tables, which is exact. A copy of a graph that holds tables is refused today rather than approximated, because a props directory holds pointers inside block payloads and a copy that walked them wrong would be a copy that read as data. +`AS COPY OF` (GG05) says what the new graph starts with rather than what it is, so it is read after the type. What it names is a graph the file holds or `CURRENT_PROPERTY_GRAPH`, which is the graph the statement is against, and a copy of an empty graph is a graph with no tables. + +A copy of a graph that holds tables is a copy by value: the catalog gets a table per table of the source, keeping the name and taking a new id, and the storage those ids address is copied block for block. Every segment block is read whole and written into a freshly allocated one, so a copy costs a read and a write per block and no decode, and what a column holds is none of the copy's business. The directories are re-encoded because a directory names the blocks its segments live in and those are now different blocks, and everything else, the labels, the validity, the key index, the CSR pair per group, the edge properties, and the tombstones a node table has, carries over byte for byte. The two graphs share nothing, which is the point: a copy that pointed at the source's segments would be a second name for one graph and the first write to either would show up in both. `DROP GRAPH` is the one statement in zu that hands blocks back. It frees the props directory of every node table in the graph, the group directory of every rel table, the tombstone chain of every node table that has one, and the table index and stats chains it rewrites, then takes the graph and its tables out of the catalog. Nothing is published along the way: the checkpoint that stores the catalog makes the catalog, the table index and the stats visible together, so a drop that fails halfway leaves the file exactly as it was. Freed blocks become allocatable at the checkpoint after the one that published the free, and `block_count` is a high-water mark that never shrinks, so what a drop returns is measured in the free list and not in the size of the file. Dropping the home graph is allowed and is the reclamation path a file with one graph has; the next load puts an empty home graph back. diff --git a/docs/api/model.json b/docs/api/model.json index 0bb8a1f8..5449b158 100644 --- a/docs/api/model.json +++ b/docs/api/model.json @@ -2420,6 +2420,14 @@ "signature": "fn check_graph(&self, schema: &str, graph_type: &GraphTypeOf) -> Result<()>", "doc": "Everything `add_graph` refuses a graph for other than a name its\nschema already holds.\n\n`CREATE OR REPLACE GRAPH` frees the blocks the old graph held\nbefore the new one is added, and a refusal after that point would\nleave a file holding neither, so the caller asks first and frees\nnothing when the answer is no." }, + { + "id": "zu::zu1::catalog::Catalog::copy_graph_tables", + "kind": "method", + "name": "copy_graph_tables", + "of": "zu::zu1::catalog::Catalog", + "signature": "fn copy_graph_tables(&mut self, source: u32, target: u32) -> Result>", + "doc": "Copies the table definitions of one graph into another, which is\nthe catalog half of `CREATE GRAPH ... AS COPY OF` (GC04).\n\nEach copy keeps the name it had, which is what makes the copy a\ncopy: a query written against the source runs against the copy\nonce `USE` names it. A name belongs to a graph rather than to\nthe file, so both graphs holding a `person` is two tables and\nnot a clash. What the copies do not keep is their ids, because\nan id is what the table index addresses storage by and the copy\ngets storage of its own.\n\nAnswers the source id, the copy's id and the kind of each table,\nwhich is what [`crate::graph::copy_graph_storage`] needs to copy\nthe blocks those ids stand for." + }, { "id": "zu::zu1::catalog::Catalog::declare_label", "kind": "method", @@ -4186,6 +4194,13 @@ "signature": "fn bulk_load_undirected_as(db: &mut zu_zu1::file::Zu1File, node_table: &str, rel_table: &str, node_count: u64, edges: &[(u32, u32)]) -> zu_common::Result", "doc": "[`bulk_load_as`] for edges with no direction (GH02).\n\nThe edge list is stored the way it is written and nothing is\nmirrored: the reverse CSR every load builds is what answers the\nother way round, and the rel table says the two ways are one edge.\nA pattern that asks for a direction is what tells them apart, so an\nundirected table costs a directed one nothing on disk." }, + { + "id": "zu::zu1::graph::copy_graph_storage", + "kind": "function", + "name": "copy_graph_storage", + "signature": "fn copy_graph_storage(db: &mut zu_zu1::file::Zu1File, tables: &[(u32, u32, zu_zu1::catalog::ElementKind)]) -> zu_common::Result<()>", + "doc": "Copies everything the tables of one graph hold into blocks of their\nown, which is the storage half of `CREATE GRAPH ... AS COPY OF`\n(GC04).\n\n`tables` pairs each source table id with the id the catalog gave its\ncopy, which the caller has already put in the new graph. The copy is\nblock for block: every segment block of the source is read whole and\nwritten into a freshly allocated one, and the directories are\nre-encoded only because a directory names the blocks its segments\nlive in and those are now different blocks. The bytes of a column,\na CSR array and a key index are the source's byte for byte, so the\ncopy costs a read and a write per block and no decode, and it does\nnot matter to it what the columns hold.\n\nNothing is shared with the source. A copy that pointed at the same\nsegments would be a second name for one graph, and the first write\nto either would show up in both; `COPY OF` is a graph that starts\nout equal and goes its own way.\n\nNothing is published here either, as in [`free_graph_storage`]: the\ntable index and the statistics are staged and the checkpoint that\nstores the catalog makes all three visible at once." + }, { "id": "zu::zu1::graph::densify_keyed", "kind": "function", diff --git a/docs/conformance/zu.json b/docs/conformance/zu.json index a2be4f11..54c0b3a3 100644 --- a/docs/conformance/zu.json +++ b/docs/conformance/zu.json @@ -5,8 +5,8 @@ "host": "darwin arm64, Apple M4", "taken": "2026-08-16", "selector": "the whole corpus apart from its large fixtures", - "cases": 385, - "pass": 224, + "cases": 386, + "pass": 225, "fail": 123, "skip": 25, "error": 13, @@ -15,7 +15,7 @@ "conditions_seen": 22, "by_kind": { "mandatory": {"cases": 70, "pass": 54, "fail": 15, "skip": 1, "error": 0}, - "optional": {"cases": 208, "pass": 111, "fail": 82, "skip": 3, "error": 12}, + "optional": {"cases": 209, "pass": 112, "fail": 82, "skip": 3, "error": 12}, "condition": {"cases": 67, "pass": 28, "fail": 17, "skip": 21, "error": 1}, "grammar": {"cases": 17, "pass": 14, "fail": 3, "skip": 0, "error": 0}, "performance": {"cases": 23, "pass": 17, "fail": 6, "skip": 0, "error": 0} diff --git a/docs/gql-conformance.md b/docs/gql-conformance.md index be4d80df..af472081 100644 --- a/docs/gql-conformance.md +++ b/docs/gql-conformance.md @@ -13,9 +13,9 @@ Every column is a run of the [gql-compat](https://github.com/tamnd/gql-compat) c | on | darwin arm64, Apple M4 | darwin arm64, Apple M4 | darwin arm64, Apple M4 | | harness | gql-compat devel | gql-compat devel | gql-compat devel | | corpus | the whole corpus apart from its large fixtures | the whole corpus apart from its large fixtures | the whole corpus apart from its large fixtures | -| cases | 385 | 377 | 385 | -| judged (pass + fail) | 257 | 324 | 347 | -| **passed** | **72** (28.0%) | **186** (57.4%) | **224** (64.6%) | +| cases | 385 | 377 | 386 | +| judged (pass + fail) | 257 | 324 | 348 | +| **passed** | **72** (28.0%) | **186** (57.4%) | **225** (64.7%) | | failed | 185 | 138 | 123 | | skipped, cannot hold the fixture | 76 | 39 | 25 | | never reached a verdict | 52 | 14 | 13 | @@ -33,7 +33,7 @@ Pass over judged, then the two exclusions. Read the exclusions first. An engine | mandatory | zu | 70 | 54 | 15 | 1 | 0 | 78.3% | | optional | ladybug | 208 | 22 | 157 | 8 | 21 | 12.3% | | optional | neo4j | 201 | 76 | 107 | 5 | 13 | 41.5% | -| optional | zu | 208 | 111 | 82 | 3 | 12 | 57.5% | +| optional | zu | 209 | 112 | 82 | 3 | 12 | 57.7% | | condition | ladybug | 67 | 0 | 4 | 63 | 0 | 0.0% | | condition | neo4j | 66 | 6 | 28 | 31 | 1 | 17.6% | | condition | zu | 67 | 28 | 17 | 21 | 1 | 62.2% | diff --git a/docs/grammar.ebnf b/docs/grammar.ebnf index 6424f1fa..da259c6c 100644 --- a/docs/grammar.ebnf +++ b/docs/grammar.ebnf @@ -16,8 +16,9 @@ query = [ use_clause ] , { reading_clause } , return_clause , graph an expression computes is not a graph this file holds. A query with no USE is against the graph the session is working in, which is the home graph. *) -use_clause = "USE" , ( "CURRENT_PROPERTY_GRAPH" | "CURRENT_GRAPH" - | [ [ "PROPERTY" ] , "GRAPH" ] , graph_name ) ; +use_clause = "USE" , graph_ref ; +graph_ref = "CURRENT_PROPERTY_GRAPH" | "CURRENT_GRAPH" + | [ [ "PROPERTY" ] , "GRAPH" ] , graph_name ; (* Catalog statements (docs/07 ยง9). They change what the file declares and answer no rows, so they are told apart from a query by their @@ -31,7 +32,7 @@ drop_schema = "DROP" , "SCHEMA" , [ "IF" , "EXISTS" ] , schema_path , [ ";" ] ; create_graph = "CREATE" , [ "OR" , "REPLACE" ] , [ "PROPERTY" ] , "GRAPH" , [ "IF" , "NOT" , "EXISTS" ] , graph_name , graph_type_ref , - [ "AS" , "COPY" , "OF" , name ] , [ ";" ] ; + [ "AS" , "COPY" , "OF" , graph_ref ] , [ ";" ] ; drop_graph = "DROP" , [ "PROPERTY" ] , "GRAPH" , [ "IF" , "EXISTS" ] , graph_name , [ ";" ] ; create_graph_type = "CREATE" , [ "OR" , "REPLACE" ] , [ "PROPERTY" ] , "GRAPH" ,