From cfc64cdaef8ad51ab7b704ce5953b6c199a12e4d Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 10:32:41 -0400 Subject: [PATCH 01/92] docs: correct the wave-2 memo's refs-only OPST claim OPST holds every object type, segmented by o_type. IndexType::for_query gates it on o_is_ref only as a conservative default, and the binary scan path already takes OPST for any constant object. --- docs/audit/burn-down/sparql12-wave2-triple-terms.md | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/docs/audit/burn-down/sparql12-wave2-triple-terms.md b/docs/audit/burn-down/sparql12-wave2-triple-terms.md index 22aaab5f88..9b15ccaa4b 100644 --- a/docs/audit/burn-down/sparql12-wave2-triple-terms.md +++ b/docs/audit/burn-down/sparql12-wave2-triple-terms.md @@ -349,9 +349,13 @@ schedule the eval epic. Verified against the core: triple term a single `Sid`/`u64` arena handle inside `FlakeValue`/`Binding` so the row width and cache footprint of the scan/join path don't grow. Guard with `query_hot_bsbm` / `query_hot_bsbm_bi`. -- **`Opst` (object-leading index) is documented refs-only** - (`comparator.rs:30,79-83`). Looking up a triple-term object by value needs - index-selection work too. +- **`Opst` (object-leading index) holds every object type**, segmented by + `o_type` (`fluree-db-core/src/comparator.rs`, module doc). `IndexType::for_query` + gates OPST on `o_is_ref` as a conservative default only; `BinaryScanOperator` + already takes OPST for any constant object. A triple-term object therefore + needs its own `o_type` partition and index-selection work, not a refs-only + exception. (An earlier revision of this memo called OPST refs-only; that was + stale.) **Net:** comparator functions are largely safe (raw-byte compare); the cost and risk are in the **enum/encoder/decoder/hash/Display arms + a new arena + enum From 4188dd7ea247833460ce6273212362eddd063c81 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 11:29:35 -0400 Subject: [PATCH 02/92] feat(core): add the RDF 1.2 triple-term value kind A triple term `<<( s p o )>>` becomes a first-class object value. `FlakeValue::TripleTerm` carries the materialized form; the index stores a dictionary handle under `OType::TRIPLE_TERM` (tag-10, decode kind `TripleTermDict`, V1 kind byte 0x15). Handles are partitioned by the inner predicate (`triple_term::term_handle`) so a predicate restriction is one `o_key` interval, and `TermKey` is the encoded base edge, ordered subject-first for the reverse tree. Adds the `rdf:reifies` and `f:tripleTerm` sids, plus a read-side `is_scan_hidden_predicate` so wildcard scans keep hiding the link form alongside the `f:reifies*` bundle while it is index-internal. Every closed match over `FlakeValue` gains an arm: formatters render the term by its display form for now, and the commit codec rejects the value until the write path moves over. --- .../src/cross_ledger/shapes_materializer.rs | 1 + fluree-db-api/src/export.rs | 16 +++ fluree-db-api/src/format/construct.rs | 4 + fluree-db-api/src/format/delimited.rs | 1 + fluree-db-api/src/format/hydration.rs | 6 +- fluree-db-api/src/format/jsonld.rs | 2 + fluree-db-api/src/format/sparql.rs | 5 + fluree-db-api/src/format/typed.rs | 4 + .../src/read/binary_index_store.rs | 4 + fluree-db-core/src/commit/codec/op_codec.rs | 7 + fluree-db-core/src/flake.rs | 2 + fluree-db-core/src/lib.rs | 14 +- fluree-db-core/src/namespaces.rs | 31 +++++ fluree-db-core/src/o_type.rs | 16 ++- fluree-db-core/src/o_type_registry.rs | 1 + fluree-db-core/src/serde/flakes_transport.rs | 35 ++++- fluree-db-core/src/triple_term.rs | 129 ++++++++++++++++++ fluree-db-core/src/value.rs | 65 ++++++++- fluree-db-core/src/value_id.rs | 5 + fluree-db-query/src/binary_scan.rs | 4 +- fluree-db-query/src/dict_overlay.rs | 3 + fluree-db-query/src/eval/helpers.rs | 1 + fluree-db-query/src/eval/metadata.rs | 2 +- fluree-db-query/src/eval/value.rs | 6 +- fluree-db-query/src/fast_whole_graph_agg.rs | 4 +- fluree-db-query/src/materializer.rs | 2 + fluree-db-query/src/property_path.rs | 4 +- fluree-db-reasoner/src/execute/derived.rs | 9 ++ fluree-db-shacl/src/constraints/datatype.rs | 3 +- fluree-db-shacl/src/constraints/pattern.rs | 5 +- fluree-db-transact/src/generate/flakes.rs | 1 + fluree-db-transact/src/import_sink.rs | 3 + fluree-vocab/src/lib.rs | 10 ++ 33 files changed, 384 insertions(+), 21 deletions(-) create mode 100644 fluree-db-core/src/triple_term.rs diff --git a/fluree-db-api/src/cross_ledger/shapes_materializer.rs b/fluree-db-api/src/cross_ledger/shapes_materializer.rs index 29d684517c..9cda40b15e 100644 --- a/fluree-db-api/src/cross_ledger/shapes_materializer.rs +++ b/fluree-db-api/src/cross_ledger/shapes_materializer.rs @@ -327,5 +327,6 @@ fn flake_value_to_lexical( ), }); } + FlakeValue::TripleTerm(_) => o.to_string(), }) } diff --git a/fluree-db-api/src/export.rs b/fluree-db-api/src/export.rs index df6224b591..ce569ec730 100644 --- a/fluree-db-api/src/export.rs +++ b/fluree-db-api/src/export.rs @@ -1418,6 +1418,12 @@ fn flake_to_jsonld( } FlakeValue::Null => serde_json::Value::Null, + FlakeValue::TripleTerm(_) => { + let dt = resolve_datatype_iri(store, o_type) + .unwrap_or_else(|| format!("{}tripleTerm", fluree_vocab::fluree::DB)); + let compact_dt = compact_iri(&dt, prefixes); + serde_json::json!({ "@value": value.to_string(), "@type": compact_dt }) + } } } @@ -1842,6 +1848,16 @@ fn write_object( } FlakeValue::Null => Ok(()), // should have been filtered above + FlakeValue::TripleTerm(_) => { + let dt = resolve_datatype_iri(store, o_type) + .unwrap_or_else(|| format!("{}tripleTerm", fluree_vocab::fluree::DB)); + let text = value.to_string(); + w.write_all(b"\"")?; + syntax::write_string(w, &text)?; + w.write_all(b"\"^^<")?; + syntax::write_iri(w, &dt)?; + w.write_all(b">") + } } } diff --git a/fluree-db-api/src/format/construct.rs b/fluree-db-api/src/format/construct.rs index d5cafb4c77..52ec5b5f6a 100644 --- a/fluree-db-api/src/format/construct.rs +++ b/fluree-db-api/src/format/construct.rs @@ -411,6 +411,7 @@ fn literal_value(val: &FlakeValue) -> Result> { FlakeValue::DayTimeDuration(v) => LiteralValue::String(Arc::from(v.to_string())), FlakeValue::Duration(v) => LiteralValue::String(Arc::from(v.to_string())), FlakeValue::GeoPoint(v) => LiteralValue::String(Arc::from(v.to_string())), + FlakeValue::TripleTerm(_) => LiteralValue::String(Arc::from(val.to_string())), })) } @@ -439,6 +440,9 @@ fn flake_value_to_ir_term(val: &FlakeValue) -> Result> { FlakeValue::Vector(_) | FlakeValue::Null | FlakeValue::Ref(_) => { return literal_value(val).map(|_| None) } + FlakeValue::TripleTerm(_) => { + Datatype::from_iri(format!("{}tripleTerm", fluree_vocab::fluree::DB)) + } }; Ok(literal_value(val)?.map(|value| IrTerm::Literal { value, diff --git a/fluree-db-api/src/format/delimited.rs b/fluree-db-api/src/format/delimited.rs index ce1f09b315..bda9e5400a 100644 --- a/fluree-db-api/src/format/delimited.rs +++ b/fluree-db-api/src/format/delimited.rs @@ -636,6 +636,7 @@ fn write_flake_value(cell: &mut Vec, val: &FlakeValue, compactor: &IriCompac } FlakeValue::Json(json_str) => cell.extend_from_slice(json_str.as_bytes()), FlakeValue::GeoPoint(v) => cell.extend_from_slice(v.to_string().as_bytes()), + FlakeValue::TripleTerm(_) => cell.extend_from_slice(val.to_string().as_bytes()), } } diff --git a/fluree-db-api/src/format/hydration.rs b/fluree-db-api/src/format/hydration.rs index 40727331b6..947ae8397a 100644 --- a/fluree-db-api/src/format/hydration.rs +++ b/fluree-db-api/src/format/hydration.rs @@ -1372,7 +1372,7 @@ impl<'a> HydrationFormatter<'a> { // (which is what the `Pattern::EdgeAnnotation` / // `AnnotationTarget` IR expansion does), but those // patterns don't go through hydration. - if fluree_db_core::is_reserved_reifies_predicate(&pred) { + if fluree_db_core::is_scan_hidden_predicate(&pred) { continue; } @@ -2267,6 +2267,7 @@ impl<'a> HydrationFormatter<'a> { FlakeValue::DayTimeDuration(v) => Ok(JsonValue::String(v.to_string())), FlakeValue::Duration(v) => Ok(JsonValue::String(v.to_string())), FlakeValue::GeoPoint(v) => Ok(JsonValue::String(v.to_string())), + FlakeValue::TripleTerm(_) => Ok(JsonValue::String(flake.o.to_string())), }; } @@ -2311,6 +2312,7 @@ impl<'a> HydrationFormatter<'a> { FlakeValue::DayTimeDuration(v) => JsonValue::String(v.to_string()), FlakeValue::Duration(v) => JsonValue::String(v.to_string()), FlakeValue::GeoPoint(v) => JsonValue::String(v.to_string()), + FlakeValue::TripleTerm(_) => JsonValue::String(flake.o.to_string()), }; Ok(json!({ @@ -2408,6 +2410,7 @@ impl<'a> HydrationFormatter<'a> { FlakeValue::DayTimeDuration(v) => json!(v.to_string()), FlakeValue::Duration(v) => json!(v.to_string()), FlakeValue::GeoPoint(v) => json!(v.to_string()), + FlakeValue::TripleTerm(_) => json!(flake.o.to_string()), }; Ok(json!({ @@ -2535,6 +2538,7 @@ impl<'a> HydrationFormatter<'a> { FlakeValue::DayTimeDuration(v) => Ok(JsonValue::String(v.to_string())), FlakeValue::Duration(v) => Ok(JsonValue::String(v.to_string())), FlakeValue::GeoPoint(v) => Ok(JsonValue::String(v.to_string())), + FlakeValue::TripleTerm(_) => Ok(JsonValue::String(flake.o.to_string())), } } diff --git a/fluree-db-api/src/format/jsonld.rs b/fluree-db-api/src/format/jsonld.rs index 0068418f4c..7f3fcdee57 100644 --- a/fluree-db-api/src/format/jsonld.rs +++ b/fluree-db-api/src/format/jsonld.rs @@ -496,6 +496,7 @@ pub(crate) fn format_binding(binding: &Binding, compactor: &IriCompactor) -> Res FlakeValue::DayTimeDuration(v) => Ok(JsonValue::String(v.to_string())), FlakeValue::Duration(v) => Ok(JsonValue::String(v.to_string())), FlakeValue::GeoPoint(v) => Ok(JsonValue::String(v.to_string())), + FlakeValue::TripleTerm(_) => Ok(JsonValue::String(val.to_string())), }; } @@ -545,6 +546,7 @@ pub(crate) fn format_binding(binding: &Binding, compactor: &IriCompactor) -> Res FlakeValue::DayTimeDuration(v) => JsonValue::String(v.to_string()), FlakeValue::Duration(v) => JsonValue::String(v.to_string()), FlakeValue::GeoPoint(v) => JsonValue::String(v.to_string()), + FlakeValue::TripleTerm(_) => JsonValue::String(val.to_string()), }; Ok(json!({ diff --git a/fluree-db-api/src/format/sparql.rs b/fluree-db-api/src/format/sparql.rs index a0df1c36ce..4635d1f431 100644 --- a/fluree-db-api/src/format/sparql.rs +++ b/fluree-db-api/src/format/sparql.rs @@ -674,6 +674,11 @@ fn format_binding( "value": v.to_string(), "datatype": dt_iri }))), + FlakeValue::TripleTerm(_) => Ok(Some(json!({ + "type": "literal", + "value": val.to_string(), + "datatype": dt_iri + }))), } } diff --git a/fluree-db-api/src/format/typed.rs b/fluree-db-api/src/format/typed.rs index 12ae5dc973..3841be5863 100644 --- a/fluree-db-api/src/format/typed.rs +++ b/fluree-db-api/src/format/typed.rs @@ -516,6 +516,10 @@ pub(crate) fn format_binding( "@value": v.to_string(), "@type": dt_iri })), + FlakeValue::TripleTerm(_) => Ok(json!({ + "@value": val.to_string(), + "@type": dt_iri + })), } } diff --git a/fluree-db-binary-index/src/read/binary_index_store.rs b/fluree-db-binary-index/src/read/binary_index_store.rs index 9ac30ce498..2b9254a6ed 100644 --- a/fluree-db-binary-index/src/read/binary_index_store.rs +++ b/fluree-db-binary-index/src/read/binary_index_store.rs @@ -1517,6 +1517,9 @@ impl BinaryIndexStore { DecodeKind::SpatialArena => Err(io::Error::other( "spatial arena decode not yet implemented in V6", )), + DecodeKind::TripleTermDict => Err(io::Error::other( + "triple-term decode needs the term dictionary (not yet wired)", + )), } } @@ -2022,6 +2025,7 @@ impl BinaryIndexStore { OType::VECTOR => Some(Sid::new(namespaces::FLUREE_DB, "embeddingVector")), OType::FULLTEXT => Some(Sid::new(namespaces::FLUREE_DB, "fullText")), OType::GEO_POINT => Some(Sid::new(namespaces::OGC_GEO, geo_names::WKT_LITERAL)), + OType::TRIPLE_TERM => Some(fluree_db_core::triple_term_datatype_sid().clone()), // Types without a stable datatype (or not representable as typed literals) // return None so callers can either skip constraints or use a safe fallback. diff --git a/fluree-db-core/src/commit/codec/op_codec.rs b/fluree-db-core/src/commit/codec/op_codec.rs index 313aae6fed..56955deaa5 100644 --- a/fluree-db-core/src/commit/codec/op_codec.rs +++ b/fluree-db-core/src/commit/codec/op_codec.rs @@ -150,6 +150,13 @@ fn encode_object( let name_id = dicts.object_ref.insert(sid.name.as_ref()); encode_varint(name_id as u64, buf); } + // Commits still carry reifications as `f:reifies*` bundles; the term + // form is interned index-side only until the write path moves over. + FlakeValue::TripleTerm(_) => { + return Err(CommitCodecError::UnsupportedValue( + "triple term objects are not encodable in commits yet".into(), + )); + } FlakeValue::Long(n) => { buf.push(OTag::Long as u8); encode_varint(zigzag_encode(*n), buf); diff --git a/fluree-db-core/src/flake.rs b/fluree-db-core/src/flake.rs index f518c8cff1..c0e535bf09 100644 --- a/fluree-db-core/src/flake.rs +++ b/fluree-db-core/src/flake.rs @@ -505,6 +505,7 @@ impl Flake { FlakeValue::Ref(sid) => 8 + sid.name.len(), FlakeValue::Vector(v) => 8 + v.len() * 8, // length prefix + 8 bytes per f64 FlakeValue::GeoPoint(_) => 8, // packed u64 + FlakeValue::TripleTerm(t) => 48 + t.s.name.len() + t.p.name.len(), }; // Metadata size @@ -550,6 +551,7 @@ impl Flake { FlakeValue::Ref(sid) => 8 + sid.name.len() as u64, FlakeValue::Vector(v) => (v.len() * 8) as u64, FlakeValue::GeoPoint(_) => 8, + FlakeValue::TripleTerm(t) => 48 + (t.s.name.len() + t.p.name.len()) as u64, }; let m_size: u64 = match &self.m { diff --git a/fluree-db-core/src/lib.rs b/fluree-db-core/src/lib.rs index c58ff4ddb2..78d6aa4f31 100644 --- a/fluree-db-core/src/lib.rs +++ b/fluree-db-core/src/lib.rs @@ -83,6 +83,7 @@ pub mod sysmem; pub mod task; pub mod temporal; pub mod tracking; +pub mod triple_term; pub mod value; pub mod value_id; pub mod vec_bi_dict; @@ -149,11 +150,12 @@ pub use namespaces::{ is_owl_equivalent_class, is_owl_equivalent_property, is_owl_functional_property, is_owl_imports, is_owl_inverse_functional_property, is_owl_inverse_of, is_owl_object_property_class, is_owl_ontology_class, is_owl_same_as, is_owl_symmetric_property, - is_owl_transitive_property, is_rdf_first, is_rdf_nil, is_rdf_property_class, is_rdf_rest, - is_rdf_type, is_rdfs_domain, is_rdfs_range, is_rdfs_subclass_of, is_rdfs_subproperty_of, - is_reifies_datatype, is_reifies_graph, is_reifies_lang, is_reifies_list_index, - is_reifies_object, is_reifies_predicate, is_reifies_subject, is_reserved_reifies_predicate, - is_schema_class, is_schema_predicate, reifies_predicate_sids, + is_owl_transitive_property, is_rdf_first, is_rdf_nil, is_rdf_property_class, is_rdf_reifies, + is_rdf_rest, is_rdf_type, is_rdfs_domain, is_rdfs_range, is_rdfs_subclass_of, + is_rdfs_subproperty_of, is_reifies_datatype, is_reifies_graph, is_reifies_lang, + is_reifies_list_index, is_reifies_object, is_reifies_predicate, is_reifies_subject, + is_reserved_reifies_predicate, is_scan_hidden_predicate, is_schema_class, is_schema_predicate, + rdf_reifies_sid, reifies_predicate_sids, triple_term_datatype_sid, }; pub use nonempty::NonEmpty; pub use ns_encoding::{ @@ -221,7 +223,7 @@ pub use tracking::{ }; pub use value::{ parse_decimal, parse_decimal_string, parse_double, parse_integer, parse_integer_string, - FlakeValue, GeoPointBits, + FlakeValue, GeoPointBits, TripleTermValue, }; pub use value_id::{ObjKey, ObjKeyError, ObjKind, ObjPair, ValueTypeTag}; pub use verified_identity::VerifiedIdentity; diff --git a/fluree-db-core/src/namespaces.rs b/fluree-db-core/src/namespaces.rs index b9866b63fa..65a046d79c 100644 --- a/fluree-db-core/src/namespaces.rs +++ b/fluree-db-core/src/namespaces.rs @@ -378,6 +378,37 @@ pub fn reifies_predicate_sids() -> [Sid; 7] { cached_reifies_predicate_sids().clone() } +/// The cached `rdf:reifies` predicate SID: the RDF 1.2 link from a +/// reifier to the triple term it reifies (`_:r rdf:reifies <<( s p o )>>`). +#[inline] +pub fn rdf_reifies_sid() -> &'static Sid { + static SID: OnceLock = OnceLock::new(); + SID.get_or_init(|| Sid::new(RDF, fluree_vocab::rdf_names::REIFIES)) +} + +/// True for `rdf:reifies`. +#[inline] +pub fn is_rdf_reifies(sid: &Sid) -> bool { + sid.namespace_code == RDF && sid.name.as_ref() == fluree_vocab::rdf_names::REIFIES +} + +/// True for the predicates wildcard scans hide from users: the seven +/// `f:reifies*` bundle predicates and, while the RDF 1.2 link form is +/// index-internal, `rdf:reifies`. Read-side only; the write firewall is +/// [`is_reserved_reifies_predicate`]. +#[inline] +pub fn is_scan_hidden_predicate(sid: &Sid) -> bool { + is_reserved_reifies_predicate(sid) || is_rdf_reifies(sid) +} + +/// The cached `f:tripleTerm` datatype SID carried by a triple-term object, +/// as `@id` is carried by a reference. +#[inline] +pub fn triple_term_datatype_sid() -> &'static Sid { + static SID: OnceLock = OnceLock::new(); + SID.get_or_init(|| Sid::new(FLUREE_DB, fluree_db_predicates::TRIPLE_TERM)) +} + /// Baseline namespace codes (code -> prefix) matching Fluree's reserved codepoints. pub fn default_namespace_codes() -> HashMap { let mut map = HashMap::new(); diff --git a/fluree-db-core/src/o_type.rs b/fluree-db-core/src/o_type.rs index 14a0890685..37a7003ace 100644 --- a/fluree-db-core/src/o_type.rs +++ b/fluree-db-core/src/o_type.rs @@ -158,8 +158,11 @@ impl OType { pub const NUM_BIG_OVERFLOW: Self = Self(0x800B); /// Spatial (complex geometry) — `o_key` is a spatial arena handle. pub const SPATIAL_COMPLEX: Self = Self(0x800C); + /// RDF 1.2 triple term — `o_key` is a triple-term dictionary handle + /// (`(inner p_id << 32) | seq`, see `fluree_db_core::triple_term`). + pub const TRIPLE_TERM: Self = Self(0x800D); - // Tag `10` payload range 0x800D–0xBFFF reserved for future Fluree domains. + // Tag `10` payload range 0x800E–0xBFFF reserved for future Fluree domains. // ── Tag `11` — rdf:langString ────────────────────────────────────── @@ -430,6 +433,7 @@ impl OType { 0x800A => DecodeKind::StringDict, // fulltext (string dict + BM25) 0x800B => DecodeKind::NumBigArena, 0x800C => DecodeKind::SpatialArena, + 0x800D => DecodeKind::TripleTermDict, _ => DecodeKind::Sentinel, // future Fluree domains } } @@ -489,6 +493,8 @@ pub enum DecodeKind { NumBigArena, /// Spatial arena handle (per-predicate). SpatialArena, + /// Triple-term dictionary handle (ledger-global, partitioned by inner predicate). + TripleTermDict, } impl DecodeKind { @@ -519,6 +525,7 @@ impl DecodeKind { 21 => Some(Self::VectorArena), 22 => Some(Self::NumBigArena), 23 => Some(Self::SpatialArena), + 24 => Some(Self::TripleTermDict), _ => None, } } @@ -574,6 +581,7 @@ impl fmt::Debug for OType { 0x800A => write!(f, "OType::FULLTEXT"), 0x800B => write!(f, "OType::NUM_BIG_OVERFLOW"), 0x800C => write!(f, "OType::SPATIAL_COMPLEX"), + 0x800D => write!(f, "OType::TRIPLE_TERM"), v if self.is_lang_string() => write!(f, "OType::LANG_STRING({})", v & 0x3FFF), v if self.is_customer_datatype() => { write!(f, "OType::CUSTOMER({})", v & 0x3FFF) @@ -735,6 +743,11 @@ mod tests { OType::NUM_BIG_OVERFLOW.decode_kind(), DecodeKind::NumBigArena ); + assert_eq!(OType::TRIPLE_TERM.decode_kind(), DecodeKind::TripleTermDict); + assert_eq!( + DecodeKind::from_u8(DecodeKind::TripleTermDict as u8), + Some(DecodeKind::TripleTermDict) + ); assert_eq!(OType::lang_string(5).decode_kind(), DecodeKind::StringDict); assert_eq!( OType::customer_datatype(10).decode_kind(), @@ -768,6 +781,7 @@ mod tests { OType::IRI_REF, OType::VECTOR, OType::NUM_BIG_OVERFLOW, + OType::TRIPLE_TERM, ] { assert!(!ot.is_string_keyed(), "{ot:?}"); } diff --git a/fluree-db-core/src/o_type_registry.rs b/fluree-db-core/src/o_type_registry.rs index 5533318cdf..15e261dfc1 100644 --- a/fluree-db-core/src/o_type_registry.rs +++ b/fluree-db-core/src/o_type_registry.rs @@ -92,6 +92,7 @@ impl OTypeRegistry { ObjKind::YEAR_MONTH_DUR => OType::XSD_YEAR_MONTH_DURATION, ObjKind::DAY_TIME_DUR => OType::XSD_DAY_TIME_DURATION, ObjKind::GEO_POINT => OType::GEO_POINT, + ObjKind::TRIPLE_TERM => OType::TRIPLE_TERM, // Blank nodes are currently represented as REF_ID SIDs whose namespace code is // `namespaces::BLANK_NODE`. We intentionally map all REF_ID to `OType::IRI_REF` diff --git a/fluree-db-core/src/serde/flakes_transport.rs b/fluree-db-core/src/serde/flakes_transport.rs index 1a56a84987..0f9762bc44 100644 --- a/fluree-db-core/src/serde/flakes_transport.rs +++ b/fluree-db-core/src/serde/flakes_transport.rs @@ -20,7 +20,7 @@ use crate::flake::{Flake, FlakeMeta}; use crate::sid::{Sid, SidInterner}; use crate::temporal::{Date, DateTime, Time}; -use crate::value::FlakeValue; +use crate::value::{FlakeValue, TripleTermValue}; use bigdecimal::BigDecimal; use num_bigint::BigInt; use serde::{Deserialize, Serialize}; @@ -125,11 +125,26 @@ pub enum TransportValue { /// JSON value as string #[serde(rename = "json")] Json(String), + /// RDF 1.2 triple term + #[serde(rename = "triple")] + TripleTerm(Box), /// Null value #[serde(rename = "null")] Null, } +/// Transport form of [`TripleTermValue`]: SIDs and the object value in their +/// transport encodings. +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)] +pub struct TransportTripleTerm { + pub s: TransportSid, + pub p: TransportSid, + pub o: TransportValue, + pub dt: TransportSid, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub lang: Option, +} + impl From<&FlakeValue> for TransportValue { fn from(v: &FlakeValue) -> Self { match v { @@ -154,6 +169,15 @@ impl From<&FlakeValue> for TransportValue { FlakeValue::Vector(v) => TransportValue::Vector(v.to_vec()), FlakeValue::Json(s) => TransportValue::Json(s.clone()), FlakeValue::GeoPoint(bits) => TransportValue::String(bits.to_string()), + FlakeValue::TripleTerm(t) => { + TransportValue::TripleTerm(Box::new(TransportTripleTerm { + s: TransportSid::from(&t.s), + p: TransportSid::from(&t.p), + o: TransportValue::from(&t.o), + dt: TransportSid::from(&t.dt), + lang: t.lang.clone(), + })) + } FlakeValue::Null => TransportValue::Null, } } @@ -202,6 +226,15 @@ impl TransportValue { } TransportValue::Vector(v) => Ok(FlakeValue::Vector(v.as_slice().into())), TransportValue::Json(s) => Ok(FlakeValue::Json(s.clone())), + TransportValue::TripleTerm(t) => { + Ok(FlakeValue::TripleTerm(Box::new(TripleTermValue { + s: t.s.to_sid(interner), + p: t.p.to_sid(interner), + o: t.o.to_flake_value(interner)?, + dt: t.dt.to_sid(interner), + lang: t.lang.clone(), + }))) + } TransportValue::Null => Ok(FlakeValue::Null), } } diff --git a/fluree-db-core/src/triple_term.rs b/fluree-db-core/src/triple_term.rs new file mode 100644 index 0000000000..249e38e47a --- /dev/null +++ b/fluree-db-core/src/triple_term.rs @@ -0,0 +1,129 @@ +//! Triple-term dictionary handles and encoded keys. +//! +//! A reification link `_:r rdf:reifies <<( s p o )>>` is one main-index flake +//! whose object is a term **handle** (`OType::TRIPLE_TERM`, `o_key`). The +//! handle is partitioned by the term's inner predicate so that every term +//! under one predicate occupies a contiguous `o_key` interval: a predicate +//! restriction on a reified-triple pattern becomes one `POST` range on +//! `rdf:reifies`. The precedent is `SubjectId`, `(ns_code << 48) | local`. +//! +//! The split is a deliberate constant: 32 bits of sequence per predicate, +//! 32 bits of predicate id. Both limits are enforced at allocation. + +use crate::o_type::OType; + +/// Bits of per-predicate sequence in a handle. +pub const TERM_SEQ_BITS: u32 = 32; +/// Mask for the sequence part of a handle. +pub const TERM_SEQ_MASK: u64 = (1u64 << TERM_SEQ_BITS) - 1; + +/// Compose a handle from the inner predicate id and a per-predicate sequence. +#[inline] +pub const fn term_handle(inner_p_id: u32, seq: u32) -> u64 { + ((inner_p_id as u64) << TERM_SEQ_BITS) | seq as u64 +} + +/// The inner predicate id a handle was allocated under. +#[inline] +pub const fn term_handle_p_id(handle: u64) -> u32 { + (handle >> TERM_SEQ_BITS) as u32 +} + +/// The per-predicate sequence of a handle. +#[inline] +pub const fn term_handle_seq(handle: u64) -> u32 { + (handle & TERM_SEQ_MASK) as u32 +} + +/// Inclusive `o_key` interval holding every term under `inner_p_id`. +#[inline] +pub const fn term_handle_range(inner_p_id: u32) -> (u64, u64) { + ( + term_handle(inner_p_id, 0), + term_handle(inner_p_id, u32::MAX), + ) +} + +/// The encoded identity of a triple term: the base edge\'s `(s_id, p_id, +/// o_type, o_key)` as the main index stores it. No graph, no list index. +/// +/// Its big-endian byte form is the reverse-tree key, ordered subject-first so +/// a subject-bound reified-triple pattern is one key range. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord)] +pub struct TermKey { + pub s_id: u64, + pub p_id: u32, + pub o_type: OType, + pub o_key: u64, +} + +impl TermKey { + /// Encoded key width in bytes. + pub const LEN: usize = 8 + 4 + 2 + 8; + + /// Big-endian key bytes: `s_id, p_id, o_type, o_key`. + #[inline] + pub fn to_be_bytes(&self) -> [u8; Self::LEN] { + let mut b = [0u8; Self::LEN]; + b[0..8].copy_from_slice(&self.s_id.to_be_bytes()); + b[8..12].copy_from_slice(&self.p_id.to_be_bytes()); + b[12..14].copy_from_slice(&self.o_type.as_u16().to_be_bytes()); + b[14..22].copy_from_slice(&self.o_key.to_be_bytes()); + b + } + + /// Decode a key written by [`Self::to_be_bytes`]. `None` on a wrong width. + #[inline] + pub fn from_be_bytes(b: &[u8]) -> Option { + if b.len() != Self::LEN { + return None; + } + Some(Self { + s_id: u64::from_be_bytes(b[0..8].try_into().ok()?), + p_id: u32::from_be_bytes(b[8..12].try_into().ok()?), + o_type: OType::from_u16(u16::from_be_bytes(b[12..14].try_into().ok()?)), + o_key: u64::from_be_bytes(b[14..22].try_into().ok()?), + }) + } + + /// Key prefix shared by every term with this subject: the first 8 bytes. + #[inline] + pub fn subject_prefix(s_id: u64) -> [u8; 8] { + s_id.to_be_bytes() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn handle_roundtrip_and_range() { + let h = term_handle(7, 42); + assert_eq!(term_handle_p_id(h), 7); + assert_eq!(term_handle_seq(h), 42); + let (lo, hi) = term_handle_range(7); + assert!(lo <= h && h <= hi); + assert!(term_handle(8, 0) > hi); + assert!(term_handle(6, u32::MAX) < lo); + } + + #[test] + fn key_roundtrip_orders_subject_first() { + let k = TermKey { + s_id: 0x0001_0000_0000_0005, + p_id: 9, + o_type: OType::IRI_REF, + o_key: 77, + }; + let b = k.to_be_bytes(); + assert_eq!(TermKey::from_be_bytes(&b), Some(k)); + let k2 = TermKey { + s_id: k.s_id + 1, + p_id: 0, + ..k + }; + assert!(k2.to_be_bytes() > b); + assert!(TermKey::from_be_bytes(&b[..10]).is_none()); + } +} diff --git a/fluree-db-core/src/value.rs b/fluree-db-core/src/value.rs index 98cf4f41f8..ab604b698d 100644 --- a/fluree-db-core/src/value.rs +++ b/fluree-db-core/src/value.rs @@ -109,6 +109,26 @@ impl fmt::Display for GeoPointBits { } } +/// An RDF 1.2 triple term: the identity of a triple `(s, p, o)`, carrying the +/// object's datatype and language tag. +/// +/// Graph-independent and without a list index, per RDF 1.2 Concepts: identical +/// triples in different graphs are the same term, and the link flake's own +/// graph scopes a reification to an occurrence. This is the materialized form +/// carried by commits and novelty; the index stores a dictionary handle in +/// `o_key` under [`crate::o_type::OType::TRIPLE_TERM`]. +/// +/// Serializes as a map (`s`, `p`, `o`, `dt`, `lang`), which no other +/// `FlakeValue` variant accepts, so it is unambiguous under `serde(untagged)`. +#[derive(Clone, Debug, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] +pub struct TripleTermValue { + pub s: Sid, + pub p: Sid, + pub o: FlakeValue, + pub dt: Sid, + pub lang: Option, +} + /// Polymorphic value type for flake objects /// /// Covers XSD datatypes supported by Fluree. @@ -166,6 +186,9 @@ pub enum FlakeValue { Json(String), /// Geographic point (geo:wktLiteral POINT) — packed 60-bit lat/lng GeoPoint(GeoPointBits), + /// RDF 1.2 triple term (`<<( s p o )>>`). Boxed: the term embeds a + /// full object value, and the enum must stay narrow on the scan path. + TripleTerm(Box), /// Null/None value Null, } @@ -227,8 +250,9 @@ impl FlakeValue { FlakeValue::String(_) => 18, FlakeValue::Json(_) => 19, FlakeValue::GeoPoint(_) => 20, + FlakeValue::TripleTerm(_) => 21, // Vector MUST be highest discriminant: empty Vector is used as max() sentinel - FlakeValue::Vector(_) => 21, + FlakeValue::Vector(_) => 22, } } @@ -306,6 +330,19 @@ impl FlakeValue { matches!(self, FlakeValue::Vector(_)) } + /// Check if this is an RDF 1.2 triple term + pub fn is_triple_term(&self) -> bool { + matches!(self, FlakeValue::TripleTerm(_)) + } + + /// Borrow the triple term, if this is one + pub fn as_triple_term(&self) -> Option<&TripleTermValue> { + match self { + FlakeValue::TripleTerm(t) => Some(t), + _ => None, + } + } + /// Try to get as i64 pub fn as_long(&self) -> Option { match self { @@ -544,6 +581,7 @@ impl FlakeValue { } // GeoPoint: compare by packed u64 (latitude-primary ordering) (FlakeValue::GeoPoint(a), FlakeValue::GeoPoint(b)) => a.cmp(b), + (FlakeValue::TripleTerm(a), FlakeValue::TripleTerm(b)) => a.cmp(b), // Should not happen since discriminants are equal _ => Ordering::Equal, } @@ -746,6 +784,25 @@ impl FlakeValue { buf[1..].copy_from_slice(&bits.as_u64().to_le_bytes()); xxh64(&buf, 0) } + FlakeValue::TripleTerm(t) => { + use xxhash_rust::xxh64::Xxh64; + let mut hasher = Xxh64::new(0); + hasher.update(&[0x16]); // type tag for TripleTerm + hasher.update(&t.s.namespace_code.to_le_bytes()); + hasher.update(t.s.name.as_bytes()); + hasher.update(&[0]); + hasher.update(&t.p.namespace_code.to_le_bytes()); + hasher.update(t.p.name.as_bytes()); + hasher.update(&[0]); + hasher.update(&t.o.canonical_hash().to_le_bytes()); + hasher.update(&t.dt.namespace_code.to_le_bytes()); + hasher.update(t.dt.name.as_bytes()); + hasher.update(&[0]); + if let Some(lang) = &t.lang { + hasher.update(lang.as_bytes()); + } + hasher.digest() + } } } } @@ -791,6 +848,7 @@ impl PartialEq for FlakeValue { .all(|(x, y)| x.to_bits() == y.to_bits()) } (FlakeValue::GeoPoint(a), FlakeValue::GeoPoint(b)) => a == b, + (FlakeValue::TripleTerm(a), FlakeValue::TripleTerm(b)) => a == b, // Numeric and temporal types already handled above _ => false, } @@ -993,6 +1051,10 @@ impl std::hash::Hash for FlakeValue { self.type_discriminant().hash(state); bits.hash(state); } + FlakeValue::TripleTerm(t) => { + self.type_discriminant().hash(state); + t.hash(state); + } } } } @@ -1034,6 +1096,7 @@ impl fmt::Display for FlakeValue { write!(f, "]") } FlakeValue::GeoPoint(bits) => write!(f, "{bits}"), + FlakeValue::TripleTerm(t) => write!(f, "<<( {} {} {} )>>", t.s, t.p, t.o), } } } diff --git a/fluree-db-core/src/value_id.rs b/fluree-db-core/src/value_id.rs index 38b8e0d484..60d7600da3 100644 --- a/fluree-db-core/src/value_id.rs +++ b/fluree-db-core/src/value_id.rs @@ -116,6 +116,11 @@ impl ObjKind { /// Precision: approximately 0.3mm at the equator. pub const GEO_POINT: Self = Self(0x14); + /// RDF 1.2 triple term — triple-term dictionary handle (full u64, + /// `(inner p_id << 32) | seq`). Only ever produced by the index-side + /// interning of reification links; commits carry the materialized term. + pub const TRIPLE_TERM: Self = Self(0x15); + /// Get the raw `u8` discriminant. #[inline] pub const fn as_u8(self) -> u8 { diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index 4acbdb858f..718ce2f61e 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -820,7 +820,7 @@ impl BinaryScanOperator { // for debug / inspection workflows. if self.p_is_var && !self.include_system_facts - && fluree_db_core::is_reserved_reifies_predicate(&flake.p) + && fluree_db_core::is_scan_hidden_predicate(&flake.p) { continue; } @@ -1201,7 +1201,7 @@ impl BinaryScanOperator { None => return false, }, }; - fluree_db_core::is_reserved_reifies_predicate(sid) + fluree_db_core::is_scan_hidden_predicate(sid) } /// Enforce within-pattern repeated-variable constraints. diff --git a/fluree-db-query/src/dict_overlay.rs b/fluree-db-query/src/dict_overlay.rs index 95b0740097..e87ee5a0b0 100644 --- a/fluree-db-query/src/dict_overlay.rs +++ b/fluree-db-query/src/dict_overlay.rs @@ -567,6 +567,9 @@ impl DictOverlay { FlakeValue::Vector(v) => Ok(self.assign_vector_handle(v)), FlakeValue::GeoPoint(bits) => Ok((ObjKind::GEO_POINT, ObjKey::from_u64(bits.as_u64()))), + FlakeValue::TripleTerm(_) => Err(io::Error::other( + "triple-term objects are interned by the index build, not the dictionary overlay", + )), } } diff --git a/fluree-db-query/src/eval/helpers.rs b/fluree-db-query/src/eval/helpers.rs index 0d4f62c81b..7dd4b9fc82 100644 --- a/fluree-db-query/src/eval/helpers.rs +++ b/fluree-db-query/src/eval/helpers.rs @@ -665,6 +665,7 @@ fn hash_flake_value(value: &FlakeValue, state: &mut impl Hasher) { } FlakeValue::Json(v) => v.hash(state), FlakeValue::GeoPoint(v) => v.0.hash(state), + FlakeValue::TripleTerm(t) => t.hash(state), FlakeValue::Null => {} } } diff --git a/fluree-db-query/src/eval/metadata.rs b/fluree-db-query/src/eval/metadata.rs index 3cd786c73c..d3876c34c1 100644 --- a/fluree-db-query/src/eval/metadata.rs +++ b/fluree-db-query/src/eval/metadata.rs @@ -522,7 +522,7 @@ fn data_properties_from_flakes(mut flakes: Vec) -> Vec { // the reifier sidecar. if matches!(flake.o, FlakeValue::Ref(_)) || fluree_db_core::is_rdf_type(&flake.p) - || fluree_db_core::is_reserved_reifies_predicate(&flake.p) + || fluree_db_core::is_scan_hidden_predicate(&flake.p) { continue; } diff --git a/fluree-db-query/src/eval/value.rs b/fluree-db-query/src/eval/value.rs index ab959a0a40..8675ac9ffc 100644 --- a/fluree-db-query/src/eval/value.rs +++ b/fluree-db-query/src/eval/value.rs @@ -972,7 +972,8 @@ impl TryFrom<&FlakeValue> for ComparableValue { | FlakeValue::GMonthDay(_) | FlakeValue::YearMonthDuration(_) | FlakeValue::DayTimeDuration(_) - | FlakeValue::Duration(_) => Ok(ComparableValue::TypedLiteral { + | FlakeValue::Duration(_) + | FlakeValue::TripleTerm(_) => Ok(ComparableValue::TypedLiteral { val: val.clone(), dtc: None, }), @@ -1006,7 +1007,8 @@ impl TryFrom for ComparableValue { | FlakeValue::GMonthDay(_) | FlakeValue::YearMonthDuration(_) | FlakeValue::DayTimeDuration(_) - | FlakeValue::Duration(_)) => Ok(ComparableValue::TypedLiteral { val, dtc: None }), + | FlakeValue::Duration(_) + | FlakeValue::TripleTerm(_)) => Ok(ComparableValue::TypedLiteral { val, dtc: None }), } } } diff --git a/fluree-db-query/src/fast_whole_graph_agg.rs b/fluree-db-query/src/fast_whole_graph_agg.rs index aba0bb4fee..a4afc842e3 100644 --- a/fluree-db-query/src/fast_whole_graph_agg.rs +++ b/fluree-db-query/src/fast_whole_graph_agg.rs @@ -631,7 +631,7 @@ fn overlay_all_subjects_count( // Mirror the pipeline's `?n ?p ?o` visibility: `f:reifies*` is // invisible to the scan but present in SPOT — its subjects must // not be counted. - if fluree_db_core::is_reserved_reifies_predicate(&flake.p) { + if fluree_db_core::is_scan_hidden_predicate(&flake.p) { declined = true; return; } @@ -1085,7 +1085,7 @@ pub(crate) fn graph_has_scan_hidden_predicates( // Unresolvable dictionary entry — err toward declining. return Ok(true); }; - if fluree_db_core::is_reserved_reifies_predicate(&sid) + if fluree_db_core::is_scan_hidden_predicate(&sid) && count_rows_for_predicate_psot(store, ctx.binary_g_id, p_id)? > 0 { return Ok(true); diff --git a/fluree-db-query/src/materializer.rs b/fluree-db-query/src/materializer.rs index 0b437265d1..4dfbe333fb 100644 --- a/fluree-db-query/src/materializer.rs +++ b/fluree-db-query/src/materializer.rs @@ -107,6 +107,7 @@ impl Hash for FlakeValueKey { FlakeValue::DayTimeDuration(v) => v.hash(state), FlakeValue::Duration(v) => v.hash(state), FlakeValue::GeoPoint(v) => v.hash(state), + FlakeValue::TripleTerm(t) => t.hash(state), } } } @@ -675,6 +676,7 @@ fn flake_value_to_comparable(val: &FlakeValue) -> Option { FlakeValue::DayTimeDuration(v) => Some(ComparableValue::String(Arc::from(v.to_string()))), FlakeValue::Duration(v) => Some(ComparableValue::String(Arc::from(v.to_string()))), FlakeValue::GeoPoint(v) => Some(ComparableValue::String(Arc::from(v.to_string()))), + FlakeValue::TripleTerm(_) => Some(ComparableValue::String(Arc::from(val.to_string()))), } } diff --git a/fluree-db-query/src/property_path.rs b/fluree-db-query/src/property_path.rs index a77435b55f..ba7ee994bb 100644 --- a/fluree-db-query/src/property_path.rs +++ b/fluree-db-query/src/property_path.rs @@ -80,7 +80,7 @@ pub fn path_max_visited() -> usize { /// properties are already excluded by the `Ref`-object filter in the scan. #[inline] pub(crate) fn is_reserved_edge_predicate(p: &Sid) -> bool { - fluree_db_core::is_rdf_type(p) || fluree_db_core::is_reserved_reifies_predicate(p) + fluree_db_core::is_rdf_type(p) || fluree_db_core::is_scan_hidden_predicate(p) } /// Re-encode a pattern-constant predicate `Sid` into the active graph's dict, @@ -684,7 +684,7 @@ impl PropertyPathOperator { // cancellation poll, so it polls here — a large `?s :p* ?o` stays // killable by the query timeout instead of running to completion. crate::fast_path_common::bail_if_cancelled(&ctx.cancellation)?; - if fluree_db_core::is_reserved_reifies_predicate(&flake.p) { + if fluree_db_core::is_scan_hidden_predicate(&flake.p) { continue; } let Flake { s, o, dt, m, .. } = flake; diff --git a/fluree-db-reasoner/src/execute/derived.rs b/fluree-db-reasoner/src/execute/derived.rs index 28ee969c75..83f0369545 100644 --- a/fluree-db-reasoner/src/execute/derived.rs +++ b/fluree-db-reasoner/src/execute/derived.rs @@ -75,6 +75,15 @@ impl DerivedSet { f.to_bits().hash(&mut hasher); } } + FlakeValue::TripleTerm(t) => { + 40u8.hash(&mut hasher); + Self::object_hash(&FlakeValue::Ref(t.s.clone())).hash(&mut hasher); + Self::object_hash(&FlakeValue::Ref(t.p.clone())).hash(&mut hasher); + Self::object_hash(&t.o).hash(&mut hasher); + t.dt.namespace_code.hash(&mut hasher); + t.dt.name.hash(&mut hasher); + t.lang.hash(&mut hasher); + } FlakeValue::Null => { 7u8.hash(&mut hasher); } diff --git a/fluree-db-shacl/src/constraints/datatype.rs b/fluree-db-shacl/src/constraints/datatype.rs index feda6f6fe5..3e4e8c6df7 100644 --- a/fluree-db-shacl/src/constraints/datatype.rs +++ b/fluree-db-shacl/src/constraints/datatype.rs @@ -95,7 +95,8 @@ fn infer_node_kind(value: &FlakeValue) -> Option { | FlakeValue::Json(_) | FlakeValue::GeoPoint(_) => Some(NodeKind::Literal), FlakeValue::Vector(_) => Some(NodeKind::Literal), // Treat vectors as literals - FlakeValue::Null => None, + // RDF 1.2 triple terms are neither IRIs, blank nodes nor literals. + FlakeValue::TripleTerm(_) | FlakeValue::Null => None, } } diff --git a/fluree-db-shacl/src/constraints/pattern.rs b/fluree-db-shacl/src/constraints/pattern.rs index ab3dd8ee2b..106a6d83fa 100644 --- a/fluree-db-shacl/src/constraints/pattern.rs +++ b/fluree-db-shacl/src/constraints/pattern.rs @@ -30,7 +30,10 @@ fn pattern_lexical_form(value: &FlakeValue) -> Option { FlakeValue::Duration(v) => Some(v.original().to_string()), FlakeValue::Json(s) => Some(s.clone()), FlakeValue::GeoPoint(v) => Some(v.to_string()), - FlakeValue::Ref(_) | FlakeValue::Vector(_) | FlakeValue::Null => None, + FlakeValue::Ref(_) + | FlakeValue::Vector(_) + | FlakeValue::TripleTerm(_) + | FlakeValue::Null => None, } } diff --git a/fluree-db-transact/src/generate/flakes.rs b/fluree-db-transact/src/generate/flakes.rs index fc75cdbe80..a95ed834b4 100644 --- a/fluree-db-transact/src/generate/flakes.rs +++ b/fluree-db-transact/src/generate/flakes.rs @@ -617,6 +617,7 @@ pub fn infer_datatype(val: &FlakeValue) -> Sid { FlakeValue::DayTimeDuration(_) => DT_DAY_TIME_DURATION.clone(), FlakeValue::Duration(_) => DT_DURATION.clone(), FlakeValue::GeoPoint(_) => DT_WKT_LITERAL.clone(), + FlakeValue::TripleTerm(_) => fluree_db_core::triple_term_datatype_sid().clone(), // Null isn't a standard RDF literal; treat as xsd:string for now (MVP). FlakeValue::Null => DT_STRING.clone(), } diff --git a/fluree-db-transact/src/import_sink.rs b/fluree-db-transact/src/import_sink.rs index 7cca28d445..e2ad615495 100644 --- a/fluree-db-transact/src/import_sink.rs +++ b/fluree-db-transact/src/import_sink.rs @@ -461,6 +461,9 @@ mod inner { (ObjKind::JSON_ID.as_u8(), ObjKey::encode_u32_id(id).as_u64()) } FlakeValue::GeoPoint(bits) => (ObjKind::GEO_POINT.as_u8(), bits.0), + // Only the link path produces term objects, through the + // chunk's term table; a materialized term here has no encoding. + FlakeValue::TripleTerm(_) => return None, FlakeValue::BigInt(bi) => { use num_bigint::BigInt; use std::convert::TryFrom; diff --git a/fluree-vocab/src/lib.rs b/fluree-vocab/src/lib.rs index 65d25745fd..cd475f40a1 100644 --- a/fluree-vocab/src/lib.rs +++ b/fluree-vocab/src/lib.rs @@ -581,6 +581,9 @@ pub mod rdf_names { /// rdf:nil local name pub const NIL: &str = "nil"; + + /// rdf:reifies local name (RDF 1.2 reifier predicate) + pub const REIFIES: &str = "reifies"; } /// OWL vocabulary constants @@ -1631,6 +1634,9 @@ pub mod fluree { /// The `@fulltext` shorthand in JSON-LD resolves to this IRI. pub const FULL_TEXT: &str = "https://ns.flur.ee/db#fullText"; + /// f:tripleTerm — datatype marker of an RDF 1.2 triple-term object + pub const TRIPLE_TERM: &str = "https://ns.flur.ee/db#tripleTerm"; + /// Full IRI for db:t predicate (used in RDF-Star annotation matching) pub const DB_T: &str = "https://ns.flur.ee/db#t"; @@ -1917,6 +1923,10 @@ pub mod db { /// edge is a list element. Always omitted in v1 (list-occurrence /// annotations are deferred). pub const REIFIES_LIST_INDEX: &str = "reifiesListIndex"; + + /// db:tripleTerm - the datatype marker Fluree gives an RDF 1.2 triple + /// term object (`<<( s p o )>>`), the way `@id` marks a reference. + pub const TRIPLE_TERM: &str = "tripleTerm"; } /// Edge-annotation system predicate IRIs (`https://ns.flur.ee/db#reifies*`). From 27783f835d63cca3473696a65f742a06e9e9a263 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 11:29:41 -0400 Subject: [PATCH 03/92] feat(index): triple-term dictionary, interned at bulk import The link `_:r rdf:reifies <<( s p o )>>` is now one main-index flake on imported ledgers, alongside the `f:reifies*` bundle it is meant to replace. Its object is a handle into a ledger-global term dictionary that mirrors the subject dictionary: one forward pack stream per inner predicate keyed by sequence, one reverse tree keyed by the encoded base edge, and references in a flagged trailing root section (`FLAG_EXT_HAS_TERM_DICT`) that the core metadata parser skips. Import interns terms after the global remap, where ids are final. The sink records each reified edge as a pseudo-record in a per-chunk term table and spools the link with the entry's ordinal; both build phases resolve the ordinal through `TermRemapCtx`, which remaps the entry with the chunk's own tables and interns it in a build-wide `TermDictBuilder`. The dictionary is uploaded next to the index artifacts and referenced from the root, and the store decodes a handle back to the materialized term. Full rebuilds and incremental builds do not intern yet (a rebuild drops the term section), and the query lowering still takes the bundle path. --- fluree-db-api/src/import.rs | 49 ++- fluree-db-api/tests/grp_import.rs | 1 + fluree-db-api/tests/it_triple_term_links.rs | 123 ++++++ .../src/dict/forward_pack.rs | 4 + fluree-db-binary-index/src/dict/mod.rs | 2 + fluree-db-binary-index/src/dict/term_dict.rs | 353 ++++++++++++++++++ .../src/format/expanded_cas.rs | 1 + .../src/format/index_root.rs | 50 ++- .../src/format/wire_helpers.rs | 102 +++++ fluree-db-binary-index/src/lib.rs | 2 +- .../src/read/binary_index_store.rs | 105 +++++- fluree-db-core/src/content_kind.rs | 5 + fluree-db-core/src/db.rs | 39 ++ fluree-db-indexer/src/build/rebuild.rs | 2 + fluree-db-indexer/src/build/root_assembly.rs | 4 + fluree-db-indexer/src/drop.rs | 1 + fluree-db-indexer/src/gc/test_support.rs | 1 + .../src/run_index/build/build_from_commits.rs | 45 ++- .../src/run_index/build/incremental_root.rs | 1 + fluree-db-indexer/src/run_index/runs/spool.rs | 123 +++++- fluree-db-transact/src/import_sink.rs | 127 ++++++- 21 files changed, 1121 insertions(+), 19 deletions(-) create mode 100644 fluree-db-api/tests/it_triple_term_links.rs create mode 100644 fluree-db-binary-index/src/dict/term_dict.rs diff --git a/fluree-db-api/src/import.rs b/fluree-db-api/src/import.rs index dfb90296be..36761d78bb 100644 --- a/fluree-db-api/src/import.rs +++ b/fluree-db-api/src/import.rs @@ -4115,6 +4115,7 @@ where let otype_registry = fluree_db_core::OTypeRegistry::new(&custom_datatype_iris); let r = fluree_db_indexer::run_index::spool::sort_remap_and_write_sorted_commit( sr.records, + sr.terms, sr.subjects, sr.strings, &vd.join(format!("chunk_{ci:05}.subjects.voc")), @@ -5649,6 +5650,7 @@ where fluree_db_core::OTypeRegistry::new(&meta_custom_datatype_iris); let meta_sorted_info = sort_remap_and_write_sorted_commit( records, + Vec::new(), meta_subjects, meta_strings, &subj_voc_path, @@ -6321,7 +6323,14 @@ where ); let mut v3_handle = tokio::task::spawn_blocking( - move || -> std::result::Result<(_, Option), ImportError> { + move || -> std::result::Result< + ( + _, + Option, + std::sync::Arc>, + ), + ImportError, + > { let commits: Vec = v3_sorted_commit_infos .iter() .enumerate() @@ -6334,6 +6343,7 @@ where string_remap_path: remap_dir.join(format!("strings_{i:05}.rmp")), lang_remap: v3_lang_remaps.get(i).cloned().unwrap_or_default(), types_map_path: info.types_map_path.clone(), + term_table: info.term_table.clone(), } }) .collect(); @@ -6362,6 +6372,10 @@ where // limit (post best-effort raise at startup/preflight). let fd_budget = fluree_db_core::fd_limit::FdBudget::detect(); + // Build-wide triple-term interner shared by every graph's build. + let term_builder = std::sync::Arc::new(std::sync::Mutex::new( + fluree_db_binary_index::dict::TermDictBuilder::new(), + )); let cfg_g0 = fluree_db_indexer::BuildConfig { run_dir: v3_runs_g0, index_dir: v3_index_dir.clone(), @@ -6375,6 +6389,7 @@ where remap_progress: Some(v3_remap_counter), build_progress: Some(v3_build_counter), stage_marker: Some(v3_stage_marker), + term_builder: Some(std::sync::Arc::clone(&term_builder)), }; std::fs::create_dir_all(&cfg_g0.run_dir).map_err(|e| index_build_error(&e))?; @@ -6404,6 +6419,7 @@ where remap_progress: None, build_progress: None, stage_marker: None, + term_builder: Some(std::sync::Arc::clone(&term_builder)), }; std::fs::create_dir_all(&cfg_g1.run_dir).map_err(|e| index_build_error(&e))?; @@ -6441,6 +6457,7 @@ where remap_progress: None, build_progress: None, stage_marker: None, + term_builder: Some(std::sync::Arc::clone(&term_builder)), }; std::fs::create_dir_all(&cfg_ng.run_dir).map_err(|e| index_build_error(&e))?; @@ -6545,7 +6562,7 @@ where "V3 index build complete" ); - Ok((result, stats_output)) + Ok((result, stats_output, term_builder)) }, ); @@ -6600,7 +6617,7 @@ where let index_start = std::time::Instant::now(); let mut current_stage = fluree_db_indexer::BUILD_STAGE_REMAP; let mut stage_start = index_start; - let (v3_result, stats_output) = loop { + let (v3_result, stats_output, term_dict_builder) = loop { tokio::select! { result = &mut v3_handle => { let stage = stage_marker.load(std::sync::atomic::Ordering::Relaxed); @@ -6615,6 +6632,31 @@ where } }; + // Persist the triple-term dictionary the build interned, if any. + let term_dict_refs = { + let builder = + std::mem::take(&mut *term_dict_builder.lock().map_err(|_| { + ImportError::IndexBuild("term dictionary lock poisoned".into()) + })?); + if builder.is_empty() { + None + } else { + let term_count = builder.len(); + let started = Instant::now(); + let refs = builder + .upload(content_store.as_ref()) + .await + .map_err(|e| ImportError::Upload(format!("triple-term dictionary: {e}")))?; + tracing::info!( + term_count, + predicates = refs.forward_packs.len(), + elapsed_ms = started.elapsed().as_millis(), + "triple-term dictionary uploaded" + ); + Some(refs) + } + }; + // Upload V3 artifacts to CAS. config.emit_progress(ImportPhase::PreparingIndex { stage: "Uploading index artifacts", @@ -6937,6 +6979,7 @@ where // dict moves into this struct literal. has_annotations: import_has_annotations, annotation_index: None, + term_dict: term_dict_refs, // Sticky-bit canonical contract lives on // `IndexRoot.had_annotation_arena` in // `fluree-db-binary-index/src/format/index_root.rs`. diff --git a/fluree-db-api/tests/grp_import.rs b/fluree-db-api/tests/grp_import.rs index 784ae1a5e4..d153429ee7 100644 --- a/fluree-db-api/tests/grp_import.rs +++ b/fluree-db-api/tests/grp_import.rs @@ -25,3 +25,4 @@ mod it_namespace_new_after_index; mod it_ns_sync_conflict; #[path = "it_pack_validation.rs"] mod it_pack_validation; +mod it_triple_term_links; diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs new file mode 100644 index 0000000000..d3833eadb4 --- /dev/null +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -0,0 +1,123 @@ +//! The RDF 1.2 link form on the bulk-import path. +//! +//! Alongside the `f:reifies*` bundle, import writes one `rdf:reifies` flake +//! per reifier whose object is a triple-term handle, and persists the term +//! dictionary that gives the handle its meaning. Reading the link back +//! through a query exercises the whole chain: sink → chunk term table → +//! global remap and interning → dictionary upload → root section → store +//! load → handle decode. + +#![cfg(feature = "native")] + +use crate::support; +use fluree_db_api::{FlureeBuilder, LedgerState}; +use serde_json::Value as JsonValue; + +const CLAIMS: &str = r#"VERSION "1.2" +@prefix ex: . + +ex:alice ex:knows ex:bob ~ ex:claim1 {| ex:confidence 0.9 ; ex:source ex:hr |} . +ex:alice ex:knows ex:carol {| ex:source ex:linkedin |} . +<< ex:bob ex:knows ex:dave >> ex:source ex:crm . +ex:carol ex:age 42 ~ ex:claim2 . +"#; + +async fn import(files: &[(&str, &str)], alias: &str) -> (fluree_db_api::Fluree, LedgerState) { + let db_dir = tempfile::tempdir().expect("db tmpdir"); + let data_dir = tempfile::tempdir().expect("data tmpdir"); + for (name, content) in files { + std::fs::write(data_dir.path().join(name), content).expect("write fixture"); + } + let fluree = FlureeBuilder::file(db_dir.path().to_string_lossy().to_string()) + .build() + .expect("build file-backed Fluree"); + fluree + .create(alias) + .import(data_dir.path()) + .threads(1) + .memory_budget_mb(256) + .cleanup(false) + .execute() + .await + .expect("import of Turtle-star must succeed"); + let ledger = fluree.ledger(alias).await.expect("load ledger"); + std::mem::forget(db_dir); + std::mem::forget(data_dir); + (fluree, ledger) +} + +fn rows(result: &JsonValue) -> Vec> { + result + .as_array() + .expect("row array") + .iter() + .map(|row| { + row.as_array() + .expect("row") + .iter() + .map(|cell| match cell { + JsonValue::String(s) => s.clone(), + JsonValue::Number(n) => n.to_string(), + other => other + .get("@id") + .or_else(|| other.get("@value")) + .map(|v| match v { + JsonValue::String(s) => s.clone(), + v => v.to_string(), + }) + .unwrap_or_else(|| other.to_string()), + }) + .collect() + }) + .collect() +} + +#[tokio::test] +async fn imported_reifiers_carry_a_decodable_triple_term_link() { + let (fluree, ledger) = import(&[("claims.ttl", CLAIMS)], "it/triple-term-links:claims").await; + + // One link per reifier, and each link's object decodes through the term + // dictionary to the base edge it reifies. + let sparql = "PREFIX ex: \n\ + PREFIX rdf: \n\ + SELECT ?r ?t WHERE { ?r rdf:reifies ?t } ORDER BY ?r"; + let result = support::query_sparql_formatted(&fluree, &ledger, sparql) + .await + .expect("link query over an imported ledger"); + let got = rows(&result); + assert_eq!(got.len(), 4, "{got:#?}"); + + let term_of = |reifier: &str| -> String { + got.iter() + .find(|r| r[0].ends_with(reifier)) + .unwrap_or_else(|| panic!("no link for {reifier}: {got:#?}"))[1] + .clone() + }; + let t1 = term_of("claim1"); + assert!( + t1.contains("alice") && t1.contains("knows") && t1.contains("bob"), + "claim1 must reify alice knows bob: {t1}" + ); + let t2 = term_of("claim2"); + assert!( + t2.contains("carol") && t2.contains("age") && t2.contains("42"), + "claim2 must reify carol age 42 (literal object): {t2}" + ); + + // While the link form is index-internal, wildcard scans keep hiding it + // exactly as they hide the `f:reifies*` bundle. + let sparql = "PREFIX ex: \n\ + SELECT ?p WHERE { ex:claim1 ?p ?o } ORDER BY ?p"; + let result = support::query_sparql_formatted(&fluree, &ledger, sparql) + .await + .expect("wildcard predicate scan"); + let preds: Vec = rows(&result).into_iter().map(|r| r[0].clone()).collect(); + assert!( + !preds.iter().any(|p| p.contains("reifies")), + "wildcard scan must hide rdf:reifies and f:reifies*: {preds:?}" + ); + assert!( + preds.iter().any(|p| p.ends_with("confidence")), + "the claim body stays visible: {preds:?}" + ); +} diff --git a/fluree-db-binary-index/src/dict/forward_pack.rs b/fluree-db-binary-index/src/dict/forward_pack.rs index fe1e35bdf6..0c9ed6b5e6 100644 --- a/fluree-db-binary-index/src/dict/forward_pack.rs +++ b/fluree-db-binary-index/src/dict/forward_pack.rs @@ -57,6 +57,10 @@ pub const KIND_STRING_FWD: u8 = 0; /// Pack kind: subject forward dictionary (one pack per namespace). pub const KIND_SUBJECT_FWD: u8 = 1; +/// Pack kind: triple-term forward dictionary (one pack stream per inner +/// predicate; entries are encoded `TermKey`s keyed by per-predicate sequence). +pub const KIND_TERM_FWD: u8 = 2; + // ============================================================================ // Pack header // ============================================================================ diff --git a/fluree-db-binary-index/src/dict/mod.rs b/fluree-db-binary-index/src/dict/mod.rs index 8ae474c25f..3d5c90b9aa 100644 --- a/fluree-db-binary-index/src/dict/mod.rs +++ b/fluree-db-binary-index/src/dict/mod.rs @@ -41,6 +41,7 @@ pub mod pack_builder; pub mod pack_reader; pub mod reader; pub mod reverse_leaf; +pub mod term_dict; pub mod varint; pub use branch::DictBranch; @@ -49,3 +50,4 @@ pub use forward_pack::ForwardPack; pub use global_dict::{LanguageTagDict, PredicateDict}; pub use pack_reader::ForwardPackReader; pub use reader::DictTreeReader; +pub use term_dict::{TermDictBuilder, TermDictReader}; diff --git a/fluree-db-binary-index/src/dict/term_dict.rs b/fluree-db-binary-index/src/dict/term_dict.rs new file mode 100644 index 0000000000..2d04a11ed3 --- /dev/null +++ b/fluree-db-binary-index/src/dict/term_dict.rs @@ -0,0 +1,353 @@ +//! Triple-term dictionary: `OType::TRIPLE_TERM` handles ↔ encoded base edges. +//! +//! One reification link is a main-index flake `_:r rdf:reifies `. +//! This dictionary gives the handle its meaning: a [`TermKey`], the base +//! edge's `(s_id, p_id, o_type, o_key)` exactly as the main index stores it. +//! +//! Layout mirrors the subject dictionary with the inner predicate in the +//! namespace role: one forward pack stream per inner predicate keyed by the +//! per-predicate sequence (the low half of the handle), and one reverse tree +//! keyed by the big-endian `TermKey` bytes (subject-first, so a subject-bound +//! reified-triple pattern is one key range). + +use crate::dict::builder::{build_reverse_tree, finalize_branch, DEFAULT_TARGET_LEAF_BYTES}; +use crate::dict::forward_pack::{encode_forward_pack, KIND_TERM_FWD}; +use crate::dict::pack_builder::{DEFAULT_TARGET_PACK_BYTES, DEFAULT_TARGET_PAGE_BYTES}; +use crate::dict::pack_reader::ForwardPackReader; +use crate::dict::reader::DictTreeReader; +use crate::dict::reverse_leaf::ReverseEntry; +use crate::format::wire_helpers::{DictTreeRefs, PackBranchEntry, TermDictRefs}; +use crate::read::leaflet_cache::LeafletCache; +use fluree_db_core::triple_term::{term_handle, term_handle_p_id, term_handle_seq, TermKey}; +use fluree_db_core::{ContentId, ContentKind, ContentStore, DictKind}; +use std::collections::{BTreeMap, HashMap}; +use std::io; +use std::path::Path; +use std::sync::Arc; + +/// The pack header's `ns_code` slot is 16 bits; the low half of the inner +/// predicate id goes there as a read-time consistency check. +#[inline] +fn pack_ns_code(p_id: u32) -> u16 { + (p_id & 0xFFFF) as u16 +} + +// ── Reader ────────────────────────────────────────────────────────────────── + +/// Read side of the triple-term dictionary. +pub struct TermDictReader { + forward: BTreeMap, + reverse: Option>, + watermarks: HashMap, + term_count: u64, +} + +impl TermDictReader { + /// Open the dictionary named by `refs`, reusing `prev`'s open handles + /// for packs and leaves whose CIDs are unchanged. + pub async fn from_refs_reusing( + cs: Arc, + cache_dir: &Path, + refs: &TermDictRefs, + leaflet_cache: Option<&Arc>, + prev: Option<&TermDictReader>, + ) -> io::Result { + let mut forward = BTreeMap::new(); + for (p_id, packs) in &refs.forward_packs { + let reader = ForwardPackReader::from_pack_refs_reusing( + Arc::clone(&cs), + cache_dir, + packs, + KIND_TERM_FWD, + pack_ns_code(*p_id), + prev.and_then(|p| p.forward.get(p_id)), + ) + .await?; + forward.insert(*p_id, reader); + } + let reverse = Some( + DictTreeReader::from_refs_reusing( + &cs, + &refs.reverse, + leaflet_cache, + Some(cache_dir), + prev.and_then(|p| p.reverse.as_ref()), + ) + .await?, + ); + Ok(Self { + forward, + reverse, + watermarks: refs.watermarks.iter().copied().collect(), + term_count: refs.term_count, + }) + } + + /// Distinct terms in the dictionary. + pub fn term_count(&self) -> u64 { + self.term_count + } + + /// Highest sequence allocated under `p_id`, if any term exists there. + pub fn watermark(&self, p_id: u32) -> Option { + self.watermarks.get(&p_id).copied() + } + + /// Inner predicates that have at least one term. + pub fn predicates(&self) -> impl Iterator + '_ { + self.forward.keys().copied() + } + + /// The handle for an encoded base edge, if it has been interned. + pub fn find_handle(&self, key: &TermKey) -> io::Result> { + match &self.reverse { + Some(tree) => tree.reverse_lookup(&key.to_be_bytes()), + None => Ok(None), + } + } + + /// The encoded base edge behind `handle`, or `None` for an unknown handle. + pub fn resolve(&self, handle: u64) -> io::Result> { + let Some(reader) = self.forward.get(&term_handle_p_id(handle)) else { + return Ok(None); + }; + let mut buf = Vec::with_capacity(TermKey::LEN); + if !reader.forward_lookup_into(term_handle_seq(handle) as u64, &mut buf)? { + return Ok(None); + } + TermKey::from_be_bytes(&buf) + .ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + format!( + "term handle {handle:#x}: malformed forward entry ({} bytes)", + buf.len() + ), + ) + }) + .map(Some) + } +} + +// ── Builder ───────────────────────────────────────────────────────────────── + +/// In-memory interner used while a build assigns handles. +/// +/// Sequences continue above any watermarks the builder was created with, so +/// an incremental build allocates fresh handles without touching existing +/// ones. +#[derive(Default)] +pub struct TermDictBuilder { + map: HashMap, + next_seq: HashMap, +} + +impl TermDictBuilder { + /// A builder whose first allocation per predicate starts at zero. + pub fn new() -> Self { + Self::default() + } + + /// A builder that continues each predicate's sequence above `watermarks`. + pub fn above_watermarks(watermarks: &[(u32, u32)]) -> Self { + Self { + map: HashMap::new(), + next_seq: watermarks + .iter() + .map(|(p_id, wm)| (*p_id, wm.saturating_add(1))) + .collect(), + } + } + + /// The handle already assigned to `key` in this builder. + pub fn get(&self, key: &TermKey) -> Option { + self.map.get(key).copied() + } + + /// The handle for `key`, allocating one under its inner predicate if new. + pub fn get_or_insert(&mut self, key: TermKey) -> io::Result { + if let Some(h) = self.map.get(&key) { + return Ok(*h); + } + let next = self.next_seq.entry(key.p_id).or_insert(0); + if *next == u32::MAX { + return Err(io::Error::other(format!( + "triple-term dictionary: predicate {} exhausted its {}-bit sequence space", + key.p_id, + fluree_db_core::triple_term::TERM_SEQ_BITS + ))); + } + let handle = term_handle(key.p_id, *next); + *next += 1; + self.map.insert(key, handle); + Ok(handle) + } + + /// Distinct terms interned so far. + pub fn len(&self) -> usize { + self.map.len() + } + + /// True when nothing has been interned. + pub fn is_empty(&self) -> bool { + self.map.is_empty() + } + + /// `(p_id, highest seq allocated)` per predicate with at least one term. + pub fn watermarks(&self) -> Vec<(u32, u32)> { + let mut wms: Vec<(u32, u32)> = self + .next_seq + .iter() + .filter(|(_, next)| **next > 0) + .map(|(p_id, next)| (*p_id, next - 1)) + .collect(); + wms.sort_unstable(); + wms + } + + /// Persist everything interned so far as a complete dictionary and + /// return its root references. + /// + /// Forward packs are written per inner predicate in sequence order, split + /// at the default pack size; the reverse tree is built from every term in + /// key order. Both are content-addressed through `cs`. + pub async fn upload(self, cs: &dyn ContentStore) -> io::Result { + let term_count = self.map.len() as u64; + let watermarks = self.watermarks(); + + // Group by predicate, ordered by seq, so each stream is contiguous from 0. + let mut by_pred: BTreeMap> = BTreeMap::new(); + let mut reverse_entries: Vec = Vec::with_capacity(self.map.len()); + for (key, handle) in &self.map { + let bytes = key.to_be_bytes(); + by_pred + .entry(term_handle_p_id(*handle)) + .or_default() + .push((term_handle_seq(*handle), bytes)); + reverse_entries.push(ReverseEntry { + key: bytes.to_vec(), + id: *handle, + }); + } + + let mut forward_packs = Vec::with_capacity(by_pred.len()); + for (p_id, mut entries) in by_pred { + entries.sort_unstable_by_key(|(seq, _)| *seq); + let kind = ContentKind::DictBlob { + dict: DictKind::TermForward { p_id }, + }; + let per_pack = (DEFAULT_TARGET_PACK_BYTES / (TermKey::LEN + 4)).max(1); + let mut packs = Vec::new(); + for chunk in entries.chunks(per_pack) { + let refs: Vec<(u64, &[u8])> = chunk + .iter() + .map(|(seq, bytes)| (*seq as u64, bytes.as_slice())) + .collect(); + let bytes = encode_forward_pack( + &refs, + KIND_TERM_FWD, + pack_ns_code(p_id), + DEFAULT_TARGET_PAGE_BYTES, + )?; + let pack_cid = put(cs, kind, &bytes).await?; + packs.push(PackBranchEntry { + first_id: refs[0].0, + last_id: refs[refs.len() - 1].0, + pack_cid, + }); + } + forward_packs.push((p_id, packs)); + } + + reverse_entries.sort_unstable_by(|a, b| a.key.cmp(&b.key)); + let reverse = upload_reverse_tree(cs, reverse_entries).await?; + + Ok(TermDictRefs { + forward_packs, + reverse, + watermarks, + term_count, + }) + } +} + +async fn put(cs: &dyn ContentStore, kind: ContentKind, bytes: &[u8]) -> io::Result { + cs.put(kind, bytes).await.map_err(io::Error::other) +} + +/// Build, upload and finalize a reverse tree from key-sorted entries. +async fn upload_reverse_tree( + cs: &dyn ContentStore, + entries: Vec, +) -> io::Result { + let kind = ContentKind::DictBlob { + dict: DictKind::TermReverse, + }; + let built = build_reverse_tree(entries, DEFAULT_TARGET_LEAF_BYTES)?; + let mut hash_to_address: HashMap = HashMap::new(); + let mut address_to_cid: HashMap = HashMap::new(); + for leaf in &built.leaves { + let cid = put(cs, kind, &leaf.bytes).await?; + let addr = cid.to_string(); + address_to_cid.insert(addr.clone(), cid); + hash_to_address.insert(leaf.hash.clone(), addr); + } + let (branch, branch_bytes, _) = finalize_branch(built.branch, &hash_to_address)?; + let branch_cid = put(cs, kind, &branch_bytes).await?; + let mut leaves = Vec::with_capacity(branch.leaves.len()); + for entry in &branch.leaves { + let cid = address_to_cid.get(&entry.address).ok_or_else(|| { + io::Error::new( + io::ErrorKind::NotFound, + format!( + "term reverse tree: no CID for leaf address {}", + entry.address + ), + ) + })?; + leaves.push(cid.clone()); + } + Ok(DictTreeRefs { + branch: branch_cid, + leaves, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use fluree_db_core::o_type::OType; + + fn key(s: u64, p: u32, o: u64) -> TermKey { + TermKey { + s_id: s, + p_id: p, + o_type: OType::IRI_REF, + o_key: o, + } + } + + #[test] + fn builder_partitions_by_predicate_and_dedups() { + let mut b = TermDictBuilder::new(); + let h1 = b.get_or_insert(key(1, 5, 2)).unwrap(); + let h2 = b.get_or_insert(key(1, 5, 3)).unwrap(); + let h3 = b.get_or_insert(key(1, 9, 2)).unwrap(); + assert_eq!(b.get_or_insert(key(1, 5, 2)).unwrap(), h1); + assert_eq!(term_handle_p_id(h1), 5); + assert_eq!(term_handle_seq(h1), 0); + assert_eq!(term_handle_seq(h2), 1); + assert_eq!((term_handle_p_id(h3), term_handle_seq(h3)), (9, 0)); + assert_eq!(b.len(), 3); + assert_eq!(b.watermarks(), vec![(5, 1), (9, 0)]); + } + + #[test] + fn builder_continues_above_watermarks() { + let mut b = TermDictBuilder::above_watermarks(&[(5, 41)]); + let h = b.get_or_insert(key(1, 5, 2)).unwrap(); + assert_eq!(term_handle_seq(h), 42); + let h0 = b.get_or_insert(key(1, 6, 2)).unwrap(); + assert_eq!(term_handle_seq(h0), 0); + } +} diff --git a/fluree-db-binary-index/src/format/expanded_cas.rs b/fluree-db-binary-index/src/format/expanded_cas.rs index 0ee8d7db2a..e64756f516 100644 --- a/fluree-db-binary-index/src/format/expanded_cas.rs +++ b/fluree-db-binary-index/src/format/expanded_cas.rs @@ -433,6 +433,7 @@ mod tests { sketch_ref: None, has_annotations: false, annotation_index: None, + term_dict: None, had_annotation_arena: false, has_list_meta: None, o_type_table: IndexRoot::build_o_type_table(&[], &[]), diff --git a/fluree-db-binary-index/src/format/index_root.rs b/fluree-db-binary-index/src/format/index_root.rs index 90b243d263..aecbbb4f47 100644 --- a/fluree-db-binary-index/src/format/index_root.rs +++ b/fluree-db-binary-index/src/format/index_root.rs @@ -19,9 +19,10 @@ use super::run_record_v2::{RunRecordV2, RECORD_V2_WIRE_SIZE}; use super::stats_wire; use super::wire_helpers::{ ensure_bytes, io_err, read_cid, read_dict_pack_refs, read_dict_tree_refs, read_i64_at, - read_string, read_string_array, read_u16_at, read_u32_at, read_u64_at, read_u8_at, write_cid, - write_dict_pack_refs, write_dict_tree_refs, write_str, write_string_array, BinaryGarbageRef, - BinaryPrevIndexRef, DictRefs, FulltextArenaRef, GraphArenaRefs, SpatialArenaRef, VectorDictRef, + read_string, read_string_array, read_term_dict_refs, read_u16_at, read_u32_at, read_u64_at, + read_u8_at, write_cid, write_dict_pack_refs, write_dict_tree_refs, write_str, + write_string_array, write_term_dict_refs, BinaryGarbageRef, BinaryPrevIndexRef, DictRefs, + FulltextArenaRef, GraphArenaRefs, SpatialArenaRef, TermDictRefs, VectorDictRef, }; use fluree_db_core::index_schema::IndexSchema; use fluree_db_core::index_stats::IndexStats; @@ -66,6 +67,8 @@ pub enum DictFamily { NumBigArena = 4, /// Spatial arena (per-predicate) SpatialArena = 5, + /// Triple-term dictionary (ledger-global, partitioned by inner predicate) + TermDict = 6, } impl DictFamily { @@ -77,6 +80,7 @@ impl DictFamily { 3 => Some(Self::VectorArena), 4 => Some(Self::NumBigArena), 5 => Some(Self::SpatialArena), + 6 => Some(Self::TermDict), _ => None, } } @@ -225,6 +229,12 @@ pub struct IndexRoot { /// can never desynchronize from a populated arena. pub annotation_index: Option, + /// Triple-term dictionary (`OType::TRIPLE_TERM` handles ↔ encoded base + /// edges). `None` until a build has interned reification links. Lives in + /// its own trailing section, flagged by `FLAG_EXT_HAS_TERM_DICT`, so its + /// lifecycle is independent of the annotation arena above. + pub term_dict: Option, + /// Sticky bit governing whether the api's /// `ApiAttachmentEventsProvider` is allowed to bootstrap an /// `Authoritative` annotation arena from a one-time base-index @@ -521,6 +531,12 @@ impl IndexRoot { None, DictFamily::SpatialArena, ), + ( + OType::TRIPLE_TERM.as_u16(), + DecodeKind::TripleTermDict, + Some(fluree::TRIPLE_TERM), + DictFamily::TermDict, + ), ]; for &(o_type, decode_kind, dt_iri, dict_family) in fluree_types { @@ -604,6 +620,9 @@ impl IndexRoot { /// Extended-flags bit: at least one indexed row carries a list index. /// Only meaningful when `FLAG_EXT_LIST_META_TRACKED` is set. const FLAG_EXT_HAS_LIST_META: u8 = 1 << 2; + /// Root carries a triple-term dictionary section after the annotation + /// section. + const FLAG_EXT_HAS_TERM_DICT: u8 = 1 << 3; /// Encode to the binary FIR6 wire format. /// @@ -696,6 +715,9 @@ impl IndexRoot { Some(false) => flags_ext |= Self::FLAG_EXT_LIST_META_TRACKED, None => {} } + if self.term_dict.is_some() { + flags_ext |= Self::FLAG_EXT_HAS_TERM_DICT; + } buf.push(flags_ext); buf.push(0); // reserved high pad byte buf.extend_from_slice(&self.index_t.to_le_bytes()); @@ -860,6 +882,11 @@ impl IndexRoot { buf.extend_from_slice(&body); } + // ---- Optional: triple-term dictionary ---- + if let Some(ref td) = self.term_dict { + write_term_dict_refs(&mut buf, td); + } + buf } @@ -1100,6 +1127,12 @@ impl IndexRoot { None }; + let term_dict = if flags_ext & Self::FLAG_EXT_HAS_TERM_DICT != 0 { + Some(read_term_dict_refs(data, &mut pos)?) + } else { + None + }; + // All optional sections consumed. `pos` should now equal the // input length — anything else means a future format added // bytes after the annotation section, or the writer emitted @@ -1156,6 +1189,7 @@ impl IndexRoot { // losing the dropped arena's history. had_annotation_arena: had_annotation_arena || annotation_index.is_some(), annotation_index, + term_dict, has_list_meta, }) } @@ -1250,6 +1284,15 @@ impl IndexRoot { ids.push(ann.reverse_branch_cid.clone()); } + // Triple-term dictionary: forward packs + reverse branch and leaves. + if let Some(ref td) = self.term_dict { + for (_, packs) in &td.forward_packs { + ids.extend(packs.iter().map(|e| e.pack_cid.clone())); + } + ids.push(td.reverse.branch.clone()); + ids.extend(td.reverse.leaves.iter().cloned()); + } + ids.sort(); ids.dedup(); ids @@ -1456,6 +1499,7 @@ mod tests { sketch_ref: None, has_annotations: false, annotation_index: None, + term_dict: None, had_annotation_arena: false, has_list_meta: None, ns_split_mode: fluree_db_core::ns_encoding::NsSplitMode::default(), diff --git a/fluree-db-binary-index/src/format/wire_helpers.rs b/fluree-db-binary-index/src/format/wire_helpers.rs index 47bbf454dc..202393ecba 100644 --- a/fluree-db-binary-index/src/format/wire_helpers.rs +++ b/fluree-db-binary-index/src/format/wire_helpers.rs @@ -122,6 +122,24 @@ pub struct DictRefs { pub string_reverse: DictTreeRefs, } +/// Triple-term dictionary references (ledger-global). +/// +/// Forward packs are grouped by inner predicate id and keyed by the +/// per-predicate sequence (the low half of a handle); the reverse tree maps +/// encoded `TermKey` bytes to the full handle. `watermarks` holds the highest +/// sequence allocated per predicate so later allocation can continue above it. +#[derive(Debug, Clone, PartialEq)] +pub struct TermDictRefs { + /// `(inner p_id, packs sorted by first_id)`, sorted by p_id. + pub forward_packs: Vec<(u32, Vec)>, + /// Reverse tree: `TermKey` bytes → handle. + pub reverse: DictTreeRefs, + /// `(inner p_id, highest seq allocated)`, sorted by p_id. + pub watermarks: Vec<(u32, u32)>, + /// Distinct terms in the dictionary. + pub term_count: u64, +} + /// Per-graph specialty arena refs (numbig, vectors, spatial). /// /// One entry per graph that has any specialty arenas. @@ -390,6 +408,90 @@ pub(crate) fn read_dict_pack_refs(data: &[u8], pos: &mut usize) -> io::Result, refs: &TermDictRefs) { + buf.push(TERM_DICT_REFS_VERSION); + let mut sorted = refs.forward_packs.clone(); + sorted.sort_by_key(|(p_id, _)| *p_id); + buf.extend_from_slice(&(sorted.len() as u32).to_le_bytes()); + for (p_id, packs) in &sorted { + buf.extend_from_slice(&p_id.to_le_bytes()); + buf.extend_from_slice(&pack_count_u16(packs.len(), "term").to_le_bytes()); + for entry in packs { + buf.extend_from_slice(&entry.first_id.to_le_bytes()); + buf.extend_from_slice(&entry.last_id.to_le_bytes()); + write_cid(buf, &entry.pack_cid); + } + } + write_dict_tree_refs(buf, &refs.reverse); + let mut wms = refs.watermarks.clone(); + wms.sort_by_key(|(p_id, _)| *p_id); + buf.extend_from_slice(&(wms.len() as u32).to_le_bytes()); + for (p_id, wm) in &wms { + buf.extend_from_slice(&p_id.to_le_bytes()); + buf.extend_from_slice(&wm.to_le_bytes()); + } + buf.extend_from_slice(&refs.term_count.to_le_bytes()); +} + +/// Read triple-term dictionary refs written by [`write_term_dict_refs`]. +pub(crate) fn read_term_dict_refs(data: &[u8], pos: &mut usize) -> io::Result { + let version = read_u8_at(data, pos)?; + if version != TERM_DICT_REFS_VERSION { + return Err(io::Error::new( + io::ErrorKind::InvalidData, + format!("term dict refs: unsupported version {version}"), + )); + } + let p_count = read_u32_at(data, pos)? as usize; + let mut forward_packs = Vec::with_capacity(p_count); + for _ in 0..p_count { + let p_id = read_u32_at(data, pos)?; + let pack_count = read_u16_at(data, pos)? as usize; + let mut packs = Vec::with_capacity(pack_count); + for _ in 0..pack_count { + let first_id = read_u64_at(data, pos)?; + let last_id = read_u64_at(data, pos)?; + let pack_cid = read_cid(data, pos)?; + packs.push(PackBranchEntry { + first_id, + last_id, + pack_cid, + }); + } + forward_packs.push((p_id, packs)); + } + let reverse = read_dict_tree_refs(data, pos)?; + let wm_count = read_u32_at(data, pos)? as usize; + let mut watermarks = Vec::with_capacity(wm_count); + for _ in 0..wm_count { + let p_id = read_u32_at(data, pos)?; + let wm = read_u32_at(data, pos)?; + watermarks.push((p_id, wm)); + } + let term_count = read_u64_at(data, pos)?; + Ok(TermDictRefs { + forward_packs, + reverse, + watermarks, + term_count, + }) +} + /// Write dict tree refs: branch CID + leaf_count:u32 + leaf CIDs. pub(crate) fn write_dict_tree_refs(buf: &mut Vec, tree: &DictTreeRefs) { write_cid(buf, &tree.branch); diff --git a/fluree-db-binary-index/src/lib.rs b/fluree-db-binary-index/src/lib.rs index 0a5fbd5b67..bc46f4560e 100644 --- a/fluree-db-binary-index/src/lib.rs +++ b/fluree-db-binary-index/src/lib.rs @@ -39,7 +39,7 @@ pub use format::index_root::IndexRoot; pub use format::run_record::{cmp_for_order, cmp_psot, cmp_spot, RunRecord, RunSortOrder}; pub use format::wire_helpers::{ BinaryGarbageRef, BinaryPrevIndexRef, DictPackRefs, DictRefs, DictTreeRefs, FulltextArenaRef, - GraphArenaRefs, PackBranchEntry, SpatialArenaRef, VectorDictRef, + GraphArenaRefs, PackBranchEntry, SpatialArenaRef, TermDictRefs, VectorDictRef, }; // ── Arena ─────────────────────────────────────────────────────────────────── diff --git a/fluree-db-binary-index/src/read/binary_index_store.rs b/fluree-db-binary-index/src/read/binary_index_store.rs index 2b9254a6ed..32699db4cc 100644 --- a/fluree-db-binary-index/src/read/binary_index_store.rs +++ b/fluree-db-binary-index/src/read/binary_index_store.rs @@ -31,6 +31,7 @@ use parking_lot::RwLock; use crate::dict::forward_pack::{KIND_STRING_FWD, KIND_SUBJECT_FWD}; use crate::dict::global_dict::{LanguageTagDict, PredicateDict}; use crate::dict::pack_reader::ForwardPackReader; +use crate::dict::term_dict::TermDictReader; use crate::dict::DictTreeReader; use crate::format::branch::{read_branch_from_bytes, BranchManifest}; use crate::format::index_root::{IndexRoot, OTypeTableEntry}; @@ -61,6 +62,8 @@ pub(crate) struct DictionarySet { /// String forward pack reader (all string IDs in one stream). pub(crate) string_forward_packs: ForwardPackReader, pub(crate) string_reverse_tree: Option>, + /// Triple-term dictionary; `None` when the root carries no term section. + pub(crate) term_dict: Option, // Kept for: DictOverlay watermark computation (query overlay resolution). // Use when: DictOverlay is wired into V3 query execution for overlay transactions. #[expect(dead_code)] @@ -367,6 +370,7 @@ impl BinaryIndexStore { subject_reverse_tree: None, string_forward_packs: ForwardPackReader::empty(), string_reverse_tree: None, + term_dict: None, subject_count: 0, string_count: 0, namespace_codes: Arc::new(HashMap::new()), @@ -1517,9 +1521,7 @@ impl BinaryIndexStore { DecodeKind::SpatialArena => Err(io::Error::other( "spatial arena decode not yet implemented in V6", )), - DecodeKind::TripleTermDict => Err(io::Error::other( - "triple-term decode needs the term dictionary (not yet wired)", - )), + DecodeKind::TripleTermDict => self.decode_triple_term(o_key), } } @@ -2238,6 +2240,87 @@ impl BinaryIndexStore { Some(proven) } + /// True when the index carries a triple-term dictionary. + pub fn has_term_dict(&self) -> bool { + self.dicts.term_dict.is_some() + } + + /// The triple-term dictionary, if the index carries one. + pub fn term_dict(&self) -> Option<&TermDictReader> { + self.dicts.term_dict.as_ref() + } + + /// The `OType::TRIPLE_TERM` handle for an encoded base edge, if interned. + pub fn find_term_handle( + &self, + key: &fluree_db_core::triple_term::TermKey, + ) -> io::Result> { + match &self.dicts.term_dict { + Some(td) => td.find_handle(key), + None => Ok(None), + } + } + + /// The encoded base edge behind a triple-term handle. + pub fn resolve_term_key( + &self, + handle: u64, + ) -> io::Result> { + match &self.dicts.term_dict { + Some(td) => td.resolve(handle), + None => Ok(None), + } + } + + /// Materialize a triple-term handle: the base edge's subject, predicate + /// and object as SIDs and a value, with the object's datatype and tag. + /// + /// Terms are graph-independent, so a per-graph object arena (NumBig) is + /// read through the default graph. + fn decode_triple_term(&self, handle: u64) -> io::Result { + let key = self.resolve_term_key(handle)?.ok_or_else(|| { + io::Error::new( + io::ErrorKind::NotFound, + format!("triple-term handle {handle:#x} not in the term dictionary"), + ) + })?; + let (ns, suffix) = self.resolve_subject_parts(key.s_id)?; + let p = self.predicate_sid(key.p_id).ok_or_else(|| { + io::Error::new( + io::ErrorKind::NotFound, + format!( + "triple-term handle {handle:#x}: unknown predicate id {}", + key.p_id + ), + ) + })?; + let o_type = key.o_type.as_u16(); + let o = self.decode_value_v3( + o_type, + key.o_key, + key.p_id, + fluree_db_core::DEFAULT_GRAPH_ID, + )?; + let dt = self + .resolve_datatype_sid_for_value(o_type, &o) + .ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + format!("triple-term handle {handle:#x}: no datatype for o_type {o_type:#06x}"), + ) + })?; + let lang = self.resolve_lang_tag(o_type).map(str::to_string); + Ok(FlakeValue::TripleTerm(Box::new( + fluree_db_core::TripleTermValue { + s: Sid::new(ns, suffix), + p, + o, + dt, + lang, + }, + ))) + } + pub fn find_subject_id_by_parts(&self, ns_code: u16, suffix: &str) -> io::Result> { match &self.dicts.subject_reverse_tree { Some(tree) => { @@ -3223,6 +3306,21 @@ async fn build_dictionary_set( let string_reverse_us = phase.elapsed().as_micros() as u64; let phase = Instant::now(); + // Triple-term dictionary (optional section). + let term_dict = match &root.term_dict { + Some(refs) => Some( + TermDictReader::from_refs_reusing( + Arc::clone(&cs), + cache_dir, + refs, + leaflet_cache, + prev.and_then(|p| p.term_dict.as_ref()), + ) + .await?, + ), + None => None, + }; + // Namespace codes: shared with the previous store when it already holds // every entry of the root's table. Codes are never reassigned within a // ledger, and the previous store's extras (codes the snapshot augmented @@ -3325,6 +3423,7 @@ async fn build_dictionary_set( subject_reverse_tree, string_forward_packs, string_reverse_tree, + term_dict, subject_count, string_count: root.string_watermark, namespace_codes, diff --git a/fluree-db-core/src/content_kind.rs b/fluree-db-core/src/content_kind.rs index 65493aabfb..b371ae0f3d 100644 --- a/fluree-db-core/src/content_kind.rs +++ b/fluree-db-core/src/content_kind.rs @@ -109,6 +109,11 @@ pub enum DictKind { VectorShard { p_id: u32 }, /// Per-predicate vector arena manifest (VAM1 JSON format). VectorManifest { p_id: u32 }, + /// Triple-term forward pack for one inner predicate (FPK1, kind 2): + /// per-predicate sequence → encoded `TermKey`. + TermForward { p_id: u32 }, + /// Triple-term reverse tree: encoded `TermKey` → handle (DTB1/DLR1). + TermReverse, } // ============================================================================ diff --git a/fluree-db-core/src/db.rs b/fluree-db-core/src/db.rs index 49161651bf..348aba8dea 100644 --- a/fluree-db-core/src/db.rs +++ b/fluree-db-core/src/db.rs @@ -722,7 +722,9 @@ fn decode_fir6_metadata(bytes: &[u8]) -> std::io::Result const FLAG_EXT_HAD_ANNOTATION_ARENA: u8 = 1 << 0; const FLAG_EXT_LIST_META_TRACKED: u8 = 1 << 1; const FLAG_EXT_HAS_LIST_META: u8 = 1 << 2; + const FLAG_EXT_HAS_TERM_DICT: u8 = 1 << 3; let flags_ext = bytes[6]; + let has_term_dict_section = flags_ext & FLAG_EXT_HAS_TERM_DICT != 0; let had_annotation_arena = flags_ext & FLAG_EXT_HAD_ANNOTATION_ARENA != 0; let has_list_meta = if flags_ext & FLAG_EXT_LIST_META_TRACKED == 0 { None @@ -841,6 +843,37 @@ fn decode_fir6_metadata(bytes: &[u8]) -> std::io::Result Ok(()) } + /// Skip the triple-term dictionary section. Matches + /// `write_term_dict_refs` in binary-index: version, per-predicate forward + /// packs, reverse tree refs, per-predicate watermarks, term count. + fn skip_term_dict_refs(bytes: &[u8], pos: &mut usize) -> std::io::Result<()> { + let version = read_u8(bytes, pos)?; + if version != 1 { + return Err(std::io::Error::new( + std::io::ErrorKind::InvalidData, + format!("FIR6: unsupported term dict section version {version}"), + )); + } + let p_count = read_u32(bytes, pos)? as usize; + for _ in 0..p_count { + let _p_id = read_u32(bytes, pos)?; + let pack_count = read_u16(bytes, pos)? as usize; + for _ in 0..pack_count { + let _first = read_u64(bytes, pos)?; + let _last = read_u64(bytes, pos)?; + skip_cid(bytes, pos)?; + } + } + skip_dict_tree_refs(bytes, pos)?; + let wm_count = read_u32(bytes, pos)? as usize; + for _ in 0..wm_count { + let _p_id = read_u32(bytes, pos)?; + let _wm = read_u32(bytes, pos)?; + } + let _term_count = read_u64(bytes, pos)?; + Ok(()) + } + fn skip_graph_arenas(bytes: &[u8], pos: &mut usize, version: u8) -> std::io::Result<()> { // Matches `write_graph_arenas_v5` in binary-index. let _g_id = read_u16(bytes, pos)?; @@ -1053,6 +1086,12 @@ fn decode_fir6_metadata(bytes: &[u8]) -> std::io::Result None }; + // Optional triple-term dictionary section: the metadata view has no use + // for it, so it is skipped structurally. + if has_term_dict_section { + skip_term_dict_refs(bytes, &mut pos)?; + } + // Trailing-byte sentinel: any unread bytes after the annotation // section indicate a future format extension or a writer bug. // Surface that explicitly rather than silently drop the bytes. diff --git a/fluree-db-indexer/src/build/rebuild.rs b/fluree-db-indexer/src/build/rebuild.rs index eb7b0c15b5..dbd2542199 100644 --- a/fluree-db-indexer/src/build/rebuild.rs +++ b/fluree-db-indexer/src/build/rebuild.rs @@ -617,6 +617,7 @@ where string_count: string_dicts[ci].len() as u64, types_map_path: Some(types_path), duplicates_removed: 0, + term_table: None, }); } @@ -797,6 +798,7 @@ where remap_progress: None, build_progress: None, stage_marker: None, + term_builder: None, }; let v3_result = crate::build_indexes_from_remapped_commits( diff --git a/fluree-db-indexer/src/build/root_assembly.rs b/fluree-db-indexer/src/build/root_assembly.rs index f7cfb67579..c51a992b3a 100644 --- a/fluree-db-indexer/src/build/root_assembly.rs +++ b/fluree-db-indexer/src/build/root_assembly.rs @@ -332,6 +332,9 @@ pub(crate) async fn encode_and_write_root_v6( sketch_ref: inputs.sketch_ref, has_annotations, annotation_index: None, + // Full rebuilds do not intern reification links yet; the term + // dictionary is produced by bulk import only. + term_dict: None, // Sticky bit flipped to `true` below if the rebuild path // seals an `Authoritative` arena. Rebuilds always start // from scratch with no prior root, so this is the only @@ -636,6 +639,7 @@ mod tests { had_annotation_arena: false, has_list_meta: None, annotation_index: None, + term_dict: None, o_type_table: IndexRoot::build_o_type_table(&[], &[]), ns_split_mode: fluree_db_core::ns_encoding::NsSplitMode::default(), } diff --git a/fluree-db-indexer/src/drop.rs b/fluree-db-indexer/src/drop.rs index 69ef499c62..e3f3adf0c6 100644 --- a/fluree-db-indexer/src/drop.rs +++ b/fluree-db-indexer/src/drop.rs @@ -245,6 +245,7 @@ mod tests { had_annotation_arena: annotation_index.is_some(), has_list_meta: None, annotation_index, + term_dict: None, o_type_table: IndexRoot::build_o_type_table(&[], &[]), ns_split_mode: fluree_db_core::ns_encoding::NsSplitMode::default(), } diff --git a/fluree-db-indexer/src/gc/test_support.rs b/fluree-db-indexer/src/gc/test_support.rs index 9ee91b574e..62eee1a0eb 100644 --- a/fluree-db-indexer/src/gc/test_support.rs +++ b/fluree-db-indexer/src/gc/test_support.rs @@ -85,6 +85,7 @@ pub(crate) fn fir6_with_named_graph_for( sketch_ref: None, has_annotations: false, annotation_index: None, + term_dict: None, had_annotation_arena: false, has_list_meta: None, o_type_table: IndexRoot::build_o_type_table(&[], &[]), diff --git a/fluree-db-indexer/src/run_index/build/build_from_commits.rs b/fluree-db-indexer/src/run_index/build/build_from_commits.rs index a0eb41a9b0..4d5b173a8d 100644 --- a/fluree-db-indexer/src/run_index/build/build_from_commits.rs +++ b/fluree-db-indexer/src/run_index/build/build_from_commits.rs @@ -21,7 +21,7 @@ use crate::run_index::runs::run_writer::{ }; use crate::run_index::runs::spool::{ link_chunk_run_files_to_flat, remap_commit_to_runs_with_op, remap_sorted_commit_v2_to_runs, - MmapStringRemap, MmapSubjectRemap, SortedCommitMergeReaderV2, SubjectRemap, + MmapStringRemap, MmapSubjectRemap, SortedCommitMergeReaderV2, SubjectRemap, TermRemapCtx, }; use crate::run_index::runs::streaming_reader::{MergeSource, StreamingRunReader}; use crate::stats::{stats_record_from_v2, SpotClassStats, DT_REF_ID}; @@ -105,6 +105,9 @@ pub struct CommitInput { pub lang_remap: Vec, /// Optional rdf:type sidecar used to rebuild the subject→class bitset table. pub types_map_path: Option, + /// Triple-term table `(path, record_count)` for the chunk's reification + /// links, if it has any (see `SortedCommitInfo::term_table`). + pub term_table: Option<(PathBuf, u64)>, } /// Configuration for the V3 build-from-commits pipeline. @@ -137,6 +140,11 @@ pub struct BuildConfig { pub build_progress: Option>, /// Shared stage marker for external progress reporting. pub stage_marker: Option>, + /// Build-wide triple-term interner. Chunks with a term table resolve + /// their link records through it; `None` disables link interning (a + /// term-carrying chunk then fails the build rather than emitting + /// dangling handles). + pub term_builder: Option>>, } /// Result of the V3 build pipeline. @@ -1162,6 +1170,7 @@ pub fn build_indexes_from_commits( let mut handles = Vec::with_capacity(worker_count); for _ in 0..worker_count { let commits_ref = commits; + let config_ref = config; let next_chunk = Arc::clone(&next_chunk); let run_dir = config.run_dir.clone(); let remap_progress = config.remap_progress.clone(); @@ -1192,6 +1201,7 @@ pub fn build_indexes_from_commits( let s_remap = MmapSubjectRemap::open(&commit.subject_remap_path)?; let str_remap = MmapStringRemap::open(&commit.string_remap_path)?; + let term_ctx = term_ctx_for(commit, config_ref)?; let mut writer = MultiOrderRunWriter::new(MultiOrderConfig { total_budget_bytes: per_thread_budget_bytes, orders: RunSortOrder::secondary_orders().to_vec(), @@ -1205,6 +1215,7 @@ pub fn build_indexes_from_commits( &str_remap, &commit.lang_remap, target_g_id, + term_ctx.as_ref(), &mut writer, worker_hook.as_mut(), remap_progress.as_deref(), @@ -1412,7 +1423,7 @@ fn build_spot_index_from_commits( let total_rows = if commits.len() <= fd_plan.spot_fan_in { // Flat merge: one long-lived reader per chunk, all open at once. // The common case, and byte-for-byte the pre-budget behavior. - let streams = open_spot_commit_readers(commits, g_id)?; + let streams = open_spot_commit_readers(commits, g_id, config)?; let mut merge = KWayMerge::new(streams, cmp_v2_spot)?; pump_spot_merge( &mut merge, @@ -1444,7 +1455,7 @@ fn build_spot_index_from_commits( let mut intermediates = Vec::with_capacity(commits.len().div_ceil(fd_plan.spot_fan_in)); for (i, group) in commits.chunks(fd_plan.spot_fan_in).enumerate() { - let streams = open_spot_commit_readers(group, g_id)?; + let streams = open_spot_commit_readers(group, g_id, config)?; let mut merge = KWayMerge::new(streams, cmp_v2_spot)?; let out_path = pass_dir.join(format!("merged_{i:06}.frn")); // Carry op bytes through verbatim (with-op intermediates), so @@ -1529,12 +1540,14 @@ fn index_build_err_to_io(e: IndexBuildError) -> io::Error { fn open_spot_commit_readers( commits: &[CommitInput], g_id: u16, + config: &BuildConfig, ) -> io::Result>> { commits .iter() .map(|commit| { let s_remap = MmapSubjectRemap::open(&commit.subject_remap_path)?; let str_remap = MmapStringRemap::open(&commit.string_remap_path)?; + let term_ctx = term_ctx_for(commit, config)?.map(Arc::new); SortedCommitMergeReaderV2::open( &commit.commit_path, commit.record_count, @@ -1542,11 +1555,27 @@ fn open_spot_commit_readers( str_remap, commit.lang_remap.clone(), g_id, + term_ctx, ) }) .collect() } +/// The term-remap context for a chunk, loaded when the chunk has a term +/// table and the build interns terms. +fn term_ctx_for(commit: &CommitInput, config: &BuildConfig) -> io::Result> { + match (&commit.term_table, &config.term_builder) { + (Some((path, count)), Some(builder)) => { + Ok(Some(TermRemapCtx::load(path, *count, Arc::clone(builder))?)) + } + (Some(_), None) => Err(io::Error::new( + io::ErrorKind::InvalidInput, + "chunk carries a triple-term table but the build has no term interner", + )), + (None, _) => Ok(None), + } +} + /// Drain a SPOT merge into the leaf writer (and class-stats collector), /// batching progress updates. Shared by the flat and hierarchical SPOT /// arms so both consume the merged sequence identically. @@ -1819,6 +1848,7 @@ mod tests { string_remap_path: str_remap_path, lang_remap: vec![], types_map_path: None, + term_table: None, }]; let config = BuildConfig { @@ -1834,6 +1864,7 @@ mod tests { remap_progress: None, build_progress: None, stage_marker: None, + term_builder: None, }; let (result, _spot_class_stats) = @@ -1938,6 +1969,7 @@ mod tests { string_remap_path: str_path, lang_remap: vec![], types_map_path: None, + term_table: None, }); } @@ -1955,6 +1987,7 @@ mod tests { remap_progress: None, build_progress: None, stage_marker: None, + term_builder: None, }; std::fs::create_dir_all(&config.run_dir).unwrap(); build_spot_index_from_commits(&commits, &config, None, None, plan) @@ -2063,6 +2096,7 @@ mod tests { string_remap_path: str_path, lang_remap: vec![], types_map_path: None, + term_table: None, }); } @@ -2090,6 +2124,7 @@ mod tests { remap_progress: None, build_progress: None, stage_marker: None, + term_builder: None, }; std::fs::create_dir_all(&config.run_dir).unwrap(); build_spot_index_from_commits(&commits, &config, Some(TYPE_P), Some(membership), plan) @@ -2170,6 +2205,7 @@ mod tests { string_remap_path: str0, lang_remap: vec![], types_map_path: Some(types0), + term_table: None, }, CommitInput { commit_path: dir.path().join("unused1.fsv2"), @@ -2178,6 +2214,7 @@ mod tests { string_remap_path: str1, lang_remap: vec![], types_map_path: Some(types1), + term_table: None, }, ]; @@ -2241,6 +2278,7 @@ mod tests { string_remap_path: str0, lang_remap: vec![], types_map_path: Some(types0), + term_table: None, }, CommitInput { commit_path: dir.path().join("unused1.fsv2"), @@ -2249,6 +2287,7 @@ mod tests { string_remap_path: str1, lang_remap: vec![], types_map_path: Some(types1), + term_table: None, }, ]; diff --git a/fluree-db-indexer/src/run_index/build/incremental_root.rs b/fluree-db-indexer/src/run_index/build/incremental_root.rs index 92d778c721..2990df5b16 100644 --- a/fluree-db-indexer/src/run_index/build/incremental_root.rs +++ b/fluree-db-indexer/src/run_index/build/incremental_root.rs @@ -392,6 +392,7 @@ mod tests { sketch_ref: None, has_annotations: false, annotation_index: None, + term_dict: None, had_annotation_arena: false, has_list_meta: None, o_type_table: IndexRoot::build_o_type_table(&[], &[]), diff --git a/fluree-db-indexer/src/run_index/runs/spool.rs b/fluree-db-indexer/src/run_index/runs/spool.rs index 1df24ba4c9..c4fa39b7a5 100644 --- a/fluree-db-indexer/src/run_index/runs/spool.rs +++ b/fluree-db-indexer/src/run_index/runs/spool.rs @@ -35,6 +35,7 @@ use fluree_db_binary_index::format::run_record_v2::RunRecordV2; use fluree_db_core::o_type_registry::OTypeRegistry; use std::io::{self, BufWriter, Read, Seek, SeekFrom, Write}; use std::path::{Path, PathBuf}; +use std::sync::Arc; // ============================================================================ // Remap table abstractions (memory-friendly) @@ -982,6 +983,7 @@ pub struct SortedCommitMergeReaderV2 { string_remap: R, lang_remap: Vec, target_g_id: u16, + term_ctx: Option>, current: Option, } @@ -993,6 +995,7 @@ impl SortedCommitMergeReaderV2 { string_remap: R, lang_remap: Vec, target_g_id: u16, + term_ctx: Option>, ) -> io::Result { let mut reader = Self { reader: SortedCommitReaderV2::open(path, record_count)?, @@ -1000,6 +1003,7 @@ impl SortedCommitMergeReaderV2 { string_remap, lang_remap, target_g_id, + term_ctx, current: None, }; reader.advance_to_next()?; @@ -1022,6 +1026,7 @@ impl SortedCommitMergeReaderV2 { &self.subject_remap, &self.string_remap, lang_remap, + self.term_ctx.as_deref(), )?; self.current = Some(record); return Ok(()); @@ -1234,12 +1239,77 @@ fn cmp_run_record_as_v2_g_spot( .then(a.i.cmp(&b.i)) } +/// Per-chunk context that turns a link record's term ordinal into a global +/// triple-term handle. +/// +/// The chunk's term table holds one V2 pseudo-record per reified base edge in +/// sorted-local ids. Resolving an ordinal remaps that entry with the same +/// tables the chunk's records use, then interns the resulting `TermKey` in +/// the build-wide dictionary. Interning is idempotent, so every phase that +/// reads the chunk resolves the same ordinal to the same handle. +pub struct TermRemapCtx { + terms: Vec, + builder: Arc>, +} + +impl TermRemapCtx { + /// Load a chunk's term table. + pub fn load( + path: &Path, + record_count: u64, + builder: Arc>, + ) -> io::Result { + let mut terms = Vec::with_capacity(record_count as usize); + for r in SortedCommitReaderV2::open(path, record_count)? { + terms.push(r?); + } + Ok(Self { terms, builder }) + } + + /// The global handle for the term at `ordinal`. + pub fn handle_for( + &self, + ordinal: u64, + subject_remap: &S, + string_remap: &R, + lang_remap: Option<&[u16]>, + ) -> io::Result { + let mut term = *self.terms.get(ordinal as usize).ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + format!( + "term ordinal {ordinal} out of range (table holds {})", + self.terms.len() + ), + ) + })?; + remap_v2_record( + &mut term, + subject_remap, + string_remap, + lang_remap, + Some(self), + )?; + let key = fluree_db_core::triple_term::TermKey { + s_id: term.s_id.as_u64(), + p_id: term.p_id, + o_type: fluree_db_core::o_type::OType::from_u16(term.o_type), + o_key: term.o_key, + }; + self.builder + .lock() + .map_err(|_| io::Error::other("term dictionary builder lock poisoned"))? + .get_or_insert(key) + } +} + #[inline] pub fn remap_v2_record( record: &mut RunRecordV2, subject_remap: &S, string_remap: &R, lang_remap: Option<&[u16]>, + term_ctx: Option<&TermRemapCtx>, ) -> io::Result<()> { use fluree_db_core::o_type::{DecodeKind, OType}; use fluree_db_core::subject_id::SubjectId; @@ -1255,6 +1325,15 @@ pub fn remap_v2_record( DecodeKind::StringDict => { record.o_key = string_remap.get(record.o_key as usize)? as u64; } + DecodeKind::TripleTermDict => { + let ctx = term_ctx.ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + "triple-term record in a chunk without a term table", + ) + })?; + record.o_key = ctx.handle_for(record.o_key, subject_remap, string_remap, lang_remap)?; + } _ => {} } @@ -1455,6 +1534,7 @@ pub fn remap_sorted_commit_v2_to_runs, writer: &mut super::run_writer::MultiOrderRunWriter, mut stats_hook: Option<&mut crate::stats::IdStatsHook>, progress: Option<&std::sync::atomic::AtomicU64>, @@ -1475,7 +1555,13 @@ pub fn remap_sorted_commit_v2_to_runs, } /// Sort, remap, and write a sorted commit file from buffered parse output. @@ -1632,6 +1722,7 @@ pub const TYPES_MAP_ENTRY_SIZE: usize = 18; #[allow(clippy::too_many_arguments)] pub fn sort_remap_and_write_sorted_commit( mut records: Vec, + mut terms: Vec, subjects: crate::run_index::resolve::chunk_dict::ChunkSubjectDict, strings: crate::run_index::resolve::chunk_dict::ChunkStringDict, subject_vocab_path: &Path, @@ -1685,7 +1776,7 @@ pub fn sort_remap_and_write_sorted_commit( // Raw tags that normalize identically collapse to one dict id. old_to_lex[*old_id as usize] = lang_dict.get_or_insert(Some(tag)); } - for record in &mut records { + for record in records.iter_mut().chain(terms.iter_mut()) { if record.lang_id != 0 { record.lang_id = old_to_lex[record.lang_id as usize]; } @@ -1723,6 +1814,11 @@ pub fn sort_remap_and_write_sorted_commit( remap_record(record, &subject_remap, &string_remap)?; } } + // Term-table entries keep their ordinal order (link records address them + // by position), so they are remapped but never sorted. + for term in &mut terms { + remap_record(term, &subject_remap, &string_remap)?; + } // A.2 step 4: Sort records by the V2-native graph-prefixed SPOT key without // materializing a second full-size record buffer. @@ -1757,6 +1853,18 @@ pub fn sort_remap_and_write_sorted_commit( } let spool_info = writer.finish()?; + let term_table = if terms.is_empty() { + None + } else { + let path = commit_path.with_extension("terms"); + let mut tw = SortedCommitWriterV2::new(&path, chunk_idx)?; + for term in &terms { + tw.push(&RunRecordV2::from_v1(term, otype_registry))?; + } + let info = tw.finish()?; + Some((info.path, info.record_count)) + }; + // A.3b: Optionally extract rdf:type edges into a tiny sidecar file. // Records are already remapped to sorted-local IDs (step A.2), so both // s_id and o_key (class) are sorted-position IDs that can be remapped to @@ -1791,6 +1899,7 @@ pub fn sort_remap_and_write_sorted_commit( string_count, types_map_path, duplicates_removed, + term_table, }) } @@ -2362,6 +2471,7 @@ mod tests { let info = sort_remap_and_write_sorted_commit( records, + Vec::new(), subj_dict, str_dict, &subj_vocab, @@ -2432,6 +2542,7 @@ mod tests { let info = sort_remap_and_write_sorted_commit( records, + Vec::new(), subj_dict, str_dict, &dir.join("subjects.voc"), @@ -2492,6 +2603,7 @@ mod tests { let lang_vocab = dir.join("languages.voc"); let info = sort_remap_and_write_sorted_commit( records, + Vec::new(), subj_dict, str_dict, &dir.join("subjects.voc"), @@ -2556,6 +2668,7 @@ mod tests { let info = sort_remap_and_write_sorted_commit( records, + Vec::new(), subj_dict, str_dict, &subj_vocab, @@ -2612,6 +2725,7 @@ mod tests { let info = sort_remap_and_write_sorted_commit( records, + Vec::new(), subj_dict, str_dict, &dir.join("subj.voc"), @@ -2645,8 +2759,9 @@ mod tests { info.record_count, &subject_remap, &string_remap, - &[], // no lang remap - 0, // target g_id (default graph) + &[], // no lang remap + 0, // target g_id (default graph) + None, // no term table &mut writer, Some(&mut stats_hook), None, diff --git a/fluree-db-transact/src/import_sink.rs b/fluree-db-transact/src/import_sink.rs index e2ad615495..ea510d5862 100644 --- a/fluree-db-transact/src/import_sink.rs +++ b/fluree-db-transact/src/import_sink.rs @@ -117,6 +117,9 @@ mod inner { pub subjects: ChunkSubjectDict, /// Chunk-local string dictionary (chunk-local ID → string bytes). pub strings: ChunkStringDict, + /// Triple-term table: one pseudo-record per reified base edge, in + /// ordinal order (see [`SpoolContext::write_link_record`]). + pub terms: Vec, } /// Result of finishing a [`SpoolContext`] via [`SpoolContext::finish_buffered`] — @@ -133,6 +136,9 @@ mod inner { pub languages: rustc_hash::FxHashMap, /// Chunk index (for deterministic ordering in merge phase). pub chunk_idx: usize, + /// Triple-term table: one pseudo-record per reified base edge, in + /// ordinal order (see [`SpoolContext::write_link_record`]). + pub terms: Vec, } /// Per-chunk context for writing spool records during parse (Phase B). @@ -149,6 +155,13 @@ mod inner { pub struct SpoolContext { /// Buffered records (insertion-order, chunk-local IDs). records: Vec, + /// Reified base edges as pseudo-records, indexed by the ordinal a + /// link record's `o_key` carries until the build interns them. + terms: Vec, + /// Global predicate id of `rdf:reifies`, assigned on first link. + rdf_reifies_pid: Option, + /// Global datatype id of `f:tripleTerm`, assigned on first link. + triple_term_dt: Option, /// Path for writing spool file (used by `finish()` backward-compat path). spool_path: std::path::PathBuf, chunk_idx: usize, @@ -193,6 +206,9 @@ mod inner { ) -> std::io::Result { Ok(Self { records: Vec::new(), + terms: Vec::new(), + rdf_reifies_pid: None, + triple_term_dt: None, spool_path: spool_path.into(), chunk_idx, subjects: ChunkSubjectDict::new(), @@ -233,6 +249,7 @@ mod inner { spool_info, subjects: self.subjects, strings: self.strings, + terms: self.terms, }) } @@ -249,6 +266,7 @@ mod inner { strings: self.strings, languages: self.languages, chunk_idx: self.chunk_idx, + terms: self.terms, } } @@ -555,6 +573,84 @@ mod inner { Ok(()) } + /// Spool the RDF 1.2 link `ann rdf:reifies <<( s p o )>>` for a reified + /// base edge. + /// + /// The base edge becomes a pseudo-record in the chunk's term table, + /// resolved exactly as the base triple's own record was (the object + /// under the inner predicate, so per-predicate arena handles agree). + /// The link record's `o_key` is that entry's ordinal; the build remaps + /// the entry to global ids and interns it, replacing the ordinal with + /// the term handle. + /// + /// `object` is the bundle's `f:reifiesObject` flake: its subject is + /// the reifier and its object, datatype and tag are the base edge's. + fn write_link_record( + &mut self, + s: &Sid, + p: &Sid, + object: &Flake, + t: i64, + ) -> Result<(), CommitCodecError> { + let ann = &object.s; + let s_id = self.assign_subject_id(s); + let p_id = self.assign_predicate_id(p); + let dt_id = self.assign_datatype_id(&object.dt)?; + let Some((o_kind, o_key)) = self.resolve_object_value(&object.o, p_id) else { + return Ok(()); + }; + let lang_id = object + .m + .as_ref() + .and_then(|m| m.lang.as_deref()) + .map(|l| self.assign_lang_id(l)) + .unwrap_or(0); + let ordinal = self.terms.len() as u64; + self.terms.push(RunRecord { + g_id: self.g_id, + s_id: SubjectId::from_u64(s_id), + p_id, + dt: dt_id, + o_kind, + op: 1, + o_key, + t: t as u32, + lang_id, + i: LIST_INDEX_NONE, + }); + + let link_p = match self.rdf_reifies_pid { + Some(p) => p, + None => { + let p = self.assign_predicate_id(fluree_db_core::rdf_reifies_sid()); + self.rdf_reifies_pid = Some(p); + p + } + }; + let link_dt = match self.triple_term_dt { + Some(d) => d, + None => { + let d = self.assign_datatype_id(fluree_db_core::triple_term_datatype_sid())?; + self.triple_term_dt = Some(d); + d + } + }; + let ann_id = self.assign_subject_id(ann); + self.records.push(RunRecord { + g_id: self.g_id, + s_id: SubjectId::from_u64(ann_id), + p_id: link_p, + dt: link_dt, + o_kind: ObjKind::TRIPLE_TERM.as_u8(), + op: 1, + o_key: ordinal, + t: t as u32, + lang_id: 0, + i: LIST_INDEX_NONE, + }); + Ok(()) + } + /// Allocate (or look up) the `g_id` for a named-graph IRI via the shared /// graph allocator. /// @@ -989,8 +1085,8 @@ mod inner { return Ok(()); } }; - for flake in bundle { - if let Err(e) = self.writer.push_flake(&flake) { + for flake in &bundle { + if let Err(e) = self.writer.push_flake(flake) { if self.encode_error.is_none() { tracing::error!("ImportSink: reifier bundle flake encode failed: {}", e); self.encode_error = Some(e); @@ -1012,6 +1108,33 @@ mod inner { } } } + // The RDF 1.2 link form rides alongside the bundle: the bundle's + // object flake carries the base edge's value, datatype and tag. + if let Some(ctx) = &mut self.spool_ctx { + let obj = bundle + .iter() + .find(|f| fluree_db_core::is_reifies_object(&f.p)); + let subj = bundle + .iter() + .find(|f| fluree_db_core::is_reifies_subject(&f.p)) + .and_then(|f| match &f.o { + FlakeValue::Ref(sid) => Some(sid), + _ => None, + }); + let pred = bundle + .iter() + .find(|f| fluree_db_core::is_reifies_predicate(&f.p)) + .and_then(|f| match &f.o { + FlakeValue::Ref(sid) => Some(sid), + _ => None, + }); + if let (Some(obj), Some(subj), Some(pred)) = (obj, subj, pred) { + let written = ctx.write_link_record(subj, pred, obj, self.t); + if let Err(e) = written { + self.encode_error.get_or_insert(e); + } + } + } Ok(()) } } From 3d8a85d0d1dc300a4379f94256ea0504c4c43dd2 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 12:41:46 -0400 Subject: [PATCH 04/92] feat(index): synthesize triple-term links on rebuild and incremental builds The resolver now assembles each commit's `f:reifies*` bundle into the RDF 1.2 link record and a per-chunk term entry (`link_synth`), so an index built from commits carries the same `rdf:reifies` flakes bulk import writes. This is what makes the CLI's post-import seal, and every later reindex, keep the links and the dictionary. Full rebuilds intern in Phase C once ids are global and publish the dictionary from the root. Incremental builds look each term up in the base dictionary, allocate fresh handles above its per-predicate watermarks, append per-predicate forward packs and update the reverse tree copy-on-write, with the replaced tree blobs recorded as garbage. Pack appends take the pack kind explicitly, since the subject appender hardcoded it. The stage cascade drops the index-side link before partitioning a reifier's flakes into bundle and body, so an indexed reifier whose body is deleted still looks orphaned and cascades; the link itself is retired by the next index pass. Tests cover import, reindex and the incremental window, including a second reifier on an existing edge. --- fluree-db-api/tests/it_triple_term_links.rs | 135 ++++++++++- .../src/dict/incremental.rs | 39 +++- .../src/dict/pack_builder.rs | 16 ++ fluree-db-binary-index/src/dict/term_dict.rs | 10 +- fluree-db-indexer/src/build/dicts.rs | 29 +++ fluree-db-indexer/src/build/incremental.rs | 107 +++++++++ fluree-db-indexer/src/build/rebuild.rs | 94 ++++++++ fluree-db-indexer/src/build/root_assembly.rs | 7 +- fluree-db-indexer/src/gc/collector.rs | 1 + .../run_index/build/incremental_resolve.rs | 105 +++++++++ .../src/run_index/build/incremental_root.rs | 12 + .../src/run_index/resolve/link_synth.rs | 213 ++++++++++++++++++ .../src/run_index/resolve/mod.rs | 1 + .../src/run_index/resolve/resolver.rs | 21 ++ fluree-db-transact/src/stage.rs | 9 +- 15 files changed, 790 insertions(+), 9 deletions(-) create mode 100644 fluree-db-indexer/src/run_index/resolve/link_synth.rs diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index d3833eadb4..09db8d7336 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -11,7 +11,7 @@ use crate::support; use fluree_db_api::{FlureeBuilder, LedgerState}; -use serde_json::Value as JsonValue; +use serde_json::{json, Value as JsonValue}; const CLAIMS: &str = r#"VERSION "1.2" @prefix ex: . @@ -121,3 +121,136 @@ async fn imported_reifiers_carry_a_decodable_triple_term_link() { "the claim body stays visible: {preds:?}" ); } + +/// A full rebuild from commits re-synthesizes the links and re-interns the +/// dictionary: the commits carry only the `f:reifies*` bundles, so this is +/// the resolver-side assembler, not the import-side term table, at work. +#[tokio::test] +async fn reindex_rebuilds_the_links_and_the_dictionary() { + let alias = "it/triple-term-links:reindex"; + let (fluree, _ledger) = import(&[("claims.ttl", CLAIMS)], alias).await; + fluree + .reindex(alias, fluree_db_api::ReindexOptions::default()) + .await + .expect("reindex of an imported annotated ledger"); + let ledger = fluree.ledger(alias).await.expect("reload after reindex"); + + let sparql = "PREFIX ex: \n\ + PREFIX rdf: \n\ + SELECT ?r ?t WHERE { ?r rdf:reifies ?t } ORDER BY ?r"; + let result = support::query_sparql_formatted(&fluree, &ledger, sparql) + .await + .expect("link query after reindex"); + let got = rows(&result); + assert_eq!(got.len(), 4, "{got:#?}"); + let t1 = &got + .iter() + .find(|r| r[0].ends_with("claim1")) + .unwrap_or_else(|| panic!("no link for claim1 after reindex: {got:#?}"))[1]; + assert!( + t1.contains("alice") && t1.contains("knows") && t1.contains("bob"), + "claim1 must still reify alice knows bob: {t1}" + ); +} + +async fn links(fluree: &fluree_db_api::Fluree, ledger: &LedgerState) -> Vec> { + let sparql = "PREFIX rdf: \n\ + SELECT ?r ?t WHERE { ?r rdf:reifies ?t } ORDER BY ?r"; + let result = support::query_sparql_formatted(fluree, ledger, sparql) + .await + .expect("link query"); + rows(&result) +} + +/// The incremental path. The first index of the ledger interns its reifier; +/// the next window adds a second reifier on the same edge, whose handle must +/// come from the base dictionary, and a reifier on a new edge, whose handle +/// is allocated above the base watermark. Both are answered from the index +/// after the incremental build appends to the dictionary. +#[tokio::test] +async fn incremental_index_appends_new_terms_and_reuses_existing_handles() { + use fluree_db_indexer::IndexerConfig; + + let fluree = FlureeBuilder::memory() + .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) + .build_memory(); + let ledger_id = "it/triple-term-links:incremental"; + let (local, handle) = + support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let ctx = json!({ "ex": "http://example.org/" }); + + local + .run_until(async { + let ledger0 = support::genesis_ledger(&fluree, ledger_id); + let first = fluree + .insert( + ledger0, + &json!({ + "@context": ctx, + "@id": "ex:alice", + "ex:knows": { + "@id": "ex:bob", + "@annotation": { "@id": "ex:claim1", "ex:source": { "@id": "ex:hr" } } + } + }), + ) + .await + .expect("first annotated insert"); + support::trigger_index_and_wait(&handle, ledger_id, first.receipt.t).await; + support::wait_for_index_application(&fluree, ledger_id, first.receipt.t).await; + let ledger1 = fluree.ledger(ledger_id).await.expect("reload after first index"); + let got = links(&fluree, &ledger1).await; + assert_eq!(got.len(), 1, "first index must intern the first reifier: {got:#?}"); + + let second = fluree + .insert( + ledger1, + &json!({ + "@context": ctx, + "@graph": [ + { + "@id": "ex:alice", + "ex:knows": { + "@id": "ex:bob", + "@annotation": { "@id": "ex:claim2", "ex:source": { "@id": "ex:crm" } } + } + }, + { + "@id": "ex:carol", + "ex:knows": { + "@id": "ex:dave", + "@annotation": { "@id": "ex:claim3", "ex:source": { "@id": "ex:hr" } } + } + } + ] + }), + ) + .await + .expect("second annotated insert"); + support::trigger_index_and_wait(&handle, ledger_id, second.receipt.t).await; + support::wait_for_index_application(&fluree, ledger_id, second.receipt.t).await; + let ledger2 = fluree + .ledger(ledger_id) + .await + .expect("reload after incremental index"); + let got = links(&fluree, &ledger2).await; + assert_eq!(got.len(), 3, "{got:#?}"); + let term = |reifier: &str| -> String { + got.iter() + .find(|r| r[0].ends_with(reifier)) + .unwrap_or_else(|| panic!("no link for {reifier}: {got:#?}"))[1] + .clone() + }; + assert_eq!( + term("claim1"), + term("claim2"), + "two reifiers of one edge must share its term" + ); + let t3 = term("claim3"); + assert!( + t3.contains("carol") && t3.contains("dave"), + "the new edge gets its own term: {t3}" + ); + }) + .await; +} diff --git a/fluree-db-binary-index/src/dict/incremental.rs b/fluree-db-binary-index/src/dict/incremental.rs index bbbabaaf26..3199d3264d 100644 --- a/fluree-db-binary-index/src/dict/incremental.rs +++ b/fluree-db-binary-index/src/dict/incremental.rs @@ -21,8 +21,7 @@ use std::io; use super::branch::{BranchLeafEntry, DictBranch}; use super::builder::LeafArtifact; use super::pack_builder::{ - build_string_forward_packs, build_subject_forward_packs_for_ns, PackArtifact, - DEFAULT_TARGET_PACK_BYTES, DEFAULT_TARGET_PAGE_BYTES, + build_string_forward_packs, PackArtifact, DEFAULT_TARGET_PACK_BYTES, DEFAULT_TARGET_PAGE_BYTES, }; use super::reverse_leaf::{encode_reverse_leaf, ReverseEntry, ReverseLeaf}; use crate::format::wire_helpers::PackBranchEntry; @@ -101,6 +100,22 @@ pub fn build_incremental_subject_packs_for_ns( ns_code: u16, existing_refs: &[PackBranchEntry], new_entries: &[(u64, &[u8])], +) -> io::Result { + build_incremental_packs_for_stream( + crate::dict::forward_pack::KIND_SUBJECT_FWD, + ns_code, + existing_refs, + new_entries, + ) +} + +/// Append packs of `kind` for one id stream: the subject dictionary's +/// per-namespace streams and the term dictionary's per-predicate streams. +pub fn build_incremental_packs_for_stream( + kind: u8, + ns_code: u16, + existing_refs: &[PackBranchEntry], + new_entries: &[(u64, &[u8])], ) -> io::Result { if new_entries.is_empty() { return Ok(IncrementalPackResult { @@ -109,7 +124,8 @@ pub fn build_incremental_subject_packs_for_ns( }); } - let result = build_subject_forward_packs_for_ns( + let result = crate::dict::pack_builder::build_forward_packs_for_stream( + kind, ns_code, new_entries, DEFAULT_TARGET_PAGE_BYTES, @@ -1043,3 +1059,20 @@ mod tests { ); } } + +#[cfg(test)] +mod stream_kind_tests { + use super::*; + use crate::dict::forward_pack::{ForwardPack, KIND_TERM_FWD}; + + #[test] + fn stream_packs_carry_the_requested_kind() { + let entries: Vec<(u64, &[u8])> = vec![(0, b"aaaa"), (1, b"bbbb")]; + let result = build_incremental_packs_for_stream(KIND_TERM_FWD, 7, &[], &entries).unwrap(); + assert_eq!(result.new_packs.len(), 1); + let pack = ForwardPack::from_bytes(&result.new_packs[0].bytes).unwrap(); + assert_eq!(pack.header().kind, KIND_TERM_FWD); + assert_eq!(pack.header().ns_code, 7); + assert_eq!(result.all_pack_refs.len(), 1); + } +} diff --git a/fluree-db-binary-index/src/dict/pack_builder.rs b/fluree-db-binary-index/src/dict/pack_builder.rs index 5528e671c4..7efe272c7e 100644 --- a/fluree-db-binary-index/src/dict/pack_builder.rs +++ b/fluree-db-binary-index/src/dict/pack_builder.rs @@ -144,6 +144,22 @@ pub fn build_subject_forward_packs_for_ns( } /// Internal: partition contiguous entries into packs. +/// Build forward packs of any `kind` for one contiguous id stream. The +/// subject and term dictionaries share this; they differ only in the kind +/// byte and what the `ns_code` header slot means. +pub fn build_forward_packs_for_stream( + kind: u8, + ns_code: u16, + entries: &[(u64, &[u8])], + target_page_bytes: usize, + target_pack_bytes: usize, +) -> io::Result { + if entries.is_empty() { + return Ok(PackBuildResult { packs: Vec::new() }); + } + build_packs_from_contiguous(entries, kind, ns_code, target_page_bytes, target_pack_bytes) +} + fn build_packs_from_contiguous( entries: &[(u64, &[u8])], kind: u8, diff --git a/fluree-db-binary-index/src/dict/term_dict.rs b/fluree-db-binary-index/src/dict/term_dict.rs index 2d04a11ed3..eeb9540d93 100644 --- a/fluree-db-binary-index/src/dict/term_dict.rs +++ b/fluree-db-binary-index/src/dict/term_dict.rs @@ -28,7 +28,7 @@ use std::sync::Arc; /// The pack header's `ns_code` slot is 16 bits; the low half of the inner /// predicate id goes there as a read-time consistency check. #[inline] -fn pack_ns_code(p_id: u32) -> u16 { +pub fn pack_ns_code(p_id: u32) -> u16 { (p_id & 0xFFFF) as u16 } @@ -193,6 +193,14 @@ impl TermDictBuilder { self.map.is_empty() } + /// Every interned term as `(handle, key)`, ascending by handle, which is + /// `(p_id, seq)` order: what an incremental pack append consumes. + pub fn entries_sorted(&self) -> Vec<(u64, TermKey)> { + let mut v: Vec<(u64, TermKey)> = self.map.iter().map(|(k, h)| (*h, *k)).collect(); + v.sort_unstable_by_key(|(h, _)| *h); + v + } + /// `(p_id, highest seq allocated)` per predicate with at least one term. pub fn watermarks(&self) -> Vec<(u32, u32)> { let mut wms: Vec<(u32, u32)> = self diff --git a/fluree-db-indexer/src/build/dicts.rs b/fluree-db-indexer/src/build/dicts.rs index 464b9fb477..4306c3cb3c 100644 --- a/fluree-db-indexer/src/build/dicts.rs +++ b/fluree-db-indexer/src/build/dicts.rs @@ -61,6 +61,35 @@ pub(crate) async fn upload_incremental_reverse_tree_async_strings( /// Core async reverse tree upload: pre-fetch affected leaves, spawn_blocking /// for CoW update, async-upload new artifacts. +/// Triple-term reverse tree append: entries are `(encoded TermKey, handle)`. +pub(crate) async fn upload_incremental_reverse_tree_async_terms( + content_store: &dyn ContentStore, + existing_refs: &DictTreeRefs, + new_terms: &[(u32, u32, Vec)], + warm_cache: Option<&LeafletCache>, +) -> Result { + use fluree_db_binary_index::dict::reverse_leaf::ReverseEntry; + use fluree_db_core::triple_term::term_handle; + + let mut entries: Vec = new_terms + .iter() + .map(|(p_id, seq, key)| ReverseEntry { + key: key.clone(), + id: term_handle(*p_id, *seq), + }) + .collect(); + entries.sort_by(|a, b| a.key.cmp(&b.key)); + + upload_incremental_reverse_tree_core( + content_store, + fluree_db_core::DictKind::TermReverse, + existing_refs, + entries, + warm_cache, + ) + .await +} + async fn upload_incremental_reverse_tree_core( content_store: &dyn ContentStore, dict: fluree_db_core::DictKind, diff --git a/fluree-db-indexer/src/build/incremental.rs b/fluree-db-indexer/src/build/incremental.rs index 51cebf667d..aab8d530a3 100644 --- a/fluree-db-indexer/src/build/incremental.rs +++ b/fluree-db-indexer/src/build/incremental.rs @@ -1272,6 +1272,113 @@ pub async fn incremental_index( root_builder.set_dict_refs(new_dict_refs); + // Triple-term dictionary: append this window's new terms. Packs are + // per inner predicate and only ever append; the reverse tree is updated + // copy-on-write like the subject tree. + if !novelty.new_terms.is_empty() { + use fluree_db_binary_index::dict::forward_pack::KIND_TERM_FWD; + use fluree_db_binary_index::dict::incremental::build_incremental_packs_for_stream; + use fluree_db_binary_index::dict::term_dict::pack_ns_code; + use fluree_db_binary_index::format::wire_helpers::PackBranchEntry; + let base = novelty.base_root.term_dict.clone(); + let mut forward_packs: Vec<(u32, Vec)> = base + .as_ref() + .map(|b| b.forward_packs.clone()) + .unwrap_or_default(); + let mut by_pred: std::collections::BTreeMap> = + std::collections::BTreeMap::new(); + for (p_id, seq, key) in &novelty.new_terms { + by_pred + .entry(*p_id) + .or_default() + .push((*seq as u64, key.as_slice())); + } + for (p_id, entries) in &by_pred { + let existing: Vec = forward_packs + .iter() + .find(|(p, _)| p == p_id) + .map(|(_, refs)| refs.clone()) + .unwrap_or_default(); + let pack_result = build_incremental_packs_for_stream( + KIND_TERM_FWD, + pack_ns_code(*p_id), + &existing, + entries, + ) + .map_err(|e| { + IndexerError::StorageWrite(format!("term fwd pack build p_id={p_id}: {e}")) + })?; + let kind = ContentKind::DictBlob { + dict: fluree_db_core::DictKind::TermForward { p_id: *p_id }, + }; + let mut updated = existing; + for pack in &pack_result.new_packs { + let pack_cid = content_store + .put(kind, &pack.bytes) + .await + .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; + updated.push(PackBranchEntry { + first_id: pack.first_id, + last_id: pack.last_id, + pack_cid, + }); + } + if let Some(entry) = forward_packs.iter_mut().find(|(p, _)| p == p_id) { + entry.1 = updated; + } else { + forward_packs.push((*p_id, updated)); + } + } + forward_packs.sort_by_key(|(p, _)| *p); + + let empty_tree = fluree_db_binary_index::DictTreeRefs { + branch: fluree_db_core::ContentId::from_hex_digest( + fluree_db_core::content_kind::CODEC_FLUREE_DICT_BLOB, + &fluree_db_core::sha256_hex(b""), + ) + .expect("valid digest"), + leaves: Vec::new(), + }; + let (base_reverse, base_count) = match &base { + Some(b) => (b.reverse.clone(), b.term_count), + None => (empty_tree, 0), + }; + let updated_tree = if base.is_some() { + super::dicts::upload_incremental_reverse_tree_async_terms( + content_store.as_ref(), + &base_reverse, + &novelty.new_terms, + warm_cache.as_deref(), + ) + .await? + } else { + // No base dictionary: build the tree from scratch through the + // same core, against an empty existing tree. + super::dicts::upload_incremental_reverse_tree_async_terms( + content_store.as_ref(), + &fluree_db_binary_index::DictTreeRefs { + branch: base_reverse.branch.clone(), + leaves: Vec::new(), + }, + &novelty.new_terms, + warm_cache.as_deref(), + ) + .await? + }; + let refs = fluree_db_binary_index::TermDictRefs { + forward_packs, + reverse: updated_tree.tree_refs, + watermarks: novelty.term_watermarks.clone(), + term_count: base_count + novelty.new_terms.len() as u64, + }; + tracing::debug!( + new_terms = novelty.new_terms.len(), + term_count = refs.term_count, + "V6 Phase 3: triple-term dictionary updated" + ); + root_builder.set_term_dict(Some(refs), updated_tree.replaced_cids); + } + // Update metadata from resolver state. let new_ns_codes: std::collections::BTreeMap = novelty .shared diff --git a/fluree-db-indexer/src/build/rebuild.rs b/fluree-db-indexer/src/build/rebuild.rs index dbd2542199..6d84710a9b 100644 --- a/fluree-db-indexer/src/build/rebuild.rs +++ b/fluree-db-indexer/src/build/rebuild.rs @@ -272,6 +272,8 @@ where _span_b.record("fetch_concurrency", fetch_concurrency); let mut shared = SharedResolverState::new_for_ledger(&ledger_id); + // Rebuilds resolve the term ordinals in Phase C, so links are synthesized. + shared.link_synth.enable(); // Pre-insert rdf:type into predicate dictionary so class tracking // works from the very first commit. @@ -443,13 +445,27 @@ where let mut subject_dicts = Vec::with_capacity(chunks.len()); let mut string_dicts = Vec::with_capacity(chunks.len()); let mut chunk_records: Vec> = Vec::with_capacity(chunks.len()); + let mut chunk_terms: Vec> = Vec::with_capacity(chunks.len()); for chunk in chunks { subject_dicts.push(chunk.subjects); string_dicts.push(chunk.strings); chunk_records.push(chunk.records); + chunk_terms.push(chunk.terms); } + // Triple-term interning happens here, once ids are global: the + // registry gives each term entry its `o_type`, the builder its + // handle. Custom datatypes are all known after Phase B. + let term_registry = { + let reserved = fluree_db_core::DatatypeDictId::RESERVED_COUNT as usize; + let custom: Vec = (reserved..shared.datatypes.len() as usize) + .filter_map(|i| shared.datatypes.resolve(i as u32).map(str::to_string)) + .collect(); + fluree_db_core::o_type_registry::OTypeRegistry::new(&custom) + }; + let mut term_builder = fluree_db_binary_index::dict::TermDictBuilder::new(); + let (subject_merge, subject_remaps) = run_index::dict_merge::merge_subject_dicts(&subject_dicts); let (string_merge, string_remaps) = @@ -569,6 +585,65 @@ where // else: inline types, no remap needed } + // Term entries get the same remap; link records then trade + // their ordinal for the global handle. + let terms = &mut chunk_terms[ci]; + for term in terms.iter_mut() { + let local_s = term.s_id.as_u64() as usize; + let global_s = *s_remap.get(local_s).ok_or_else(|| { + IndexerError::StorageWrite(format!( + "term subject remap miss: chunk {ci}, local_s={local_s}" + )) + })?; + term.s_id = fluree_db_core::subject_id::SubjectId::from_u64(global_s); + let kind = fluree_db_core::value_id::ObjKind::from_u8(term.o_kind); + if kind == fluree_db_core::value_id::ObjKind::REF_ID { + let local_o = term.o_key as usize; + term.o_key = *s_remap.get(local_o).ok_or_else(|| { + IndexerError::StorageWrite(format!( + "term object remap miss: chunk {ci}, local_o={local_o}" + )) + })?; + } else if kind == fluree_db_core::value_id::ObjKind::LEX_ID + || kind == fluree_db_core::value_id::ObjKind::JSON_ID + { + let local_str = fluree_db_core::value_id::ObjKey::from_u64(term.o_key) + .decode_u32_id() as usize; + let global_str = *str_remap.get(local_str).ok_or_else(|| { + IndexerError::StorageWrite(format!( + "term string remap miss: chunk {ci}, local_str={local_str}" + )) + })?; + term.o_key = + fluree_db_core::value_id::ObjKey::encode_u32_id(global_str).as_u64(); + } + } + let triple_term = fluree_db_core::value_id::ObjKind::TRIPLE_TERM.as_u8(); + for record in records.iter_mut() { + if record.o_kind != triple_term { + continue; + } + let term = terms.get(record.o_key as usize).ok_or_else(|| { + IndexerError::StorageWrite(format!( + "term ordinal {} out of range in chunk {ci}", + record.o_key + )) + })?; + let key = fluree_db_core::triple_term::TermKey { + s_id: term.s_id.as_u64(), + p_id: term.p_id, + o_type: term_registry.resolve( + fluree_db_core::value_id::ObjKind::from_u8(term.o_kind), + fluree_db_core::DatatypeDictId::from_u16(term.dt), + term.lang_id, + ), + o_key: term.o_key, + }; + record.o_key = term_builder + .get_or_insert(key) + .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; + } + // Sort by (g_id, SPOT). records.sort_unstable_by(fluree_db_binary_index::format::run_record::cmp_g_spot); @@ -625,6 +700,7 @@ where // copies immediately. For large datasets (e.g. 60M flakes) this // reclaims ~2-6 GB of heap before the index build phase. drop(chunk_records); + drop(chunk_terms); drop(subject_remaps); drop(string_remaps); drop(subject_dicts); @@ -1055,6 +1131,23 @@ where .instrument(tracing::debug_span!("upload_dicts_v3")) .await?; + let term_dict = if term_builder.is_empty() { + None + } else { + let term_count = term_builder.len(); + let refs = term_builder + .upload(&content_store) + .await + .map_err(|e| IndexerError::StorageWrite(format!("term dictionary: {e}")))?; + tracing::info!( + term_count, + predicates = refs.forward_packs.len(), + links = shared.link_synth.links_emitted, + "triple-term dictionary uploaded" + ); + Some(refs) + }; + // Build namespace codes BTreeMap from shared.ns_prefixes. let ns_codes: std::collections::BTreeMap = shared .ns_prefixes @@ -1186,6 +1279,7 @@ where sketch_ref, attachment_events: config.attachment_events.clone(), prev_index: prev_index.clone(), + term_dict, }; let result = super::root_assembly::encode_and_write_root_v6( diff --git a/fluree-db-indexer/src/build/root_assembly.rs b/fluree-db-indexer/src/build/root_assembly.rs index c51a992b3a..164961b49c 100644 --- a/fluree-db-indexer/src/build/root_assembly.rs +++ b/fluree-db-indexer/src/build/root_assembly.rs @@ -240,6 +240,9 @@ pub(crate) struct Fir6Inputs { /// prior root — without it a full rebuild under `Augment` coverage /// silently drops a previously-sealed arena. pub prev_index: Option, + /// Triple-term dictionary interned by this build, if any reification + /// links were synthesized. + pub term_dict: Option, } /// Encode an `IndexRoot` (FIR6), write to CAS, and return an `IndexResult`. @@ -332,9 +335,7 @@ pub(crate) async fn encode_and_write_root_v6( sketch_ref: inputs.sketch_ref, has_annotations, annotation_index: None, - // Full rebuilds do not intern reification links yet; the term - // dictionary is produced by bulk import only. - term_dict: None, + term_dict: inputs.term_dict, // Sticky bit flipped to `true` below if the rebuild path // seals an `Authoritative` arena. Rebuilds always start // from scratch with no prior root, so this is the only diff --git a/fluree-db-indexer/src/gc/collector.rs b/fluree-db-indexer/src/gc/collector.rs index 0c54d48a2a..2db3d67ae2 100644 --- a/fluree-db-indexer/src/gc/collector.rs +++ b/fluree-db-indexer/src/gc/collector.rs @@ -2337,6 +2337,7 @@ mod tests { sketch_ref: None, attachment_events: None, prev_index, + term_dict: None, } } diff --git a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs index 0b5efd3116..37c1b0a5ad 100644 --- a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs +++ b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs @@ -202,6 +202,12 @@ pub struct IncrementalNovelty { pub base_numbig_counts: HashMap<(u16, u32), usize>, /// String text bytes for fulltext assertion entries. pub fulltext_string_bytes: HashMap>, + /// Triple terms first interned by this window as `(p_id, seq, key bytes)`, + /// ascending by `(p_id, seq)`; their handles continue above the base + /// root's per-predicate watermarks. + pub new_terms: Vec<(u32, u32, Vec)>, + /// Per-predicate term watermarks after this window (base merged with new). + pub term_watermarks: Vec<(u32, u32)>, } // ============================================================================ @@ -280,6 +286,8 @@ pub async fn resolve_incremental_commits_v6( // 4. Seed SharedResolverState from V6 root. let mut shared = SharedResolverState::from_index_root(&root)?; + // Incremental builds resolve term ordinals in step 9a, so links are synthesized. + shared.link_synth.enable(); // Enable spatial hook for non-POINT geometry detection. shared.spatial_hook = Some(crate::spatial_hook::SpatialHook::new()); @@ -568,6 +576,7 @@ pub async fn resolve_incremental_commits_v6( commit_count, "V6 incremental resolve: timings (no records)" ); + let base_term_watermarks = root_term_watermarks(&root); return Ok(IncrementalNovelty { records: Vec::new(), ops: Vec::new(), @@ -586,6 +595,8 @@ pub async fn resolve_incremental_commits_v6( base_vector_counts, base_numbig_counts, fulltext_string_bytes: HashMap::new(), + new_terms: Vec::new(), + term_watermarks: base_term_watermarks, }); } @@ -653,6 +664,90 @@ pub async fn resolve_incremental_commits_v6( } let t_remap_records_ms = t0.elapsed().as_millis() as u64; + // 9a. Triple terms: remap the chunk's term table the same way, then give + // every link record its global handle — an existing one from the base + // dictionary, or a fresh one above the base watermarks. + let mut chunk_terms = chunk.terms; + for term in &mut chunk_terms { + remap_record(term, &reconcile.subject_remap, &reconcile.string_remap)?; + } + let (new_terms, term_watermarks) = { + let base_refs = root.term_dict.as_ref(); + let base_reader = match base_refs { + Some(refs) => { + let cache_dir = config + .artifact_cache_dir + .clone() + .unwrap_or_else(std::env::temp_dir); + Some( + fluree_db_binary_index::dict::TermDictReader::from_refs_reusing( + Arc::clone(&cs), + &cache_dir, + refs, + None, + None, + ) + .await + .map_err(|e| { + IncrementalResolveError::DictTreeLoad(format!("term dictionary: {e}")) + })?, + ) + } + None => None, + }; + let base_wms: Vec<(u32, u32)> = base_refs.map(|r| r.watermarks.clone()).unwrap_or_default(); + let mut builder = + fluree_db_binary_index::dict::TermDictBuilder::above_watermarks(&base_wms); + let triple_term = ObjKind::TRIPLE_TERM.as_u8(); + for record in &mut v1_records { + if record.o_kind != triple_term { + continue; + } + let term = chunk_terms.get(record.o_key as usize).ok_or_else(|| { + IncrementalResolveError::Io(std::io::Error::new( + std::io::ErrorKind::InvalidData, + format!("term ordinal {} out of range", record.o_key), + )) + })?; + let key = fluree_db_core::triple_term::TermKey { + s_id: term.s_id.as_u64(), + p_id: term.p_id, + o_type: o_type_registry.resolve( + ObjKind::from_u8(term.o_kind), + fluree_db_core::DatatypeDictId::from_u16(term.dt), + term.lang_id, + ), + o_key: term.o_key, + }; + let existing = match (&base_reader, builder.get(&key)) { + (_, Some(h)) => Some(h), + (Some(reader), None) => reader.find_handle(&key)?, + (None, None) => None, + }; + record.o_key = match existing { + Some(h) => h, + None => builder.get_or_insert(key)?, + }; + } + let new_terms: Vec<(u32, u32, Vec)> = builder + .entries_sorted() + .into_iter() + .map(|(h, key)| { + ( + fluree_db_core::triple_term::term_handle_p_id(h), + fluree_db_core::triple_term::term_handle_seq(h), + key.to_be_bytes().to_vec(), + ) + }) + .collect(); + let mut wms: std::collections::BTreeMap = base_wms.into_iter().collect(); + for (p_id, wm) in builder.watermarks() { + wms.insert(p_id, wm); + } + (new_terms, wms.into_iter().collect::>()) + }; + drop(chunk_terms); + // VECTOR_ID handles are already globally-correct: chunk inserts // appended to the pre-loaded base arena (step 4b) so they return // `base_count..base_count+chunk_count` directly, and retractions / @@ -766,9 +861,19 @@ pub async fn resolve_incremental_commits_v6( base_vector_counts, base_numbig_counts, fulltext_string_bytes, + new_terms, + term_watermarks, }) } +/// The base root's per-predicate term watermarks, or none. +fn root_term_watermarks(root: &IndexRoot) -> Vec<(u32, u32)> { + root.term_dict + .as_ref() + .map(|r| r.watermarks.clone()) + .unwrap_or_default() +} + // ============================================================================ // Internal: Fact lifecycle dedup // ============================================================================ diff --git a/fluree-db-indexer/src/run_index/build/incremental_root.rs b/fluree-db-indexer/src/run_index/build/incremental_root.rs index 2990df5b16..1d7b12bca2 100644 --- a/fluree-db-indexer/src/run_index/build/incremental_root.rs +++ b/fluree-db-indexer/src/run_index/build/incremental_root.rs @@ -118,6 +118,18 @@ impl IncrementalRootBuilder { } /// Update subject and string watermarks. + /// Replace the triple-term dictionary refs, recording the old reverse + /// tree's CIDs that the update superseded as garbage. Packs only append, + /// so none are replaced here. + pub fn set_term_dict( + &mut self, + refs: Option, + replaced: Vec, + ) { + self.root.term_dict = refs; + self.replaced_cids.extend(replaced); + } + pub fn set_watermarks(&mut self, subject_watermarks: Vec, string_watermark: u32) { self.root.subject_watermarks = subject_watermarks; self.root.string_watermark = string_watermark; diff --git a/fluree-db-indexer/src/run_index/resolve/link_synth.rs b/fluree-db-indexer/src/run_index/resolve/link_synth.rs new file mode 100644 index 0000000000..d15eb3d389 --- /dev/null +++ b/fluree-db-indexer/src/run_index/resolve/link_synth.rs @@ -0,0 +1,213 @@ +//! Synthesizes the RDF 1.2 link record for every `f:reifies*` bundle the +//! resolver sees, so rebuilds carry the same `_:r rdf:reifies ` flakes +//! bulk import writes. +//! +//! A bundle's three required slots arrive as consecutive records about the +//! reifier subject (the writer emits them together, and a SPOT-sorted commit +//! keeps one subject's records adjacent). The assembler keys a pending bundle +//! on `(g_id, reifier, t, op)`, fills the subject, predicate and object slots +//! as their records pass, and on completion appends the base edge as a +//! pseudo-record to the chunk's term table and a link record, whose `o_key` +//! is that entry's ordinal, to the chunk's records. The build remaps the +//! entry to global ids and interns it, replacing the ordinal with the handle. +//! +//! Disabled unless a build path opts in: a path that has not learned to +//! resolve the ordinals must not see link records. + +use super::global_dict::PredicateDict; +use super::resolver::RebuildChunk; +use fluree_db_binary_index::format::run_record::{RunRecord, LIST_INDEX_NONE}; +use fluree_db_core::commit::codec::raw_reader::{RawObject, RawOp}; +use fluree_db_core::subject_id::SubjectId; +use fluree_db_core::value_id::ObjKind; +use fluree_vocab::{db, fluree}; +use std::collections::HashMap; + +/// Per-build state for link synthesis. +#[derive(Debug, Default)] +pub struct LinkSynth { + enabled: bool, + /// `[f:reifiesSubject, f:reifiesPredicate, f:reifiesObject]` predicate ids, + /// each looked up (never allocated) as soon as the dictionary holds it. + /// Resolved per slot because the first bundle's records enter the + /// dictionary one at a time, and each must already match its slot. + slots: [Option; 3], + /// Predicate-dictionary length at the last slot lookup, so the lookup + /// repeats only when new predicates appeared. + slots_checked_at: u32, + rdf_reifies: Option, + triple_term_dt: Option, + pending: Option, + /// Link records emitted so far. + pub links_emitted: u64, +} + +#[derive(Debug)] +struct Pending { + g_id: u16, + ann: u64, + t: u32, + op: u8, + s: Option, + p: Option, + /// `(o_kind, o_key, dt, lang_id)` of the base edge's object. + o: Option<(u8, u64, u16, u16)>, +} + +impl LinkSynth { + /// A disabled assembler; see [`Self::enable`]. + pub fn new() -> Self { + Self::default() + } + + /// Turn synthesis on. Only a build path that resolves term ordinals may + /// do this. + pub fn enable(&mut self) { + self.enabled = true; + } + + /// Refresh the reserved-slot predicate ids when the dictionary grew. + fn refresh_slots(&mut self, predicates: &PredicateDict) { + if self.slots.iter().all(Option::is_some) || predicates.len() == self.slots_checked_at { + return; + } + self.slots_checked_at = predicates.len(); + let names = [ + db::REIFIES_SUBJECT, + db::REIFIES_PREDICATE, + db::REIFIES_OBJECT, + ]; + for (slot, name) in self.slots.iter_mut().zip(names) { + if slot.is_none() { + *slot = predicates.get(&format!("{}{}", fluree::DB, name)); + } + } + } + + /// Feed one resolved record (with its raw op, for the predicate slot's + /// IRI). Must be called for every record in commit order. + pub fn observe( + &mut self, + raw: &RawOp<'_>, + record: &RunRecord, + predicates: &mut PredicateDict, + datatypes: &mut PredicateDict, + ns_prefixes: &HashMap, + chunk: &mut RebuildChunk, + ) { + if !self.enabled { + return; + } + self.refresh_slots(predicates); + let Some(slot) = self.slots.iter().position(|s| *s == Some(record.p_id)) else { + return; + }; + + let key = (record.g_id, record.s_id.as_u64(), record.t, record.op); + let same = self + .pending + .as_ref() + .is_some_and(|p| (p.g_id, p.ann, p.t, p.op) == key); + if !same { + self.flush(chunk, predicates, datatypes); + self.pending = Some(Pending { + g_id: record.g_id, + ann: record.s_id.as_u64(), + t: record.t, + op: record.op, + s: None, + p: None, + o: None, + }); + } + let pending = self.pending.as_mut().expect("pending bundle set above"); + match slot { + 0 => { + if ObjKind::from_u8(record.o_kind) == ObjKind::REF_ID { + pending.s = Some(record.o_key); + } + } + 1 => { + if let RawObject::Ref { ns_code, name } = raw.o { + let prefix = ns_prefixes + .get(&ns_code) + .map(std::string::String::as_str) + .unwrap_or(""); + pending.p = Some(predicates.get_or_insert_parts(prefix, name)); + } + } + _ => { + // Resolved under `f:reifiesObject`, which matches the base + // edge's encoding for every kind except the per-predicate + // arenas; those bundles are left to the bundle path. + let kind = ObjKind::from_u8(record.o_kind); + if kind != ObjKind::NUM_BIG && kind != ObjKind::VECTOR_ID { + pending.o = Some((record.o_kind, record.o_key, record.dt, record.lang_id)); + } + } + } + } + + /// Emit the pending bundle if complete. Called on a key change and at + /// the end of each commit. + pub fn flush( + &mut self, + chunk: &mut RebuildChunk, + predicates: &mut PredicateDict, + datatypes: &mut PredicateDict, + ) { + let Some(pending) = self.pending.take() else { + return; + }; + let (Some(s), Some(p), Some((o_kind, o_key, dt, lang_id))) = + (pending.s, pending.p, pending.o) + else { + return; + }; + let link_p = *self + .rdf_reifies + .get_or_insert_with(|| predicates.get_or_insert(fluree_vocab::rdf::REIFIES)); + let link_dt = match self.triple_term_dt { + Some(d) => d, + None => { + let raw = datatypes.get_or_insert(fluree::TRIPLE_TERM); + let Ok(d) = u16::try_from(raw) else { + tracing::warn!( + dt_id = raw, + "f:tripleTerm datatype id exceeds u16; link skipped" + ); + return; + }; + self.triple_term_dt = Some(d); + d + } + }; + let ordinal = chunk.terms.len() as u64; + chunk.terms.push(RunRecord { + g_id: pending.g_id, + s_id: SubjectId::from_u64(s), + p_id: p, + dt, + o_kind, + op: 1, + o_key, + t: pending.t, + lang_id, + i: LIST_INDEX_NONE, + }); + chunk.records.push(RunRecord { + g_id: pending.g_id, + s_id: SubjectId::from_u64(pending.ann), + p_id: link_p, + dt: link_dt, + o_kind: ObjKind::TRIPLE_TERM.as_u8(), + op: pending.op, + o_key: ordinal, + t: pending.t, + lang_id: 0, + i: LIST_INDEX_NONE, + }); + chunk.flake_count += 1; + self.links_emitted += 1; + } +} diff --git a/fluree-db-indexer/src/run_index/resolve/mod.rs b/fluree-db-indexer/src/run_index/resolve/mod.rs index 273cfbdcc7..2690794237 100644 --- a/fluree-db-indexer/src/run_index/resolve/mod.rs +++ b/fluree-db-indexer/src/run_index/resolve/mod.rs @@ -1,4 +1,5 @@ pub mod chunk_dict; pub mod global_dict; pub mod lang_remap; +pub mod link_synth; pub mod resolver; diff --git a/fluree-db-indexer/src/run_index/resolve/resolver.rs b/fluree-db-indexer/src/run_index/resolve/resolver.rs index 947c5d4832..676066e0a6 100644 --- a/fluree-db-indexer/src/run_index/resolve/resolver.rs +++ b/fluree-db-indexer/src/run_index/resolve/resolver.rs @@ -1088,6 +1088,9 @@ pub struct SharedResolverState { pub graphs: super::global_dict::PredicateDict, /// Global language tag dict (shared across all chunks — no per-chunk remap needed). pub languages: super::global_dict::LanguageTagDict, + /// RDF 1.2 link synthesis for `f:reifies*` bundles; off until a build + /// path that resolves term ordinals enables it. + pub link_synth: super::link_synth::LinkSynth, /// Per-graph, per-predicate overflow numeric arenas (BigInt/BigDecimal). /// Outer key = g_id, inner key = p_id. pub numbigs: @@ -1174,6 +1177,7 @@ impl SharedResolverState { fulltext_hook_config: crate::fulltext_hook::FulltextHookConfig::default(), schema_hook: None, saw_list_meta: false, + link_synth: super::link_synth::LinkSynth::new(), } } @@ -1302,6 +1306,7 @@ impl SharedResolverState { fulltext_hook_config: crate::fulltext_hook::FulltextHookConfig::default(), schema_hook: None, saw_list_meta: false, + link_synth: super::link_synth::LinkSynth::new(), }) } @@ -1446,6 +1451,15 @@ impl SharedResolverState { return Ok(()); }; + self.link_synth.observe( + &raw_op, + &record, + &mut self.predicates, + &mut self.datatypes, + &self.ns_prefixes, + chunk, + ); + // Feed raw op to spatial hook (needs raw WKT string + resolved IDs). // Note: record.s_id is chunk-local here; subject IDs in spatial entries // must be remapped after dict merge (Phase C). @@ -1488,6 +1502,9 @@ impl SharedResolverState { Ok(()) })?; + self.link_synth + .flush(chunk, &mut self.predicates, &mut self.datatypes); + // Emit txn-meta records into the same chunk. let meta_count = self.emit_txn_meta_chunk( commit_hash_hex, @@ -2123,6 +2140,9 @@ pub struct RebuildChunk { pub strings: super::chunk_dict::ChunkStringDict, /// Buffered RunRecords (with chunk-local subject/string IDs). pub records: Vec, + /// Reified base edges as pseudo-records, addressed by ordinal from the + /// link records' `o_key` (see `link_synth`). + pub terms: Vec, /// Running count of flakes (records) in this chunk. pub flake_count: u64, } @@ -2133,6 +2153,7 @@ impl RebuildChunk { subjects: super::chunk_dict::ChunkSubjectDict::new(), strings: super::chunk_dict::ChunkStringDict::new(), records: Vec::new(), + terms: Vec::new(), flake_count: 0, } } diff --git a/fluree-db-transact/src/stage.rs b/fluree-db-transact/src/stage.rs index 0f0273f427..966e973e8f 100644 --- a/fluree-db-transact/src/stage.rs +++ b/fluree-db-transact/src/stage.rs @@ -226,8 +226,10 @@ async fn cascade_attachment_retracts( // reifies it live, pointing at a triple that no longer exists. let mut all_ann_flakes = all_ann_flakes; stamp_graph(&mut all_ann_flakes, flake.g.as_ref()); + // The index-side `rdf:reifies` link is neither bundle nor body. let (bundle, metadata): (Vec, Vec) = all_ann_flakes .into_iter() + .filter(|f| !fluree_db_core::is_rdf_reifies(&f.p)) .partition(|f| is_reserved_reifies_predicate(&f.p)); if bundle.is_empty() { continue; @@ -352,6 +354,7 @@ async fn cascade_attachment_retracts( ); let (bundle, current_metadata): (Vec, Vec) = all_flakes .into_iter() + .filter(|f| !fluree_db_core::is_rdf_reifies(&f.p)) .partition(|f| is_reserved_reifies_predicate(&f.p)); if bundle.is_empty() { continue; // not an annotation subject @@ -500,7 +503,11 @@ async fn cascade_attachment_retracts( ) .await?; for asserted in all_flakes { - if is_reserved_reifies_predicate(&asserted.p) { + // Bundle flakes are retracted by the caller; the index-side + // link is retired by the next index pass, not by a commit. + if is_reserved_reifies_predicate(&asserted.p) + || fluree_db_core::is_rdf_reifies(&asserted.p) + { continue; } let mut retract = asserted.clone(); From d674e499fdea067cbc72c8afdac15e03bce8827d Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 13:21:51 -0400 Subject: [PATCH 05/92] feat(query): lower reified-triple patterns to the rdf:reifies link Behind `FLUREE_ANNOTATION_TERMS=1`, the SPARQL lowering routes the reified-triple forms (`<< s p o >>` and `?r rdf:reifies <<( s p o )>>`) through the link flake instead of the `f:reifies*` bundle chain: one `annotation rdf:reifies ?__term_N` triple, then `BIND(SUBJECT(?__term_N) AS ?s)` for a component variable no earlier pattern binds, or a `FILTER` equality when one does, since `BIND` overwrites. A constant predicate becomes `FILTER(PREDICATE(?__term_N) =

)`, which the filter pushdown turns into `ObjectBounds::term_predicate` and the scan into one `POST` handle interval: the seek keys narrow the leaflets, and each row's `o_key` is checked against the interval in encoded form, so a count never decodes a term. A fully constant edge composes to a constant term the scan resolves through the reverse tree. `SUBJECT`, `PREDICATE`, `OBJECT` and `isTRIPLE` now evaluate over `FlakeValue::TripleTerm`; `TRIPLE()` stays deferred. Link scans emit late-materialized handles. The annotation syntax `s p o {| |}` and the JSON-LD surface still take the bundle chain. The link tests move to their own binary because the flag is process-wide; the new test pins every access-path shape from the design doc, including a component variable bound by an earlier pattern. --- fluree-db-api/Cargo.toml | 5 + fluree-db-api/src/format/materialize.rs | 21 ++- fluree-db-api/tests/grp_import.rs | 1 - fluree-db-api/tests/grp_triple_terms.rs | 6 + fluree-db-api/tests/it_triple_term_links.rs | 88 ++++++++++++ fluree-db-core/src/query_bounds.rs | 24 ++++ fluree-db-query/src/binary_scan.rs | 74 +++++++++- fluree-db-query/src/eval/dispatch.rs | 4 + fluree-db-query/src/eval/rdf.rs | 62 +++++++++ fluree-db-query/src/execute.rs | 2 + fluree-db-query/src/execute/pushdown.rs | 7 +- fluree-db-query/src/execute/where_plan.rs | 1 + fluree-db-query/src/ir/expression.rs | 8 ++ fluree-db-query/src/object_binding.rs | 9 ++ fluree-db-query/src/planner.rs | 37 +++++ fluree-db-query/src/range_semijoin.rs | 1 + fluree-db-sparql/src/lower/annotation.rs | 143 +++++++++++++++++++- fluree-db-sparql/src/lower/expression.rs | 24 ++-- fluree-db-sparql/src/lower/mod.rs | 9 ++ fluree-db-sparql/src/lower/rdf_star.rs | 10 +- 20 files changed, 504 insertions(+), 32 deletions(-) create mode 100644 fluree-db-api/tests/grp_triple_terms.rs diff --git a/fluree-db-api/Cargo.toml b/fluree-db-api/Cargo.toml index 8918aa160b..ad8dda3597 100644 --- a/fluree-db-api/Cargo.toml +++ b/fluree-db-api/Cargo.toml @@ -182,6 +182,11 @@ path = "tests/grp_graphsource.rs" name = "grp_import" path = "tests/grp_import.rs" +# Own binary: the triple-term link lowering tests set a process-wide flag. +[[test]] +name = "grp_triple_terms" +path = "tests/grp_triple_terms.rs" + # Standalone (NOT in grp_import): re-execs itself and lowers RLIMIT_NOFILE # hard+soft in the child, which must not share a process with other tests. [[test]] diff --git a/fluree-db-api/src/format/materialize.rs b/fluree-db-api/src/format/materialize.rs index fca3170c07..84813f3025 100644 --- a/fluree-db-api/src/format/materialize.rs +++ b/fluree-db-api/src/format/materialize.rs @@ -90,13 +90,20 @@ fn materialize_encoded_lit(binding: &Binding, gv: &BinaryGraphView) -> std::io:: // NUM_BIG arena values share one EncodedLit whose dt_id is hardcoded // to decimal — recover xsd:integer vs xsd:decimal from the decoded // value, not dt_id (issue #1329). - let dt_sid = other.overflow_numeric_datatype_sid().unwrap_or_else(|| { - store - .dt_sids() - .get(*dt_id as usize) - .cloned() - .unwrap_or_else(|| Sid::new(0, "")) - }); + let dt_sid = other + .overflow_numeric_datatype_sid() + .or_else(|| { + other + .is_triple_term() + .then(|| fluree_db_core::triple_term_datatype_sid().clone()) + }) + .unwrap_or_else(|| { + store + .dt_sids() + .get(*dt_id as usize) + .cloned() + .unwrap_or_else(|| Sid::new(0, "")) + }); let meta = store.decode_meta(*lang_id, *i_val); let dtc = match meta.and_then(|m| m.lang.map(std::sync::Arc::from)) { Some(lang) => DatatypeConstraint::LangTag(lang), diff --git a/fluree-db-api/tests/grp_import.rs b/fluree-db-api/tests/grp_import.rs index d153429ee7..784ae1a5e4 100644 --- a/fluree-db-api/tests/grp_import.rs +++ b/fluree-db-api/tests/grp_import.rs @@ -25,4 +25,3 @@ mod it_namespace_new_after_index; mod it_ns_sync_conflict; #[path = "it_pack_validation.rs"] mod it_pack_validation; -mod it_triple_term_links; diff --git a/fluree-db-api/tests/grp_triple_terms.rs b/fluree-db-api/tests/grp_triple_terms.rs new file mode 100644 index 0000000000..df82b0003a --- /dev/null +++ b/fluree-db-api/tests/grp_triple_terms.rs @@ -0,0 +1,6 @@ +#[path = "support/mod.rs"] +mod support; + +// Own binary: the link-lowering tests set a process-wide environment flag. +#[path = "it_triple_term_links.rs"] +mod it_triple_term_links; diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 09db8d7336..8ddbc93975 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -254,3 +254,91 @@ async fn incremental_index_appends_new_terms_and_reuses_existing_handles() { }) .await; } + +/// The link-based lowering (`FLUREE_ANNOTATION_TERMS=1`): reified-triple +/// patterns scan `rdf:reifies` and decompose or constrain the term instead +/// of walking the bundle chain. Every shape the design doc's access-path +/// table names, on the same imported claims. +#[tokio::test] +async fn link_lowering_answers_reified_triple_shapes() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let (fluree, ledger) = import(&[("claims.ttl", CLAIMS)], "it/triple-term-links:lowering").await; + let q = |body: &str| { + format!( + "PREFIX ex: \n\ + PREFIX rdf: \n{body}" + ) + }; + let run = |body: &str| { + let sparql = q(body); + let fluree = &fluree; + let ledger = &ledger; + async move { + let result = support::query_sparql_formatted(fluree, ledger, &sparql) + .await + .unwrap_or_else(|e| panic!("{sparql}: {e}")); + rows(&result) + } + }; + + // Wildcard: every reifier's edge, decomposed from the term. + let got = + run("SELECT ?s ?p ?o WHERE { ?r rdf:reifies <<( ?s ?p ?o )>> } ORDER BY ?s ?p ?o").await; + assert_eq!(got.len(), 4, "{got:#?}"); + assert!( + got.iter() + .any(|r| r[0].ends_with("alice") && r[1].ends_with("knows") && r[2].ends_with("bob")), + "{got:#?}" + ); + assert!( + got.iter() + .any(|r| r[0].ends_with("carol") && r[1].ends_with("age") && r[2] == "42"), + "{got:#?}" + ); + + // Predicate-bound with no body join: the handle interval alone must + // exclude carol's `ex:age` claim. + let got = + run("SELECT ?s ?o WHERE { ?r rdf:reifies <<( ?s ex:knows ?o )>> } ORDER BY ?s ?o").await; + assert_eq!(got.len(), 3, "{got:#?}"); + assert!(got.iter().all(|r| r[1] != "42"), "{got:#?}"); + + // Predicate-bound: the handle interval. + let got = + run("SELECT ?s ?o WHERE { << ?s ex:knows ?o >> ex:source ?src } ORDER BY ?s ?o").await; + assert_eq!(got.len(), 3, "{got:#?}"); + assert!( + got.iter() + .all(|r| r[0].ends_with("alice") || r[0].ends_with("bob")), + "{got:#?}" + ); + + // Fully bound: the composed constant term. + let got = run("SELECT ?src WHERE { << ex:alice ex:knows ex:bob >> ex:source ?src }").await; + assert_eq!(got.len(), 1, "{got:#?}"); + assert!(got[0][0].ends_with("hr"), "{got:#?}"); + + // Subject-bound with a literal object. + let got = run("SELECT ?p ?o WHERE { ?r rdf:reifies <<( ex:carol ?p ?o )>> }").await; + assert_eq!(got.len(), 1, "{got:#?}"); + assert!(got[0][0].ends_with("age") && got[0][1] == "42", "{got:#?}"); + + // Object- and predicate-bound with the reifier's body joined. + let got = run( + "SELECT ?s ?src WHERE { ?r rdf:reifies <<( ?s ex:knows ex:bob )>> . ?r ex:source ?src }", + ) + .await; + assert_eq!(got.len(), 1, "{got:#?}"); + assert!( + got[0][0].ends_with("alice") && got[0][1].ends_with("hr"), + "{got:#?}" + ); + + // A component variable bound earlier joins instead of being rebound. + let got = run("SELECT ?o2 WHERE { ?r1 rdf:reifies <<( ex:alice ex:knows ?o1 )>> . ?r2 rdf:reifies <<( ?o1 ex:knows ?o2 )>> }").await; + assert_eq!(got.len(), 1, "{got:#?}"); + assert!( + got[0][0].ends_with("dave"), + "bob knows dave via alice knows bob: {got:#?}" + ); +} diff --git a/fluree-db-core/src/query_bounds.rs b/fluree-db-core/src/query_bounds.rs index 180b5fdc37..2294b1b04d 100644 --- a/fluree-db-core/src/query_bounds.rs +++ b/fluree-db-core/src/query_bounds.rs @@ -133,6 +133,10 @@ pub struct ObjectBounds { /// Upper bound: (value, inclusive) /// For `?x < 100`, use `(100, false)`. For `?x <= 100`, use `(100, true)`. pub upper: Option<(FlakeValue, bool)>, + /// Restrict triple-term objects to terms whose inner predicate is this + /// SID. Handles are partitioned by inner predicate, so a scan turns this + /// into one `o_key` interval; on materialized values it checks the term. + pub term_predicate: Option, } impl ObjectBounds { @@ -153,6 +157,20 @@ impl ObjectBounds { self } + /// Bounds that only restrict a triple term's inner predicate. + pub fn term_predicate(sid: crate::sid::Sid) -> Self { + Self { + term_predicate: Some(sid), + ..Self::default() + } + } + + /// True when a lower or upper value bound is set (as opposed to only a + /// term-predicate restriction, which an index seek can enforce alone). + pub fn has_value_bounds(&self) -> bool { + self.lower.is_some() || self.upper.is_some() + } + /// Check if a value satisfies the bounds /// /// Uses **type class comparison**: @@ -160,6 +178,12 @@ impl ObjectBounds { /// - Temporal types are only comparable within the same kind (Date vs Date, etc.) /// - Other types require exact type match pub fn matches(&self, value: &FlakeValue) -> bool { + if let Some(p) = &self.term_predicate { + match value { + FlakeValue::TripleTerm(t) if &t.p == p => {} + _ => return false, + } + } // Check lower bound if let Some((lower, inclusive)) = &self.lower { match Self::class_cmp(value, lower) { diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index 718ce2f61e..5f449cf88b 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -215,6 +215,11 @@ pub struct BinaryScanOperator { #[expect(dead_code)] index_hint: Option, object_bounds: Option, + /// Set at open when a term-predicate bound became the POST seek range: + /// the inclusive `o_key` interval every emitted row must fall in. The + /// seek keys only narrow the leaflets read, so rows are checked against + /// it in encoded form, without a decode. + term_o_key_range: Option<(u64, u64)>, /// Bound object value, if the triple pattern's object is a constant. bound_o: Option, /// `bound_o` as its persisted `(o_type, o_key)` when it is an IRI the @@ -746,6 +751,7 @@ impl BinaryScanOperator { emit, index_hint, object_bounds, + term_o_key_range: None, bound_o: None, bound_o_encoded: None, check_s_eq_o, @@ -1440,8 +1446,17 @@ impl BinaryScanOperator { continue; } } + if let Some((lo, hi)) = self.term_o_key_range { + if o_key < lo || o_key > hi { + continue; + } + } + let bounds_need_value = self.object_bounds.as_ref().is_some_and(|b| { + b.has_value_bounds() + || (b.term_predicate.is_some() && self.term_o_key_range.is_none()) + }); let needs_o_decode = (self.bound_o.is_some() && self.bound_o_encoded.is_none()) - || self.object_bounds.is_some() + || bounds_need_value || (!late_materialize && self.o_var_pos.is_some()); // BinaryGraphView::decode_value is novelty-aware: dict-backed types // (IriRef, StringDict, JsonArena) automatically route through @@ -1472,7 +1487,7 @@ impl BinaryScanOperator { } } - if let Some(bounds) = &self.object_bounds { + if let Some(bounds) = self.object_bounds.as_ref().filter(|_| bounds_need_value) { let Some(val) = decoded_o.as_ref() else { return Err(QueryError::Internal( "object bounds require object decoding".to_string(), @@ -2443,8 +2458,25 @@ impl Operator for BinaryScanOperator { let mut range_min_okey: Option = None; let mut range_max_okey: Option = None; let mut range_o_type: Option = None; + let mut term_o_key_range: Option<(u64, u64)> = None; if order == RunSortOrder::Post && filter.p_id.is_some() && self.bound_o.is_none() { if let Some(bounds) = self.object_bounds.as_ref() { + // A term-predicate bound is one contiguous handle interval: + // handles are `(inner p_id << 32) | seq`. + if let Some(sid) = bounds.term_predicate.as_ref() { + if let Some(p) = store_ref + .sid_to_iri(sid) + .and_then(|iri| store_ref.find_predicate_id(&iri)) + { + let (lo, hi) = fluree_db_core::triple_term::term_handle_range(p); + let ot = OType::TRIPLE_TERM.as_u16(); + range_o_type = Some(ot); + range_min_okey = Some(lo); + range_max_okey = Some(hi); + filter.o_type = Some(ot); + term_o_key_range = Some((lo, hi)); + } + } let supports_range = |ot: OType| -> bool { matches!( ot, @@ -2527,6 +2559,8 @@ impl Operator for BinaryScanOperator { // Create cursor. If any of (s_id, p_id, o_type, o_key) are bound OR we have a // temporal object-key range (POST + bounds), construct a narrow min/max key range // so we can seek into the branch manifest rather than scanning all leaves. + self.term_o_key_range = term_o_key_range; + let use_range = |filter: &BinaryFilter| { filter.s_id.is_some() || filter.p_id.is_some() @@ -3933,6 +3967,42 @@ fn value_to_otype_okey( find_numbig_okey(val, store, numbig_ctx) } FlakeValue::Decimal(_) => find_numbig_okey(val, store, numbig_ctx), + // A constant triple term composes to its handle through the term + // dictionary; a term that was never interned matches nothing, which + // the caller's decode-and-compare fallback preserves. + FlakeValue::TripleTerm(term) => { + let s_id = resolve_subject_v3(&term.s, store, dict_novelty)?; + let p_id = store + .sid_to_iri(&term.p) + .and_then(|iri| store.find_predicate_id(&iri)) + .ok_or_else(|| { + std::io::Error::new( + std::io::ErrorKind::Unsupported, + "triple term predicate is not a known predicate", + ) + })?; + let (o_type, o_key) = value_to_otype_okey( + &term.o, + &term.dt, + term.lang.as_deref(), + store, + dict_novelty, + None, + )?; + let key = fluree_db_core::triple_term::TermKey { + s_id, + p_id, + o_type, + o_key, + }; + match store.find_term_handle(&key)? { + Some(handle) => Ok((OType::TRIPLE_TERM, handle)), + None => Err(std::io::Error::new( + std::io::ErrorKind::Unsupported, + "triple term is not interned", + )), + } + } // Not handled: Vector (arena + HNSW identity; raw-merge is the // intended lane). _ => Err(std::io::Error::new( diff --git a/fluree-db-query/src/eval/dispatch.rs b/fluree-db-query/src/eval/dispatch.rs index eedcfca23b..c6164dc312 100644 --- a/fluree-db-query/src/eval/dispatch.rs +++ b/fluree-db-query/src/eval/dispatch.rs @@ -113,6 +113,10 @@ impl Function { Function::Datatype { strict } => rdf::eval_datatype(args, row, ctx, *strict), Function::LangMatches => rdf::eval_lang_matches(args, row, ctx), Function::SameTerm => rdf::eval_same_term(args, row, ctx), + Function::TripleSubject => rdf::eval_triple_subject(args, row, ctx), + Function::TriplePredicate => rdf::eval_triple_predicate(args, row, ctx), + Function::TripleObject => rdf::eval_triple_object(args, row, ctx), + Function::IsTriple => rdf::eval_is_triple(args, row, ctx), Function::Iri => rdf::eval_iri(args, row, ctx), Function::Bnode => rdf::eval_bnode(args, row, ctx), diff --git a/fluree-db-query/src/eval/rdf.rs b/fluree-db-query/src/eval/rdf.rs index d6ae6ad7b8..77e98e883c 100644 --- a/fluree-db-query/src/eval/rdf.rs +++ b/fluree-db-query/src/eval/rdf.rs @@ -447,6 +447,68 @@ pub fn eval_bnode( } } +// ── SPARQL 1.2 triple-term functions ──────────────────────────────────────── + +/// The triple term an argument evaluates to, or `None` when it is unbound +/// or not a term (a SPARQL type error, which yields no value). +fn triple_term_arg( + args: &[Expression], + row: &R, + ctx: Option<&ExecutionContext<'_>>, + name: &str, +) -> Result> { + check_arity(args, 1, name)?; + Ok(match args[0].eval_to_comparable(row, ctx)? { + Some(ComparableValue::TypedLiteral { + val: fluree_db_core::FlakeValue::TripleTerm(t), + .. + }) => Some(*t), + _ => None, + }) +} + +pub fn eval_triple_subject( + args: &[Expression], + row: &R, + ctx: Option<&ExecutionContext<'_>>, +) -> Result> { + Ok(triple_term_arg(args, row, ctx, "SUBJECT")?.map(|t| ComparableValue::Sid(t.s))) +} + +pub fn eval_triple_predicate( + args: &[Expression], + row: &R, + ctx: Option<&ExecutionContext<'_>>, +) -> Result> { + Ok(triple_term_arg(args, row, ctx, "PREDICATE")?.map(|t| ComparableValue::Sid(t.p))) +} + +pub fn eval_triple_object( + args: &[Expression], + row: &R, + ctx: Option<&ExecutionContext<'_>>, +) -> Result> { + Ok(triple_term_arg(args, row, ctx, "OBJECT")? + .and_then(|t| ComparableValue::try_from(&t.o).ok())) +} + +pub fn eval_is_triple( + args: &[Expression], + row: &R, + ctx: Option<&ExecutionContext<'_>>, +) -> Result> { + check_arity(args, 1, "ISTRIPLE")?; + Ok(args[0].eval_to_comparable(row, ctx)?.map(|v| { + ComparableValue::Bool(matches!( + v, + ComparableValue::TypedLiteral { + val: fluree_db_core::FlakeValue::TripleTerm(_), + .. + } + )) + })) +} + #[cfg(test)] mod tests { use super::*; diff --git a/fluree-db-query/src/execute.rs b/fluree-db-query/src/execute.rs index 16356b7bd0..ec5c70e8a1 100644 --- a/fluree-db-query/src/execute.rs +++ b/fluree-db-query/src/execute.rs @@ -349,10 +349,12 @@ mod tests { let a = ObjectBounds { lower: Some((FlakeValue::Long(10), false)), upper: Some((FlakeValue::Long(100), true)), + term_predicate: None, }; let b = ObjectBounds { lower: Some((FlakeValue::Long(20), true)), upper: Some((FlakeValue::Long(80), false)), + term_predicate: None, }; let merged = merge_object_bounds(&a, &b); diff --git a/fluree-db-query/src/execute/pushdown.rs b/fluree-db-query/src/execute/pushdown.rs index 8f90f58a9b..13ebcfa552 100644 --- a/fluree-db-query/src/execute/pushdown.rs +++ b/fluree-db-query/src/execute/pushdown.rs @@ -44,7 +44,6 @@ pub fn extract_bounds_from_filters( consumed_indices.push(idx); } } - (bounds, consumed_indices) } @@ -137,6 +136,10 @@ pub fn merge_object_bounds(a: &ObjectBounds, b: &ObjectBounds) -> ObjectBounds { ObjectBounds { lower: merge_lower_bound(a.lower.as_ref(), b.lower.as_ref()), upper: merge_upper_bound(a.upper.as_ref(), b.upper.as_ref()), + term_predicate: a + .term_predicate + .clone() + .or_else(|| b.term_predicate.clone()), } } @@ -299,10 +302,12 @@ mod tests { let a = ObjectBounds { lower: Some((FlakeValue::Long(10), false)), upper: Some((FlakeValue::Long(100), true)), + term_predicate: None, }; let b = ObjectBounds { lower: Some((FlakeValue::Long(20), true)), upper: Some((FlakeValue::Long(80), false)), + term_predicate: None, }; let merged = merge_object_bounds(&a, &b); diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index 936967eba5..f379e78134 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -5448,6 +5448,7 @@ mod tests { ObjectBounds { lower: Some((FlakeValue::String("2026-01-01".to_string()), true)), upper: None, + term_predicate: None, }, ); diff --git a/fluree-db-query/src/ir/expression.rs b/fluree-db-query/src/ir/expression.rs index cb2177d076..a714fac995 100644 --- a/fluree-db-query/src/ir/expression.rs +++ b/fluree-db-query/src/ir/expression.rs @@ -975,6 +975,14 @@ pub enum Function { }, LangMatches, SameTerm, + /// SPARQL 1.2 `SUBJECT(term)`: the subject of a triple term. + TripleSubject, + /// SPARQL 1.2 `PREDICATE(term)`: the predicate of a triple term. + TriplePredicate, + /// SPARQL 1.2 `OBJECT(term)`: the object of a triple term. + TripleObject, + /// SPARQL 1.2 `isTRIPLE(term)`. + IsTriple, // ========================================================================= // Fluree-specific functions diff --git a/fluree-db-query/src/object_binding.rs b/fluree-db-query/src/object_binding.rs index 15bcc95670..99de85dc05 100644 --- a/fluree-db-query/src/object_binding.rs +++ b/fluree-db-query/src/object_binding.rs @@ -153,6 +153,15 @@ pub(crate) fn late_materialized_object_binding( i_val: encoded_i_val(o_i), t, }), + DecodeKind::TripleTermDict => Some(Binding::EncodedLit { + o_kind: ObjKind::TRIPLE_TERM.as_u8(), + o_key, + p_id, + dt_id: 0, + lang_id: 0, + i_val: encoded_i_val(o_i), + t, + }), DecodeKind::NumBigArena => Some(Binding::EncodedLit { o_kind: ObjKind::NUM_BIG.as_u8(), o_key, diff --git a/fluree-db-query/src/planner.rs b/fluree-db-query/src/planner.rs index d6b196a6e1..2fab24969e 100644 --- a/fluree-db-query/src/planner.rs +++ b/fluree-db-query/src/planner.rs @@ -859,6 +859,9 @@ pub fn extract_object_bounds_for_var( filter: &Expression, object_var: VarId, ) -> Option { + if let Some(sid) = term_predicate_constraint(filter, object_var) { + return Some(ObjectBounds::term_predicate(sid)); + } // Only proceed if filter is range-safe let constraints = extract_range_constraints(filter)?; @@ -880,6 +883,40 @@ pub fn extract_object_bounds_for_var( // Generalized Selectivity Scoring for All Pattern Types // ============================================================================= +/// `PREDICATE(?t) =

` (either operand order) on the object variable of a +/// triple-term scan: the one filter shape the scan can enforce as a handle +/// interval, since handles are partitioned by inner predicate. +fn term_predicate_constraint( + filter: &Expression, + object_var: VarId, +) -> Option { + let Expression::Call { + func: Function::Eq, + args, + } = filter + else { + return None; + }; + if args.len() != 2 { + return None; + } + let is_pred_of = |e: &Expression| { + matches!(e, Expression::Call { func: Function::TriplePredicate, args } + if args.len() == 1 && args[0] == Expression::Var(object_var)) + }; + let const_sid = |e: &Expression| match e { + Expression::Const(FlakeValue::Ref(sid)) => Some(sid.clone()), + _ => None, + }; + if is_pred_of(&args[0]) { + const_sid(&args[1]) + } else if is_pred_of(&args[1]) { + const_sid(&args[0]) + } else { + None + } +} + /// Cardinality estimate for a generalized pattern. /// /// Each variant carries only the data meaningful for that category: diff --git a/fluree-db-query/src/range_semijoin.rs b/fluree-db-query/src/range_semijoin.rs index f6f3f5f161..bd766eff30 100644 --- a/fluree-db-query/src/range_semijoin.rs +++ b/fluree-db-query/src/range_semijoin.rs @@ -663,6 +663,7 @@ impl RangeSemiJoinOperator { let bounds = ObjectBounds { lower: envelope.lower.clone().map(|v| (v, true)), upper: envelope.upper.clone().map(|v| (v, true)), + term_predicate: None, }; let narrow = ColumnSet::single(ColumnId::SId).union(ColumnSet::single(ColumnId::OKey)); let mixed = narrow diff --git a/fluree-db-sparql/src/lower/annotation.rs b/fluree-db-sparql/src/lower/annotation.rs index f9109b97b5..7d82538f93 100644 --- a/fluree-db-sparql/src/lower/annotation.rs +++ b/fluree-db-sparql/src/lower/annotation.rs @@ -84,6 +84,10 @@ impl LoweringContext<'_, E> { let mut out = Vec::new(); let annotation_ref = self.lower_subject(reifier)?; let edge = self.lower_triple_term(triple_term, &mut out)?; + if link_terms_enabled() { + self.lower_reified_link(annotation_ref, edge, &mut out); + return Ok(out); + } out.push(Pattern::AnnotationTarget { annotation: annotation_ref, edge, @@ -118,11 +122,19 @@ impl LoweringContext<'_, E> { let p = self.lower_predicate(&qt.predicate)?; let (o, dtc) = self.lower_object_desugared(&qt.object, cache, out, true)?; - out.push(Pattern::AnnotationTarget { - annotation: annotation_ref.clone(), - edge: IrTriplePattern { s, p, o, dtc }, - body: Vec::new(), - }); + if link_terms_enabled() { + self.lower_reified_link( + annotation_ref.clone(), + IrTriplePattern { s, p, o, dtc }, + out, + ); + } else { + out.push(Pattern::AnnotationTarget { + annotation: annotation_ref.clone(), + edge: IrTriplePattern { s, p, o, dtc }, + body: Vec::new(), + }); + } cache.insert(qt.span, annotation_ref.clone()); Ok(annotation_ref) } @@ -257,3 +269,124 @@ impl LoweringContext<'_, E> { Ok(IrTriplePattern { s, p, o, dtc }) } } + +/// `FLUREE_ANNOTATION_TERMS=1` routes reified-triple patterns through the +/// `rdf:reifies` link flake and the term dictionary instead of the +/// `f:reifies*` bundle chain. Read per lowering so tests can flip it. +pub(super) fn link_terms_enabled() -> bool { + std::env::var("FLUREE_ANNOTATION_TERMS").is_ok_and(|v| v == "1") +} + +impl LoweringContext<'_, E> { + /// Lower a reified-triple pattern to the link form: one + /// `annotation rdf:reifies ?__term` triple whose object is a triple-term + /// handle, then each component of the reified edge either bound from the + /// term (`BIND(SUBJECT(?__term) AS ?s)`) or constrained against it + /// (`FILTER(PREDICATE(?__term) =

)`). A predicate constraint is what the + /// planner turns into one handle interval; a fully constant edge composes + /// to a constant term the scan looks up directly. + pub(super) fn lower_reified_link( + &mut self, + annotation_ref: Ref, + edge: IrTriplePattern, + out: &mut Vec, + ) { + use fluree_db_core::FlakeValue; + use fluree_db_query::ir::{Expression, Function}; + + let reifies = self.encoder.encode_ref(fluree_vocab::rdf::REIFIES); + + // Fully constant edge: compose the term itself. + if let Some(term) = constant_term(&edge) { + out.push(Pattern::Triple(IrTriplePattern { + s: annotation_ref, + p: reifies, + o: IrTerm::Value(FlakeValue::TripleTerm(Box::new(term))), + dtc: None, + })); + return; + } + + let name = format!("?__term_{}", self.term_counter); + self.term_counter += 1; + let t = self.vars.get_or_insert(&name); + out.push(Pattern::Triple(IrTriplePattern { + s: annotation_ref, + p: reifies, + o: IrTerm::Var(t), + dtc: None, + })); + self.link_bound_vars.insert(t); + + let accessor = |f: Function| Expression::call(f, vec![Expression::Var(t)]); + let mut component = |func: Function, term: IrTerm, out: &mut Vec| match term { + IrTerm::Var(v) => { + if self.link_bound_vars.insert(v) { + out.push(Pattern::Bind { + var: v, + expr: accessor(func), + }); + } else { + out.push(Pattern::Filter(Expression::call( + Function::Eq, + vec![accessor(func), Expression::Var(v)], + ))); + } + } + IrTerm::Sid(sid) => out.push(Pattern::Filter(Expression::call( + Function::Eq, + vec![accessor(func), Expression::Const(FlakeValue::Ref(sid))], + ))), + IrTerm::Iri(iri) => match self.encoder.encode_iri(&iri) { + Some(sid) => out.push(Pattern::Filter(Expression::call( + Function::Eq, + vec![accessor(func), Expression::Const(FlakeValue::Ref(sid))], + ))), + // An IRI in no registered namespace names nothing in this + // ledger, so the pattern cannot match. + None => out.push(Pattern::Filter(Expression::Const(FlakeValue::Boolean( + false, + )))), + }, + IrTerm::Value(v) => out.push(Pattern::Filter(Expression::call( + Function::Eq, + vec![accessor(func), Expression::Const(v)], + ))), + }; + component(Function::TripleSubject, edge.s.into(), out); + component(Function::TriplePredicate, edge.p.into(), out); + component(Function::TripleObject, edge.o, out); + } +} + +/// The materialized term for an edge whose three positions are constants +/// and whose object datatype is known; `None` otherwise. +fn constant_term(edge: &IrTriplePattern) -> Option { + use fluree_db_core::FlakeValue; + let s = match &edge.s { + Ref::Sid(s) => s.clone(), + _ => return None, + }; + let p = match &edge.p { + Ref::Sid(p) => p.clone(), + _ => return None, + }; + let (o, dt, lang) = match (&edge.o, &edge.dtc) { + (IrTerm::Sid(sid), _) => ( + FlakeValue::Ref(sid.clone()), + fluree_db_core::edge::id_datatype_sid(), + None, + ), + (IrTerm::Value(v), Some(DatatypeConstraint::Explicit(dt))) => (v.clone(), dt.clone(), None), + (IrTerm::Value(v), Some(DatatypeConstraint::LangTag(tag))) => ( + v.clone(), + fluree_db_core::Sid::new( + fluree_vocab::namespaces::RDF, + fluree_vocab::rdf_names::LANG_STRING, + ), + Some(tag.to_string()), + ), + _ => return None, + }; + Some(fluree_db_core::TripleTermValue { s, p, o, dt, lang }) +} diff --git a/fluree-db-sparql/src/lower/expression.rs b/fluree-db-sparql/src/lower/expression.rs index 276758e0f2..9471020cd3 100644 --- a/fluree-db-sparql/src/lower/expression.rs +++ b/fluree-db-sparql/src/lower/expression.rs @@ -233,16 +233,9 @@ impl LoweringContext<'_, E> { // parse time, but have no evaluable implementation yet: defer per // burn-down decision D-1 (accept-then-defer). A query that reaches here // fails at lower time with a clean `not_implemented`, not a parse error. - if matches!( - name, - FunctionName::Triple - | FunctionName::Subject - | FunctionName::Predicate - | FunctionName::Object - | FunctionName::IsTriple - ) { + if matches!(name, FunctionName::Triple) { return Err(LowerError::not_implemented( - "SPARQL 1.2 triple-term functions (TRIPLE/SUBJECT/PREDICATE/OBJECT/isTRIPLE)", + "SPARQL 1.2 TRIPLE(s, p, o) construction", span, )); } @@ -327,13 +320,14 @@ impl LoweringContext<'_, E> { FunctionName::CosineSimilarity => Function::CosineSimilarity, FunctionName::EuclideanDistance => Function::EuclideanDistance, + // SPARQL 1.2 triple-term accessors over `FlakeValue::TripleTerm`. + FunctionName::Subject => Function::TripleSubject, + FunctionName::Predicate => Function::TriplePredicate, + FunctionName::Object => Function::TripleObject, + FunctionName::IsTriple => Function::IsTriple, // Handled by the `not_implemented` early return above. - FunctionName::Triple - | FunctionName::Subject - | FunctionName::Predicate - | FunctionName::Object - | FunctionName::IsTriple => { - unreachable!("triple-term functions defer via the early return") + FunctionName::Triple => { + unreachable!("TRIPLE() defers via the early return") } // Extension functions diff --git a/fluree-db-sparql/src/lower/mod.rs b/fluree-db-sparql/src/lower/mod.rs index 928276ee09..6f6e3022af 100644 --- a/fluree-db-sparql/src/lower/mod.rs +++ b/fluree-db-sparql/src/lower/mod.rs @@ -400,6 +400,13 @@ struct LoweringContext<'a, E> { pp_counter: u32, /// Monotonic counter for generating expression-based ORDER BY bind variables (`?__order_by_0`, …). order_counter: u32, + /// Monotonic counter for the triple-term variables the link lowering + /// mints (`?__term_0`, …). + term_counter: u32, + /// Variables some earlier pattern binds. The link lowering binds a term's + /// components with `BIND` only for variables not in this set, and joins + /// with a `FILTER` equality otherwise, since `BIND` overwrites. + link_bound_vars: std::collections::HashSet, /// Original SPARQL source text (for extracting SERVICE body text). source_text: Option<&'a str>, } @@ -424,6 +431,8 @@ impl<'a, E: IriEncoder> LoweringContext<'a, E> { agg_counter: 0, pp_counter: 0, order_counter: 0, + term_counter: 0, + link_bound_vars: std::collections::HashSet::new(), source_text, } } diff --git a/fluree-db-sparql/src/lower/rdf_star.rs b/fluree-db-sparql/src/lower/rdf_star.rs index ef9945838a..a42ca580d4 100644 --- a/fluree-db-sparql/src/lower/rdf_star.rs +++ b/fluree-db-sparql/src/lower/rdf_star.rs @@ -109,7 +109,15 @@ impl LoweringContext<'_, E> { match &tp.annotation { Some(ann) => self.lower_annotation_units(edge, ann, &mut result)?, - None => result.push(Pattern::Triple(edge)), + None => { + for v in [edge.s.as_var(), edge.p.as_var(), edge.o.as_var()] + .into_iter() + .flatten() + { + self.link_bound_vars.insert(v); + } + result.push(Pattern::Triple(edge)); + } } } From 26b0d88a8a974974e7e7d905dc14156f92e045fb Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 14:37:03 -0400 Subject: [PATCH 06/92] feat(query): encoded term components and term-identity constraints on the link lowering The flagged link lowering was slower than the bundle chain: a composed constant term never reached the dictionary (the simple encoder had no TripleTerm arm, so a fully bound pattern fell to a decode-and-compare scan of every link), every SUBJECT/PREDICATE/OBJECT accessor materialized the whole term per row, and positions nothing reads were still bound. - compose_term_handle is shared by both encoders; an uninterned term is NotFound, so the scan skips straight to novelty. - An accessor on a late-materialized handle reads the dictionary key and binds the component encoded (EncodedSid, EncodedPid, EncodedLit with its datatype and tag); OBJECT on a materialized term keeps the datatype and tag too. - elide_unread_term_binds drops component BINDs nothing reads, the twin of elide_redundant_chain for the bundle chain. - Variable positions are always BIND: the agreement check is the join in every scope, and bind_unifies normalizes representations (a scan's EncodedSid against a VALUES row's Sid) in BindOperator and apply_inline. The lowering's cross-scope binding set is gone. - Constant positions are sameTerm filters the pushdown turns into ObjectBounds::{term_subject, term_object} beside term_predicate; the scan checks them on the handle's dictionary key (term_key_filter). A literal object carries its datatype or tag as a resolved binding, so "5"^^xsd:int and "chat"@fr match by term identity, as the composed constant does. Two constraints on one component that disagree keep the second filter instead of one silently winning. - Import skips arena-scoped (decimal, big-integer, vector) objects as rebuild does: per-(graph, predicate) handles are no graph-independent term identity. --- fluree-db-api/tests/it_triple_term_links.rs | 175 +++++++++++++++ fluree-db-core/src/query_bounds.rs | 53 ++++- fluree-db-query/src/binary_scan.rs | 157 ++++++++++--- fluree-db-query/src/bind.rs | 8 +- fluree-db-query/src/eval.rs | 232 +++++++++++--------- fluree-db-query/src/eval/rdf.rs | 148 ++++++++++++- fluree-db-query/src/execute.rs | 4 + fluree-db-query/src/execute/pushdown.rs | 31 +++ fluree-db-query/src/execute/where_plan.rs | 60 ++++- fluree-db-query/src/object_binding.rs | 41 ++++ fluree-db-query/src/operator/inline.rs | 2 +- fluree-db-query/src/planner.rs | 78 ++++--- fluree-db-query/src/range_semijoin.rs | 2 + fluree-db-sparql/src/lower/annotation.rs | 89 +++++--- fluree-db-sparql/src/lower/mod.rs | 5 - fluree-db-sparql/src/lower/rdf_star.rs | 10 +- fluree-db-transact/src/import_sink.rs | 8 + 17 files changed, 878 insertions(+), 225 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 8ddbc93975..c6c6676612 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -342,3 +342,178 @@ async fn link_lowering_answers_reified_triple_shapes() { "bob knows dave via alice knows bob: {got:#?}" ); } + +const LITERAL_CLAIMS: &str = r#"VERSION "1.2" +@prefix ex: . +@prefix xsd: . + +ex:doc ex:title "chat"@fr {| ex:source ex:fr |} . +ex:doc ex:title "chat"@en {| ex:source ex:en |} . +ex:doc ex:title "chat" {| ex:source ex:plain |} . +ex:doc ex:size "5"^^xsd:int {| ex:source ex:int |} . +ex:doc ex:size 5 {| ex:source ex:integer |} . +"#; + +async fn run_link_query( + fluree: &fluree_db_api::Fluree, + ledger: &LedgerState, + body: String, +) -> Vec> { + let sparql = format!( + "PREFIX ex: \n\ + PREFIX rdf: \n\ + PREFIX xsd: \n{body}" + ); + let result = support::query_sparql_formatted(fluree, ledger, &sparql) + .await + .unwrap_or_else(|e| panic!("{sparql}: {e}")); + rows(&result) +} + +/// A reified pattern's literal object is a term: its language tag or +/// datatype is part of the match, both when the whole edge composes to a +/// constant term and when only the object is constant. Decomposing a term +/// keeps the tag and datatype too. +#[tokio::test] +async fn link_lowering_matches_literal_objects_by_term() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let (fluree, ledger) = import( + &[("literals.ttl", LITERAL_CLAIMS)], + "it/triple-term-links:literals", + ) + .await; + let run = |body: &str| run_link_query(&fluree, &ledger, body.to_string()); + + for (object, source) in [ + ("\"chat\"@fr", "fr"), + ("\"chat\"@en", "en"), + ("\"chat\"", "plain"), + ] { + for subject in ["ex:doc", "?s"] { + let got = run(&format!( + "SELECT ?src WHERE {{ << {subject} ex:title {object} >> ex:source ?src }}" + )) + .await; + assert_eq!(got.len(), 1, "{subject} {object}: {got:#?}"); + assert!(got[0][0].ends_with(source), "{subject} {object}: {got:#?}"); + } + } + for (object, source) in [("\"5\"^^xsd:int", "int"), ("5", "integer")] { + for subject in ["ex:doc", "?s"] { + let got = run(&format!( + "SELECT ?src WHERE {{ << {subject} ex:size {object} >> ex:source ?src }}" + )) + .await; + assert_eq!(got.len(), 1, "{subject} {object}: {got:#?}"); + assert!(got[0][0].ends_with(source), "{subject} {object}: {got:#?}"); + } + } + + let got = run("SELECT ?o (LANG(?o) AS ?l) WHERE { ?r rdf:reifies <<( ex:doc ex:title ?o )>> } ORDER BY ?l").await; + assert_eq!(got.len(), 3, "{got:#?}"); + let langs: Vec<&str> = got.iter().map(|r| r[1].as_str()).collect(); + assert_eq!(langs, ["", "en", "fr"], "{got:#?}"); + + let got = run("SELECT (DATATYPE(?o) AS ?dt) WHERE { ?r rdf:reifies <<( ex:doc ex:size ?o )>> } ORDER BY ?dt").await; + assert_eq!(got.len(), 2, "{got:#?}"); + assert!(got[0][0].ends_with("int"), "{got:#?}"); + assert!(got[1][0].ends_with("integer"), "{got:#?}"); +} + +/// A component variable is always bound with `BIND`, whose agreement check +/// is the join: the same variable in two UNION branches is fresh in each, +/// a VALUES row or the reifier itself constrains it, and the value the row +/// already holds may be in any representation. +#[tokio::test] +async fn link_lowering_joins_component_variables_in_every_scope() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let (fluree, ledger) = import(&[("claims.ttl", CLAIMS)], "it/triple-term-links:scopes").await; + let run = |body: &str| run_link_query(&fluree, &ledger, body.to_string()); + + let got = run( + "SELECT ?o WHERE { { << ex:alice ex:knows ?o >> ex:source ex:hr } \ + UNION { << ex:bob ex:knows ?o >> ex:source ex:crm } } ORDER BY ?o", + ) + .await; + assert_eq!(got.len(), 2, "{got:#?}"); + assert!( + got[0][0].ends_with("bob") && got[1][0].ends_with("dave"), + "{got:#?}" + ); + + let got = run( + "SELECT ?s ?o WHERE { VALUES ?s { ex:alice ex:bob } << ?s ex:knows ?o >> ex:source ?src } \ + ORDER BY ?s ?o", + ) + .await; + assert_eq!(got.len(), 3, "{got:#?}"); + assert!( + got[2][0].ends_with("bob") && got[2][1].ends_with("dave"), + "{got:#?}" + ); + + let got = run("SELECT ?x WHERE { ?x rdf:reifies <<( ?x ?p ?o )>> }").await; + assert!( + got.is_empty(), + "no reifier is its own edge's subject: {got:#?}" + ); + let got = run("SELECT ?x WHERE { ?x rdf:reifies <<( ?y ?p ?o )>> }").await; + assert_eq!(got.len(), 4, "{got:#?}"); + + let got = run("SELECT ?s WHERE { ?s ex:age ?age . << ?s ?p ?o >> ex:source ?src }").await; + assert!(got.is_empty(), "carol's claim has no source: {got:#?}"); + let got = run( + "SELECT ?s ?o WHERE { ?s ex:knows ?o . ?r rdf:reifies <<( ?s ex:knows ?o )>> } \ + ORDER BY ?s ?o", + ) + .await; + assert_eq!(got.len(), 3, "{got:#?}"); + assert!( + got[2][0].ends_with("bob") && got[2][1].ends_with("dave"), + "{got:#?}" + ); +} + +/// Two inner-predicate constraints on one term that name different +/// predicates admit no handle: the second filter stays in the plan. +#[tokio::test] +async fn link_lowering_keeps_contradictory_predicate_filters() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let (fluree, ledger) = import( + &[("claims.ttl", CLAIMS)], + "it/triple-term-links:contradiction", + ) + .await; + let run = |body: &str| run_link_query(&fluree, &ledger, body.to_string()); + + let got = run("SELECT ?r WHERE { ?r rdf:reifies ?t . FILTER(PREDICATE(?t) = ex:knows) }").await; + assert_eq!(got.len(), 3, "{got:#?}"); + let got = run( + "SELECT ?r WHERE { ?r rdf:reifies ?t . FILTER(PREDICATE(?t) = ex:knows) \ + FILTER(PREDICATE(?t) = ex:age) }", + ) + .await; + assert!(got.is_empty(), "{got:#?}"); +} + +/// Positions the query never reads are not decomposed at all; the count is +/// still the count of matching links. +#[tokio::test] +async fn link_lowering_counts_without_decomposing_unread_positions() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let (fluree, ledger) = import(&[("claims.ttl", CLAIMS)], "it/triple-term-links:count").await; + let got = run_link_query( + &fluree, + &ledger, + "SELECT (COUNT(*) AS ?n) WHERE { << ?s ?p ?o >> ex:source ?src }".to_string(), + ) + .await; + assert_eq!(got, vec![vec!["3".to_string()]], "{got:#?}"); + let got = run_link_query( + &fluree, + &ledger, + "SELECT (COUNT(*) AS ?n) WHERE { << ?s ex:knows ?o >> ex:source ?src }".to_string(), + ) + .await; + assert_eq!(got, vec![vec!["3".to_string()]], "{got:#?}"); +} diff --git a/fluree-db-core/src/query_bounds.rs b/fluree-db-core/src/query_bounds.rs index 2294b1b04d..c2af4629ae 100644 --- a/fluree-db-core/src/query_bounds.rs +++ b/fluree-db-core/src/query_bounds.rs @@ -137,6 +137,13 @@ pub struct ObjectBounds { /// SID. Handles are partitioned by inner predicate, so a scan turns this /// into one `o_key` interval; on materialized values it checks the term. pub term_predicate: Option, + /// Restrict triple-term objects to terms whose inner subject is this + /// SID. A scan compares the handle's dictionary key without decoding. + pub term_subject: Option, + /// Restrict triple-term objects to terms whose inner object is this + /// term: the value with its datatype or language tag, since + /// `<< ?s :p "chat"@fr >>` names a term, not a value. + pub term_object: Option<(FlakeValue, crate::DatatypeConstraint)>, } impl ObjectBounds { @@ -165,12 +172,34 @@ impl ObjectBounds { } } + /// Bounds that only restrict a triple term's inner subject. + pub fn term_subject(sid: crate::sid::Sid) -> Self { + Self { + term_subject: Some(sid), + ..Self::default() + } + } + + /// Bounds that only restrict a triple term's inner object. + pub fn term_object(value: FlakeValue, dtc: crate::DatatypeConstraint) -> Self { + Self { + term_object: Some((value, dtc)), + ..Self::default() + } + } + /// True when a lower or upper value bound is set (as opposed to only a /// term-predicate restriction, which an index seek can enforce alone). pub fn has_value_bounds(&self) -> bool { self.lower.is_some() || self.upper.is_some() } + /// True when the inner subject or object of a triple term is constrained; + /// a scan enforces these on the handle's dictionary key. + pub fn has_term_component_bounds(&self) -> bool { + self.term_subject.is_some() || self.term_object.is_some() + } + /// Check if a value satisfies the bounds /// /// Uses **type class comparison**: @@ -178,10 +207,26 @@ impl ObjectBounds { /// - Temporal types are only comparable within the same kind (Date vs Date, etc.) /// - Other types require exact type match pub fn matches(&self, value: &FlakeValue) -> bool { - if let Some(p) = &self.term_predicate { - match value { - FlakeValue::TripleTerm(t) if &t.p == p => {} - _ => return false, + if self.term_predicate.is_some() || self.has_term_component_bounds() { + let FlakeValue::TripleTerm(t) = value else { + return false; + }; + if self.term_predicate.as_ref().is_some_and(|p| &t.p != p) { + return false; + } + if self.term_subject.as_ref().is_some_and(|s| &t.s != s) { + return false; + } + if let Some((o, dtc)) = &self.term_object { + let same_type = match dtc { + crate::DatatypeConstraint::LangTag(tag) => { + t.lang.as_deref() == Some(tag.as_ref()) + } + crate::DatatypeConstraint::Explicit(dt) => t.lang.is_none() && &t.dt == dt, + }; + if !same_type || &t.o != o { + return false; + } } } // Check lower bound diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index 5f449cf88b..fea5288306 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -171,6 +171,14 @@ fn inline_ops_need_t(ops: &[InlineOperator]) -> bool { // BinaryScanOperator // ============================================================================ +/// Encoded form of the inner-subject and inner-object bounds on a term +/// variable (see `ObjectBounds::term_subject` / `term_object`). +#[derive(Debug, Clone, Copy)] +struct TermKeyFilter { + s_id: Option, + o: Option<(u16, u64)>, +} + /// Scan operator: streams leaflets from `BinaryCursor`, eagerly decoding /// `ColumnBatch` rows into `Binding::Sid` / `Binding::Lit` values. pub struct BinaryScanOperator { @@ -220,6 +228,10 @@ pub struct BinaryScanOperator { /// seek keys only narrow the leaflets read, so rows are checked against /// it in encoded form, without a decode. term_o_key_range: Option<(u64, u64)>, + /// Set at open when the bounds constrain a term's inner subject or + /// object and the store encodes them: every emitted term handle's + /// dictionary key must match, checked without decoding the term. + term_key_filter: Option, /// Bound object value, if the triple pattern's object is a constant. bound_o: Option, /// `bound_o` as its persisted `(o_type, o_key)` when it is an IRI the @@ -752,6 +764,7 @@ impl BinaryScanOperator { index_hint, object_bounds, term_o_key_range: None, + term_key_filter: None, bound_o: None, bound_o_encoded: None, check_s_eq_o, @@ -1451,9 +1464,28 @@ impl BinaryScanOperator { continue; } } + if let Some(f) = self.term_key_filter { + // Only a term can satisfy a term-component bound. + if o_type != OType::TRIPLE_TERM.as_u16() { + continue; + } + let Some(key) = store_arc + .resolve_term_key(o_key) + .map_err(|e| QueryError::from_io("resolve_term_key", e))? + else { + continue; + }; + if f.s_id.is_some_and(|s| s != key.s_id) + || f.o + .is_some_and(|(ot, ok)| ot != key.o_type.as_u16() || ok != key.o_key) + { + continue; + } + } let bounds_need_value = self.object_bounds.as_ref().is_some_and(|b| { b.has_value_bounds() || (b.term_predicate.is_some() && self.term_o_key_range.is_none()) + || (b.has_term_component_bounds() && self.term_key_filter.is_none()) }); let needs_o_decode = (self.bound_o.is_some() && self.bound_o_encoded.is_none()) || bounds_need_value @@ -2561,6 +2593,26 @@ impl Operator for BinaryScanOperator { // so we can seek into the branch manifest rather than scanning all leaves. self.term_o_key_range = term_o_key_range; + // Inner-subject / inner-object bounds on a term variable: encode + // them once here and compare each handle's dictionary key in the + // row loop. A component the dictionaries do not hold names no + // interned term, so no base row can match; only novelty remains. + self.term_key_filter = None; + if let Some(bounds) = self + .object_bounds + .as_ref() + .filter(|b| b.has_term_component_bounds()) + { + match encode_term_key_filter(bounds, store_ref, ctx.dict_novelty.as_ref()) { + Ok(f) => self.term_key_filter = Some(f), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => { + return self.open_overlay_only_fallback(ctx, &s_sid, &p_sid).await; + } + // Unencodable: the row loop decodes and checks the term. + Err(_) => {} + } + } + let use_range = |filter: &BinaryFilter| { filter.s_id.is_some() || filter.p_id.is_some() @@ -3967,42 +4019,7 @@ fn value_to_otype_okey( find_numbig_okey(val, store, numbig_ctx) } FlakeValue::Decimal(_) => find_numbig_okey(val, store, numbig_ctx), - // A constant triple term composes to its handle through the term - // dictionary; a term that was never interned matches nothing, which - // the caller's decode-and-compare fallback preserves. - FlakeValue::TripleTerm(term) => { - let s_id = resolve_subject_v3(&term.s, store, dict_novelty)?; - let p_id = store - .sid_to_iri(&term.p) - .and_then(|iri| store.find_predicate_id(&iri)) - .ok_or_else(|| { - std::io::Error::new( - std::io::ErrorKind::Unsupported, - "triple term predicate is not a known predicate", - ) - })?; - let (o_type, o_key) = value_to_otype_okey( - &term.o, - &term.dt, - term.lang.as_deref(), - store, - dict_novelty, - None, - )?; - let key = fluree_db_core::triple_term::TermKey { - s_id, - p_id, - o_type, - o_key, - }; - match store.find_term_handle(&key)? { - Some(handle) => Ok((OType::TRIPLE_TERM, handle)), - None => Err(std::io::Error::new( - std::io::ErrorKind::Unsupported, - "triple term is not interned", - )), - } - } + FlakeValue::TripleTerm(term) => compose_term_handle(term, store, dict_novelty), // Not handled: Vector (arena + HNSW identity; raw-merge is the // intended lane). _ => Err(std::io::Error::new( @@ -4452,6 +4469,73 @@ fn otype_from_dt_sid(dt_sid: &Sid, store: &BinaryIndexStore) -> Option { Some(OType::customer_datatype(dt_id)) } +/// The inner-subject / inner-object bounds of a term variable in encoded +/// form. `NotFound` when a component is absent from the dictionaries (no +/// interned term can carry it); any other error leaves the bounds to the +/// decoded check. +fn encode_term_key_filter( + bounds: &fluree_db_core::ObjectBounds, + store: &BinaryIndexStore, + dict_novelty: Option<&Arc>, +) -> std::io::Result { + let s_id = match bounds.term_subject.as_ref() { + Some(sid) => Some(resolve_subject_v3(sid, store, dict_novelty)?), + None => None, + }; + let o = match bounds.term_object.as_ref() { + Some((value, dtc)) => { + let (dt, lang) = match dtc { + DatatypeConstraint::Explicit(dt) => (dt.clone(), None), + DatatypeConstraint::LangTag(tag) => ( + Sid::new(namespaces::RDF, rdf_names::LANG_STRING), + Some(tag.as_ref()), + ), + }; + let (ot, key) = value_to_otype_okey(value, &dt, lang, store, dict_novelty, None)?; + Some((ot.as_u16(), key)) + } + None => None, + }; + Ok(TermKeyFilter { s_id, o }) +} + +/// A constant triple term's handle, composed through the term dictionary. +/// A subject, predicate, object or term the dictionary does not hold is +/// `NotFound`: the link index only ever carries interned handles, so the +/// pattern cannot match a base row and the caller may skip to novelty. +pub(crate) fn compose_term_handle( + term: &fluree_db_core::TripleTermValue, + store: &BinaryIndexStore, + dict_novelty: Option<&Arc>, +) -> std::io::Result<(OType, u64)> { + use std::io::{Error, ErrorKind}; + let s_id = resolve_subject_v3(&term.s, store, dict_novelty)?; + let p_id = store + .sid_to_p_id(&term.p) + .ok_or_else(|| Error::new(ErrorKind::NotFound, "triple term predicate is not known"))?; + let (o_type, o_key) = value_to_otype_okey( + &term.o, + &term.dt, + term.lang.as_deref(), + store, + dict_novelty, + None, + )?; + let key = fluree_db_core::triple_term::TermKey { + s_id, + p_id, + o_type, + o_key, + }; + match store.find_term_handle(&key)? { + Some(handle) => Ok((OType::TRIPLE_TERM, handle)), + None => Err(Error::new( + ErrorKind::NotFound, + "triple term is not interned", + )), + } +} + /// Simplified FlakeValue → (OType, o_key) translation for fast-path operators. /// /// Uses default OType for each value variant (no dt_sid/lang context needed). @@ -4468,6 +4552,7 @@ pub(crate) fn value_to_otype_okey_simple( FlakeValue::Null => Ok((OType::NULL, 0)), FlakeValue::Boolean(b) => Ok((OType::XSD_BOOLEAN, *b as u64)), FlakeValue::Long(n) => Ok((OType::XSD_INTEGER, ObjKey::encode_i64(*n).as_u64())), + FlakeValue::TripleTerm(term) => compose_term_handle(term, store, None), FlakeValue::Double(d) => { // Encoding failures are NOT NotFound: the value could still exist in // the base index under a representation we can't compute, so callers diff --git a/fluree-db-query/src/bind.rs b/fluree-db-query/src/bind.rs index feb3700e9f..df962e0028 100644 --- a/fluree-db-query/src/bind.rs +++ b/fluree-db-query/src/bind.rs @@ -256,8 +256,12 @@ impl Operator for BindOperator { match existing { Some(Binding::Unbound) | None => true, // Unbound - can bind Some(existing_val) => { - // Check if same value - existing_val == &computed || matches!(computed, Binding::Unbound) + matches!(computed, Binding::Unbound) + || crate::object_binding::bind_unifies( + existing_val, + &computed, + Some(ctx), + ) } } }; diff --git a/fluree-db-query/src/eval.rs b/fluree-db-query/src/eval.rs index c1d192832b..b4bde059c3 100644 --- a/fluree-db-query/src/eval.rs +++ b/fluree-db-query/src/eval.rs @@ -51,7 +51,7 @@ pub use value::{ArithmeticError, ComparableValue, ComparisonError, NullValueErro use crate::binding::{Binding, BindingRow, RowAccess}; use crate::context::ExecutionContext; use crate::error::{QueryError, Result}; -use crate::ir::{Expression, FlakeValue}; +use crate::ir::{Expression, FlakeValue, Function}; use crate::parse::UnresolvedDatatypeConstraint; use crate::var_registry::VarId; use fluree_db_core::ids::DatatypeDictId; @@ -151,108 +151,7 @@ impl Expression { ctx: Option<&ExecutionContext<'_>>, ) -> Result> { match self { - Expression::Var(var) => match row.get(*var) { - Some(Binding::Lit { val, dtc, .. }) => Ok(lit_to_comparable(val, dtc, ctx)), - Some(Binding::EncodedLit { - o_kind, - o_key, - p_id, - dt_id, - lang_id, - .. - }) => { - let Some(decoded) = ctx.and_then(|c| { - c.decode_encoded_value(*o_kind, *o_key, *p_id, *dt_id, *lang_id) - }) else { - return Ok(None); - }; - let val = decoded.map_err(|e| { - decode_lookup_error( - "decode encoded literal", - format!( - "o_kind={o_kind}, o_key={o_key}, p_id={p_id}, dt_id={dt_id}, lang_id={lang_id}" - ), - e, - ) - })?; - // xsd:float is folded to `FlakeValue::Double` at decode (the - // NUM_F64 fast path in `context.rs`), dropping the float tag - // the Lit path keeps via `lit_to_comparable`. Re-tag it from - // the in-scope `dt_id` so `datatype(?f + ?f)` stays xsd:float - // on the late-materialized (`EncodedLit`) path — one integer - // compare on the hot decode arm (#1470). - if *dt_id == DatatypeDictId::FLOAT.as_u16() { - if let FlakeValue::Double(d) = val { - return Ok(Some(ComparableValue::Float(d as f32))); - } - } - // A stored language-tagged literal decodes to a bare string - // (`FlakeValue::String` cannot carry the tag), so `=`/`!=`/ - // `IN` were tag-blind exactly on the production-typical - // indexed path while the Lit path compares tag-aware - // (#1468). Re-tag from the in-scope `lang_id` — symmetric - // to the FLOAT re-tag above and to the `lang_id` check in - // `binding_effective_bool`; one integer compare on the hot - // arm, the meta decode only runs for lang-tagged rows. - if *lang_id != 0 && matches!(&val, FlakeValue::String(_)) { - return match ctx.and_then(|c| c.lang_tag_for_id(*lang_id)) { - Some(tag) => Ok(Some(ComparableValue::TypedLiteral { - val, - dtc: Some(crate::parse::UnresolvedDatatypeConstraint::LangTag(tag)), - })), - // An UNRESOLVABLE nonzero lang_id (an - // overlay-ephemeral id the persisted store can't - // see — unreachable through today's scan paths, - // pinned by the post-index-novelty test) must - // surface as an unknown value, never degrade to a - // tag-blind bare string (the exact silent-equality - // bug this arm exists to fix). - None => Ok(None), - }; - } - Ok(ComparableValue::try_from(&val).ok()) - } - Some(Binding::Sid { sid, .. }) => Ok(Some(ComparableValue::Sid(sid.clone()))), - Some(Binding::IriMatch { iri, .. }) => { - Ok(Some(ComparableValue::Iri(Arc::clone(iri)))) - } - Some(Binding::Iri(iri)) => Ok(Some(ComparableValue::Iri(Arc::clone(iri)))), - Some(Binding::EncodedSid { s_id, .. }) => { - let Some(resolved) = ctx.and_then(|c| c.resolve_subject_iri(*s_id)) else { - return Ok(None); - }; - match resolved { - Ok(iri) => Ok(Some(ComparableValue::Iri(Arc::from(iri)))), - Err(e) => Err(decode_lookup_error( - "resolve subject IRI", - format!("s_id={s_id}"), - e, - )), - } - } - Some(Binding::EncodedPid { p_id }) => { - let Some(store) = ctx.and_then(|c| c.binary_store.as_deref()) else { - return Ok(None); - }; - match store.resolve_predicate_iri(*p_id) { - Some(iri) => Ok(Some(ComparableValue::Iri(Arc::from(iri)))), - None => Err(QueryError::dictionary_lookup(format!( - "resolve predicate IRI: unknown p_id={p_id}" - ))), - } - } - Some(Binding::Unbound | Binding::Poisoned) | None => Ok(None), - Some(Binding::Grouped(_)) => { - debug_assert!(false, "Grouped binding in filter evaluation"); - Ok(None) - } - // A path or list is not a scalar — no comparable value. The - // relevant functions (`length`, `size`/`head`/…) read the - // binding directly via dispatch / the binding-producing path. - Some( - Binding::Path { .. } | Binding::Rel(_) | Binding::List(_) | Binding::Map(_), - ) => Ok(None), - }, + Expression::Var(var) => binding_to_comparable(row.get(*var), ctx), // FlakeValue::Null is the only variant TryFrom rejects (with // NullValueError); a constant Null evaluates to "no value". @@ -267,8 +166,9 @@ impl Expression { | Expression::Reduce { .. } | Expression::PatternComprehension { .. } => Ok(None), - // A resolved value (pattern-comprehension list) — its comparable form. - Expression::Resolved(b) => Ok(list::element_to_comparable(b)), + // A resolved value: its comparable form, decoded exactly as a + // bound variable is, so a literal keeps its datatype or tag. + Expression::Resolved(b) => binding_to_comparable(Some(b.as_ref()), ctx), // A list predicate is a boolean scalar. Expression::ListPredicate { @@ -392,6 +292,19 @@ impl Expression { return Ok((**b).clone()); } + // A term accessor on a late-materialized handle binds the component + // encoded, without materializing the term. + if let Expression::Call { + func: + func @ (Function::TripleSubject | Function::TriplePredicate | Function::TripleObject), + args, + } = self + { + if let Some(binding) = rdf::encoded_term_component(func, args, row, ctx)? { + return Ok(binding); + } + } + // Scoped list-iteration and eval-time member access produce structured // values directly (a List / the accumulator / a looked-up value). match self { @@ -527,6 +440,115 @@ pub(crate) fn stored_float_f64( /// xsd:double and xsd:string/plain-string paths yield the same `ComparableValue` /// but each pay one cheap datatype check (float-vs-double, resp. /// xsd:string-vs-foreign/lang). Foreign string literals are rare (BSBM has none). +/// The comparable form of a binding, decoding an encoded one through `ctx`. +/// Every expression surface that reads a binding goes through here so a +/// variable and a resolved constant compare the same way. +pub(crate) fn binding_to_comparable( + binding: Option<&Binding>, + ctx: Option<&ExecutionContext<'_>>, +) -> Result> { + match binding { + Some(Binding::Lit { val, dtc, .. }) => Ok(lit_to_comparable(val, dtc, ctx)), + Some(Binding::EncodedLit { + o_kind, + o_key, + p_id, + dt_id, + lang_id, + .. + }) => { + let Some(decoded) = + ctx.and_then(|c| c.decode_encoded_value(*o_kind, *o_key, *p_id, *dt_id, *lang_id)) + else { + return Ok(None); + }; + let val = decoded.map_err(|e| { + decode_lookup_error( + "decode encoded literal", + format!( + "o_kind={o_kind}, o_key={o_key}, p_id={p_id}, dt_id={dt_id}, lang_id={lang_id}" + ), + e, + ) + })?; + // xsd:float is folded to `FlakeValue::Double` at decode (the + // NUM_F64 fast path in `context.rs`), dropping the float tag + // the Lit path keeps via `lit_to_comparable`. Re-tag it from + // the in-scope `dt_id` so `datatype(?f + ?f)` stays xsd:float + // on the late-materialized (`EncodedLit`) path — one integer + // compare on the hot decode arm (#1470). + if *dt_id == DatatypeDictId::FLOAT.as_u16() { + if let FlakeValue::Double(d) = val { + return Ok(Some(ComparableValue::Float(d as f32))); + } + } + // A stored language-tagged literal decodes to a bare string + // (`FlakeValue::String` cannot carry the tag), so `=`/`!=`/ + // `IN` were tag-blind exactly on the production-typical + // indexed path while the Lit path compares tag-aware + // (#1468). Re-tag from the in-scope `lang_id` — symmetric + // to the FLOAT re-tag above and to the `lang_id` check in + // `binding_effective_bool`; one integer compare on the hot + // arm, the meta decode only runs for lang-tagged rows. + if *lang_id != 0 && matches!(&val, FlakeValue::String(_)) { + return match ctx.and_then(|c| c.lang_tag_for_id(*lang_id)) { + Some(tag) => Ok(Some(ComparableValue::TypedLiteral { + val, + dtc: Some(crate::parse::UnresolvedDatatypeConstraint::LangTag(tag)), + })), + // An UNRESOLVABLE nonzero lang_id (an + // overlay-ephemeral id the persisted store can't + // see — unreachable through today's scan paths, + // pinned by the post-index-novelty test) must + // surface as an unknown value, never degrade to a + // tag-blind bare string (the exact silent-equality + // bug this arm exists to fix). + None => Ok(None), + }; + } + Ok(ComparableValue::try_from(&val).ok()) + } + Some(Binding::Sid { sid, .. }) => Ok(Some(ComparableValue::Sid(sid.clone()))), + Some(Binding::IriMatch { iri, .. }) => Ok(Some(ComparableValue::Iri(Arc::clone(iri)))), + Some(Binding::Iri(iri)) => Ok(Some(ComparableValue::Iri(Arc::clone(iri)))), + Some(Binding::EncodedSid { s_id, .. }) => { + let Some(resolved) = ctx.and_then(|c| c.resolve_subject_iri(*s_id)) else { + return Ok(None); + }; + match resolved { + Ok(iri) => Ok(Some(ComparableValue::Iri(Arc::from(iri)))), + Err(e) => Err(decode_lookup_error( + "resolve subject IRI", + format!("s_id={s_id}"), + e, + )), + } + } + Some(Binding::EncodedPid { p_id }) => { + let Some(store) = ctx.and_then(|c| c.binary_store.as_deref()) else { + return Ok(None); + }; + match store.resolve_predicate_iri(*p_id) { + Some(iri) => Ok(Some(ComparableValue::Iri(Arc::from(iri)))), + None => Err(QueryError::dictionary_lookup(format!( + "resolve predicate IRI: unknown p_id={p_id}" + ))), + } + } + Some(Binding::Unbound | Binding::Poisoned) | None => Ok(None), + Some(Binding::Grouped(_)) => { + debug_assert!(false, "Grouped binding in filter evaluation"); + Ok(None) + } + // A path or list is not a scalar — no comparable value. The + // relevant functions (`length`, `size`/`head`/…) read the + // binding directly via dispatch / the binding-producing path. + Some(Binding::Path { .. } | Binding::Rel(_) | Binding::List(_) | Binding::Map(_)) => { + Ok(None) + } + } +} + fn lit_to_comparable( val: &FlakeValue, dtc: &DatatypeConstraint, diff --git a/fluree-db-query/src/eval/rdf.rs b/fluree-db-query/src/eval/rdf.rs index 77e98e883c..2bb9a043e4 100644 --- a/fluree-db-query/src/eval/rdf.rs +++ b/fluree-db-query/src/eval/rdf.rs @@ -5,7 +5,8 @@ use crate::binding::{Binding, RowAccess}; use crate::context::ExecutionContext; use crate::error::{QueryError, Result}; -use crate::ir::Expression; +use crate::ir::{Expression, Function}; +use crate::object_binding::{late_materialized_object_binding, materialized_object_binding}; use fluree_db_binary_index::BinaryIndexStore; use fluree_db_core::{DatatypeDictId, Sid}; use std::sync::Arc; @@ -467,11 +468,135 @@ fn triple_term_arg( }) } +/// The encoded base edge behind an argument bound to a late-materialized +/// triple-term handle, with the link's `t` and the context to read it +/// through. One forward-dictionary lookup, where materializing the term +/// costs three dictionary reads and a boxed value per accessor per row. +/// `None` for any other argument, including a materialized term, which +/// takes the value path. +fn encoded_term<'c, R: RowAccess>( + args: &[Expression], + row: &R, + ctx: Option<&'c ExecutionContext<'_>>, +) -> Result< + Option<( + fluree_db_core::triple_term::TermKey, + i64, + &'c ExecutionContext<'c>, + )>, +> { + let [Expression::Var(v)] = args else { + return Ok(None); + }; + let Some(Binding::EncodedLit { + o_kind, o_key, t, .. + }) = row.get(*v) + else { + return Ok(None); + }; + if *o_kind != fluree_db_core::value_id::ObjKind::TRIPLE_TERM.as_u8() { + return Ok(None); + } + let Some(ctx) = ctx else { + return Ok(None); + }; + let Some(store) = ctx.binary_store.as_deref() else { + return Ok(None); + }; + let key = store + .resolve_term_key(*o_key) + .map_err(|e| QueryError::from_io("resolve_term_key", e))? + .ok_or_else(|| { + QueryError::Internal(format!( + "triple-term handle {o_key:#x} has no dictionary entry" + )) + })?; + Ok(Some((key, *t, ctx))) +} + +/// The base edge's object as a binding, in the encoded form a scan would +/// bind it when the kind allows, else materialized with its datatype or +/// language tag. +fn term_object_binding( + key: &fluree_db_core::triple_term::TermKey, + t: i64, + ctx: &ExecutionContext<'_>, +) -> Result { + let o_type = key.o_type.as_u16(); + if let Some(b) = + late_materialized_object_binding(o_type, key.o_key, key.p_id, t, u32::MAX, None) + { + return Ok(b); + } + let store = ctx + .binary_store + .as_deref() + .ok_or_else(|| QueryError::Internal("term object decode without a store".into()))?; + let val = store + .decode_value_v3(o_type, key.o_key, key.p_id, ctx.binary_g_id) + .map_err(|e| QueryError::from_io("decode_value_v3", e))?; + Ok(materialized_object_binding( + store, + o_type, + key.p_id, + val, + Some(t), + None, + )) +} + +/// `BIND(SUBJECT|PREDICATE|OBJECT(?term) AS ?v)` on an encoded handle: the +/// component in the encoded form a scan would bind it, so the BIND is one +/// dictionary lookup and no materialization. `None` takes the value path. +pub(crate) fn encoded_term_component( + func: &Function, + args: &[Expression], + row: &R, + ctx: Option<&ExecutionContext<'_>>, +) -> Result> { + let Some((key, t, ctx)) = encoded_term(args, row, ctx)? else { + return Ok(None); + }; + let binding = match func { + Function::TripleSubject => Binding::encoded_sid(key.s_id), + Function::TriplePredicate => Binding::EncodedPid { p_id: key.p_id }, + Function::TripleObject => term_object_binding(&key, t, ctx)?, + _ => return Ok(None), + }; + Ok(Some(binding)) +} + +/// A materialized term's object with its datatype or language tag, which +/// `OBJECT()` must keep: `"chat"@fr` is not `"chat"`. +fn term_object_comparable( + term: &fluree_db_core::TripleTermValue, + ctx: Option<&ExecutionContext<'_>>, +) -> Option { + if let fluree_db_core::FlakeValue::Ref(sid) = &term.o { + return Some(ComparableValue::Sid(sid.clone())); + } + let dtc = match &term.lang { + Some(lang) => fluree_db_core::DatatypeConstraint::LangTag(Arc::from(lang.as_str())), + None => fluree_db_core::DatatypeConstraint::Explicit(term.dt.clone()), + }; + super::lit_to_comparable(&term.o, &dtc, ctx) +} + pub fn eval_triple_subject( args: &[Expression], row: &R, ctx: Option<&ExecutionContext<'_>>, ) -> Result> { + if let Some((key, _, ctx)) = encoded_term(args, row, ctx)? { + let store = ctx + .binary_store + .as_deref() + .expect("encoded term has a store"); + let iri = store + .resolve_subject_iri(key.s_id) + .map_err(|e| QueryError::from_io("resolve_subject_iri", e))?; + return Ok(Some(ComparableValue::Sid(store.encode_iri(&iri)))); + } Ok(triple_term_arg(args, row, ctx, "SUBJECT")?.map(|t| ComparableValue::Sid(t.s))) } @@ -480,6 +605,20 @@ pub fn eval_triple_predicate( row: &R, ctx: Option<&ExecutionContext<'_>>, ) -> Result> { + if let Some((key, _, ctx)) = encoded_term(args, row, ctx)? { + let store = ctx + .binary_store + .as_deref() + .expect("encoded term has a store"); + let sid = store + .p_sid_table() + .get(key.p_id as usize) + .cloned() + .ok_or_else(|| { + QueryError::Internal(format!("triple-term predicate {} is not known", key.p_id)) + })?; + return Ok(Some(ComparableValue::Sid(sid))); + } Ok(triple_term_arg(args, row, ctx, "PREDICATE")?.map(|t| ComparableValue::Sid(t.p))) } @@ -488,8 +627,11 @@ pub fn eval_triple_object( row: &R, ctx: Option<&ExecutionContext<'_>>, ) -> Result> { - Ok(triple_term_arg(args, row, ctx, "OBJECT")? - .and_then(|t| ComparableValue::try_from(&t.o).ok())) + if let Some((key, t, ctx)) = encoded_term(args, row, ctx)? { + let object = term_object_binding(&key, t, ctx)?; + return super::binding_to_comparable(Some(&object), Some(ctx)); + } + Ok(triple_term_arg(args, row, ctx, "OBJECT")?.and_then(|t| term_object_comparable(&t, ctx))) } pub fn eval_is_triple( diff --git a/fluree-db-query/src/execute.rs b/fluree-db-query/src/execute.rs index ec5c70e8a1..72fc992869 100644 --- a/fluree-db-query/src/execute.rs +++ b/fluree-db-query/src/execute.rs @@ -350,11 +350,15 @@ mod tests { lower: Some((FlakeValue::Long(10), false)), upper: Some((FlakeValue::Long(100), true)), term_predicate: None, + term_subject: None, + term_object: None, }; let b = ObjectBounds { lower: Some((FlakeValue::Long(20), true)), upper: Some((FlakeValue::Long(80), false)), term_predicate: None, + term_subject: None, + term_object: None, }; let merged = merge_object_bounds(&a, &b); diff --git a/fluree-db-query/src/execute/pushdown.rs b/fluree-db-query/src/execute/pushdown.rs index 13ebcfa552..0d18aea9bd 100644 --- a/fluree-db-query/src/execute/pushdown.rs +++ b/fluree-db-query/src/execute/pushdown.rs @@ -31,6 +31,12 @@ pub fn extract_bounds_from_filters( let mut filter_vars_matched = 0; for &var in &object_vars { if let Some(new_bounds) = extract_object_bounds_for_var(expr, var) { + if bounds + .get(&var) + .is_some_and(|existing| term_predicates_conflict(existing, &new_bounds)) + { + continue; + } filter_vars_matched += 1; bounds .entry(var) @@ -90,6 +96,12 @@ pub fn extract_lookahead_bounds_with_consumption( // Try to extract bounds for each object variable for &var in &object_vars { if let Some(new_bounds) = extract_object_bounds_for_var(expr, var) { + if bounds + .get(&var) + .is_some_and(|existing| term_predicates_conflict(existing, &new_bounds)) + { + continue; + } filter_vars_matched += 1; // Merge with existing bounds for this var (intersection) bounds @@ -127,6 +139,19 @@ pub fn count_filter_vars(expr: &Expression) -> usize { vars.len() } +/// Two constraints on the same component of one term variable that name +/// different terms. Nothing satisfies both, and `merge_object_bounds` keeps +/// only one, so the second filter must stay in the plan (and empties it) +/// rather than be consumed. +fn term_predicates_conflict(a: &ObjectBounds, b: &ObjectBounds) -> bool { + fn differ(x: &Option, y: &Option) -> bool { + matches!((x, y), (Some(x), Some(y)) if x != y) + } + differ(&a.term_predicate, &b.term_predicate) + || differ(&a.term_subject, &b.term_subject) + || differ(&a.term_object, &b.term_object) +} + /// Merge two ObjectBounds, taking the tighter constraint for each bound /// /// For lower bounds, takes the higher value (more restrictive). @@ -140,6 +165,8 @@ pub fn merge_object_bounds(a: &ObjectBounds, b: &ObjectBounds) -> ObjectBounds { .term_predicate .clone() .or_else(|| b.term_predicate.clone()), + term_subject: a.term_subject.clone().or_else(|| b.term_subject.clone()), + term_object: a.term_object.clone().or_else(|| b.term_object.clone()), } } @@ -303,11 +330,15 @@ mod tests { lower: Some((FlakeValue::Long(10), false)), upper: Some((FlakeValue::Long(100), true)), term_predicate: None, + term_subject: None, + term_object: None, }; let b = ObjectBounds { lower: Some((FlakeValue::Long(20), true)), upper: Some((FlakeValue::Long(80), false)), term_predicate: None, + term_subject: None, + term_object: None, }; let merged = merge_object_bounds(&a, &b); diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index f379e78134..e4d2a0a8b5 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -2311,6 +2311,51 @@ fn values_cell_as_ref_term(binding: &crate::binding::Binding) -> Option { } } +/// Drop `BIND(SUBJECT|PREDICATE|OBJECT(?term) AS ?v)` when nothing reads +/// `?v`: no other pattern, the seed, the post-WHERE pipeline or the +/// projection. A variable the seed binds keeps its BIND, which is then the +/// equality check on that position. The `f:reifies*` twin of this rule is +/// `elide_redundant_chain`. +fn elide_unread_term_binds( + patterns: &[Pattern], + needed_vars: &HashSet, + required_where_vars: Option<&[VarId]>, + seed_schema: &HashSet, +) -> Option> { + fn is_term_accessor(expr: &Expression) -> bool { + matches!( + expr, + Expression::Call { + func: Function::TripleSubject | Function::TriplePredicate | Function::TripleObject, + args, + } if matches!(args.as_slice(), [Expression::Var(_)]) + ) + } + if !patterns + .iter() + .any(|p| matches!(p, Pattern::Bind { expr, .. } if is_term_accessor(expr))) + { + return None; + } + let mut counts: HashMap = HashMap::new(); + let mut all_vars: HashSet = HashSet::new(); + collect_var_stats(patterns, &mut counts, &mut all_vars); + let unread = |var: VarId| { + counts.get(&var).copied().unwrap_or(0) <= 1 + && !needed_vars.contains(&var) + && !required_where_vars.is_some_and(|r| r.contains(&var)) + && !seed_schema.contains(&var) + }; + let kept: Vec = patterns + .iter() + .filter( + |p| !matches!(p, Pattern::Bind { var, expr } if is_term_accessor(expr) && unread(*var)), + ) + .cloned() + .collect(); + (kept.len() != patterns.len()).then_some(kept) +} + /// Drop VALUES columns that are UNDEF in every row, and a VALUES left with no /// columns and exactly one row. /// @@ -2591,13 +2636,24 @@ pub fn build_where_operators_seeded_with_needed( let inlined_storage = inline_singleton_values_objects(patterns); let patterns: &[Pattern] = inlined_storage.as_deref().unwrap_or(patterns); + let seed_vars = seed.as_deref().map(|op| seed_vars(op)).unwrap_or_default(); + + // A reified-edge position nobody reads costs a dictionary lookup per row + // and removes none; the link lowering binds every variable position. + let term_bind_storage = elide_unread_term_binds( + patterns, + needed_vars, + required_where_vars, + &seed_vars.schema, + ); + let patterns: &[Pattern] = term_bind_storage.as_deref().unwrap_or(patterns); + // Apply generalized pattern reordering upfront for all pattern lists. // // reorder_patterns determines optimal placement of all patterns // (triples, compound patterns like UNION/OPTIONAL/MINUS/EXISTS/Subquery) // using selectivity-based cost estimation. This subsumes the per-block // reorder_patterns_seeded calls that previously handled triple-only blocks. - let seed_vars = seed.as_deref().map(|op| seed_vars(op)).unwrap_or_default(); let reordered_storage = reorder_patterns_with_seed(patterns, stats.as_deref(), &seed_vars); let patterns = &reordered_storage; @@ -5449,6 +5505,8 @@ mod tests { lower: Some((FlakeValue::String("2026-01-01".to_string()), true)), upper: None, term_predicate: None, + term_subject: None, + term_object: None, }, ); diff --git a/fluree-db-query/src/object_binding.rs b/fluree-db-query/src/object_binding.rs index 99de85dc05..48fa28039b 100644 --- a/fluree-db-query/src/object_binding.rs +++ b/fluree-db-query/src/object_binding.rs @@ -448,6 +448,47 @@ pub(crate) fn normalize_for_key_cow<'a>( } } +/// Whether a BIND's computed value agrees with the value the row already +/// holds for its variable. Plain equality is representation-bound: a scan +/// binds `EncodedSid`, a VALUES row or a decoded term binds `Sid`, and the +/// two never compare equal. Normalize both sides to their dictionary form +/// first, as the join surfaces do, and compare a predicate id against a +/// subject id through the shared SID. +pub(crate) fn bind_unifies( + existing: &Binding, + computed: &Binding, + ctx: Option<&crate::context::ExecutionContext<'_>>, +) -> bool { + if existing == computed { + return true; + } + let Some(ctx) = ctx else { + return false; + }; + if ctx.is_multi_ledger() { + return false; + } + let Some(store) = ctx.binary_store.as_deref() else { + return false; + }; + let as_sid = |b: &Binding| -> Option { + match b { + Binding::EncodedPid { p_id } => store + .p_sid_table() + .get(*p_id as usize) + .cloned() + .map(Binding::sid), + _ => None, + } + }; + let a = as_sid(existing); + let b = as_sid(computed); + let a = a.as_ref().unwrap_or(existing); + let b = b.as_ref().unwrap_or(computed); + normalize_for_key_cow(a, Some(store), None).as_ref() + == normalize_for_key_cow(b, Some(store), None).as_ref() +} + /// True if this is an arena-backed (NUM_BIG) encoded literal. pub(crate) fn is_numbig_encoded(binding: &Binding) -> bool { matches!( diff --git a/fluree-db-query/src/operator/inline.rs b/fluree-db-query/src/operator/inline.rs index 9ea1e278dd..de77be0fe5 100644 --- a/fluree-db-query/src/operator/inline.rs +++ b/fluree-db-query/src/operator/inline.rs @@ -84,7 +84,7 @@ pub fn apply_inline( Some(pos) => match (&bindings[pos], &value) { (Binding::Unbound, _) => bindings[pos] = value, (_, Binding::Unbound) => continue, - (a, b) if a == b => continue, + (a, b) if crate::object_binding::bind_unifies(a, b, ctx) => continue, _ => return Ok(false), }, } diff --git a/fluree-db-query/src/planner.rs b/fluree-db-query/src/planner.rs index 2fab24969e..b0a011178f 100644 --- a/fluree-db-query/src/planner.rs +++ b/fluree-db-query/src/planner.rs @@ -859,8 +859,8 @@ pub fn extract_object_bounds_for_var( filter: &Expression, object_var: VarId, ) -> Option { - if let Some(sid) = term_predicate_constraint(filter, object_var) { - return Some(ObjectBounds::term_predicate(sid)); + if let Some(bounds) = term_component_constraint(filter, object_var) { + return Some(bounds); } // Only proceed if filter is range-safe let constraints = extract_range_constraints(filter)?; @@ -883,37 +883,63 @@ pub fn extract_object_bounds_for_var( // Generalized Selectivity Scoring for All Pattern Types // ============================================================================= -/// `PREDICATE(?t) =

` (either operand order) on the object variable of a -/// triple-term scan: the one filter shape the scan can enforce as a handle -/// interval, since handles are partitioned by inner predicate. -fn term_predicate_constraint( - filter: &Expression, - object_var: VarId, -) -> Option { - let Expression::Call { - func: Function::Eq, - args, - } = filter - else { +/// A constant on one component of a triple-term variable — `PREDICATE(?t) = +///

`, `sameTerm(SUBJECT(?t), )`, `sameTerm(OBJECT(?t), "v"^^dt)` — +/// in either operand order, as bounds the scan enforces on the handle: the +/// predicate as one handle interval, the subject and object against the +/// handle's dictionary key. An object literal must come as a resolved +/// binding under `sameTerm`, which carries its datatype or tag; a plain +/// `=` on a literal is value equality and stays a filter. +fn term_component_constraint(filter: &Expression, object_var: VarId) -> Option { + let Expression::Call { func, args } = filter else { return None; }; + let same_term = match func { + Function::SameTerm => true, + Function::Eq => false, + _ => return None, + }; if args.len() != 2 { return None; } - let is_pred_of = |e: &Expression| { - matches!(e, Expression::Call { func: Function::TriplePredicate, args } - if args.len() == 1 && args[0] == Expression::Var(object_var)) - }; - let const_sid = |e: &Expression| match e { - Expression::Const(FlakeValue::Ref(sid)) => Some(sid.clone()), + let accessor = |e: &Expression| match e { + Expression::Call { func, args } + if args.len() == 1 && args[0] == Expression::Var(object_var) => + { + match func { + Function::TripleSubject | Function::TriplePredicate | Function::TripleObject => { + Some(func.clone()) + } + _ => None, + } + } _ => None, }; - if is_pred_of(&args[0]) { - const_sid(&args[1]) - } else if is_pred_of(&args[1]) { - const_sid(&args[0]) - } else { - None + let (func, constant) = accessor(&args[0]) + .map(|f| (f, &args[1])) + .or_else(|| accessor(&args[1]).map(|f| (f, &args[0])))?; + match (func, constant) { + (Function::TriplePredicate, Expression::Const(FlakeValue::Ref(sid))) => { + Some(ObjectBounds::term_predicate(sid.clone())) + } + (Function::TripleSubject, Expression::Const(FlakeValue::Ref(sid))) => { + Some(ObjectBounds::term_subject(sid.clone())) + } + (Function::TripleObject, Expression::Const(FlakeValue::Ref(sid))) => { + Some(ObjectBounds::term_object( + FlakeValue::Ref(sid.clone()), + fluree_db_core::DatatypeConstraint::Explicit( + fluree_db_core::edge::id_datatype_sid(), + ), + )) + } + (Function::TripleObject, Expression::Resolved(b)) if same_term => match b.as_ref() { + crate::binding::Binding::Lit { val, dtc, .. } => { + Some(ObjectBounds::term_object(val.clone(), dtc.clone())) + } + _ => None, + }, + _ => None, } } diff --git a/fluree-db-query/src/range_semijoin.rs b/fluree-db-query/src/range_semijoin.rs index bd766eff30..d12b28feed 100644 --- a/fluree-db-query/src/range_semijoin.rs +++ b/fluree-db-query/src/range_semijoin.rs @@ -664,6 +664,8 @@ impl RangeSemiJoinOperator { lower: envelope.lower.clone().map(|v| (v, true)), upper: envelope.upper.clone().map(|v| (v, true)), term_predicate: None, + term_subject: None, + term_object: None, }; let narrow = ColumnSet::single(ColumnId::SId).union(ColumnSet::single(ColumnId::OKey)); let mixed = narrow diff --git a/fluree-db-sparql/src/lower/annotation.rs b/fluree-db-sparql/src/lower/annotation.rs index 7d82538f93..aa95773480 100644 --- a/fluree-db-sparql/src/lower/annotation.rs +++ b/fluree-db-sparql/src/lower/annotation.rs @@ -282,9 +282,13 @@ impl LoweringContext<'_, E> { /// `annotation rdf:reifies ?__term` triple whose object is a triple-term /// handle, then each component of the reified edge either bound from the /// term (`BIND(SUBJECT(?__term) AS ?s)`) or constrained against it - /// (`FILTER(PREDICATE(?__term) =

)`). A predicate constraint is what the - /// planner turns into one handle interval; a fully constant edge composes - /// to a constant term the scan looks up directly. + /// (`FILTER(sameTerm(PREDICATE(?__term),

))`). A variable is always a + /// `BIND`: when the variable is already bound, `BIND` keeps the row only + /// if the values agree, which is the join, and that holds in every scope + /// (a UNION branch, an OPTIONAL, a subquery seed) without the lowering + /// tracking who binds what. A predicate constraint is what the planner + /// turns into one handle interval; a fully constant edge composes to a + /// constant term the scan looks up directly. pub(super) fn lower_reified_link( &mut self, annotation_ref: Ref, @@ -292,6 +296,7 @@ impl LoweringContext<'_, E> { out: &mut Vec, ) { use fluree_db_core::FlakeValue; + use fluree_db_query::binding::Binding; use fluree_db_query::ir::{Expression, Function}; let reifies = self.encoder.encode_ref(fluree_vocab::rdf::REIFIES); @@ -316,46 +321,64 @@ impl LoweringContext<'_, E> { o: IrTerm::Var(t), dtc: None, })); - self.link_bound_vars.insert(t); let accessor = |f: Function| Expression::call(f, vec![Expression::Var(t)]); - let mut component = |func: Function, term: IrTerm, out: &mut Vec| match term { - IrTerm::Var(v) => { - if self.link_bound_vars.insert(v) { - out.push(Pattern::Bind { - var: v, - expr: accessor(func), - }); - } else { - out.push(Pattern::Filter(Expression::call( - Function::Eq, - vec![accessor(func), Expression::Var(v)], - ))); - } - } - IrTerm::Sid(sid) => out.push(Pattern::Filter(Expression::call( - Function::Eq, - vec![accessor(func), Expression::Const(FlakeValue::Ref(sid))], - ))), + let same_term = |f: Function, constant: Expression| { + Pattern::Filter(Expression::call( + Function::SameTerm, + vec![accessor(f), constant], + )) + }; + let IrTriplePattern { s, p, o, dtc } = edge; + let component = |func: Function, term: IrTerm, out: &mut Vec| match term { + IrTerm::Var(v) => out.push(Pattern::Bind { + var: v, + expr: accessor(func), + }), + IrTerm::Sid(sid) => out.push(same_term(func, Expression::Const(FlakeValue::Ref(sid)))), IrTerm::Iri(iri) => match self.encoder.encode_iri(&iri) { - Some(sid) => out.push(Pattern::Filter(Expression::call( - Function::Eq, - vec![accessor(func), Expression::Const(FlakeValue::Ref(sid))], - ))), + Some(sid) => out.push(same_term(func, Expression::Const(FlakeValue::Ref(sid)))), // An IRI in no registered namespace names nothing in this // ledger, so the pattern cannot match. None => out.push(Pattern::Filter(Expression::Const(FlakeValue::Boolean( false, )))), }, - IrTerm::Value(v) => out.push(Pattern::Filter(Expression::call( - Function::Eq, - vec![accessor(func), Expression::Const(v)], - ))), + // A literal matches by term identity: its datatype or language + // tag is part of what `<< ?s :p "chat"@fr >>` asks for. + IrTerm::Value(v) => match &dtc { + Some(dtc) => out.push(same_term( + func, + Expression::Resolved(Box::new(Binding::Lit { + val: v, + dtc: dtc.clone(), + t: None, + op: None, + p_id: None, + })), + )), + None => out.push(Pattern::Filter(Expression::call( + Function::Eq, + vec![accessor(func), Expression::Const(v)], + ))), + }, }; - component(Function::TripleSubject, edge.s.into(), out); - component(Function::TriplePredicate, edge.p.into(), out); - component(Function::TripleObject, edge.o, out); + // Constraints first, so the planner's filter lookahead over the link + // triple reaches every one of them before a BIND intervenes. + let mut binds = Vec::new(); + for (func, term) in [ + (Function::TripleSubject, IrTerm::from(s)), + (Function::TriplePredicate, IrTerm::from(p)), + (Function::TripleObject, o), + ] { + match term { + IrTerm::Var(_) => binds.push((func, term)), + constant => component(func, constant, out), + } + } + for (func, term) in binds { + component(func, term, out); + } } } diff --git a/fluree-db-sparql/src/lower/mod.rs b/fluree-db-sparql/src/lower/mod.rs index 6f6e3022af..c322e32cff 100644 --- a/fluree-db-sparql/src/lower/mod.rs +++ b/fluree-db-sparql/src/lower/mod.rs @@ -403,10 +403,6 @@ struct LoweringContext<'a, E> { /// Monotonic counter for the triple-term variables the link lowering /// mints (`?__term_0`, …). term_counter: u32, - /// Variables some earlier pattern binds. The link lowering binds a term's - /// components with `BIND` only for variables not in this set, and joins - /// with a `FILTER` equality otherwise, since `BIND` overwrites. - link_bound_vars: std::collections::HashSet, /// Original SPARQL source text (for extracting SERVICE body text). source_text: Option<&'a str>, } @@ -432,7 +428,6 @@ impl<'a, E: IriEncoder> LoweringContext<'a, E> { pp_counter: 0, order_counter: 0, term_counter: 0, - link_bound_vars: std::collections::HashSet::new(), source_text, } } diff --git a/fluree-db-sparql/src/lower/rdf_star.rs b/fluree-db-sparql/src/lower/rdf_star.rs index a42ca580d4..ef9945838a 100644 --- a/fluree-db-sparql/src/lower/rdf_star.rs +++ b/fluree-db-sparql/src/lower/rdf_star.rs @@ -109,15 +109,7 @@ impl LoweringContext<'_, E> { match &tp.annotation { Some(ann) => self.lower_annotation_units(edge, ann, &mut result)?, - None => { - for v in [edge.s.as_var(), edge.p.as_var(), edge.o.as_var()] - .into_iter() - .flatten() - { - self.link_bound_vars.insert(v); - } - result.push(Pattern::Triple(edge)); - } + None => result.push(Pattern::Triple(edge)), } } diff --git a/fluree-db-transact/src/import_sink.rs b/fluree-db-transact/src/import_sink.rs index ea510d5862..eb166c4da2 100644 --- a/fluree-db-transact/src/import_sink.rs +++ b/fluree-db-transact/src/import_sink.rs @@ -599,6 +599,14 @@ mod inner { let Some((o_kind, o_key)) = self.resolve_object_value(&object.o, p_id) else { return Ok(()); }; + // Per-(graph, predicate) arena handles are not a graph-independent + // object identity; rebuild skips these bundles too. + if matches!( + ObjKind::from_u8(o_kind), + ObjKind::NUM_BIG | ObjKind::VECTOR_ID + ) { + return Ok(()); + } let lang_id = object .m .as_ref() From 270984e19ddd09a59934aedf78ec94e7c8b53c7b Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 14:37:12 -0400 Subject: [PATCH 07/92] fix(index): replay attachment ops per reifier instead of assembling bundles per commit A re-point that changes one slot writes only that slot's retract and assert (sync and upsert cancel the unchanged slots), and a full re-point's six ops interleave under the per-commit sort, so an assembler that needed all three slots in one (graph, reifier, t, op) group emitted no link events and the old link survived the build. The resolver now only collects f:reifies* ops (RebuildChunk.attachments) and the build replays them per reifier once ids are global: ops in t order, retracts before asserts within one t, a link retract and assert wherever the attachment changes from or to a complete edge. A rebuild replays the whole history from nothing and writes the links as one more sorted commit file; an incremental build seeds each touched reifier with the attachment the base index holds (one SPOT point lookup) and appends the links to the window. --- fluree-db-api/tests/it_triple_term_links.rs | 89 ++++ fluree-db-indexer/src/build/rebuild.rs | 61 ++- .../run_index/build/incremental_resolve.rs | 155 +++++++ .../src/run_index/resolve/link_synth.rs | 411 ++++++++++++------ .../src/run_index/resolve/resolver.rs | 9 +- 5 files changed, 590 insertions(+), 135 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index c6c6676612..6417860812 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -517,3 +517,92 @@ async fn link_lowering_counts_without_decomposing_unread_positions() { .await; assert_eq!(got, vec![vec!["3".to_string()]], "{got:#?}"); } + +/// A re-point that changes one slot writes only that slot's retract and +/// assert (sync and upsert cancel the unchanged slots). A rebuild replays +/// the reifier's whole history, so the link moves with it. +#[tokio::test] +async fn reindex_follows_a_partial_repoint() { + let alias = "it/triple-term-links:reindex-repoint"; + let (fluree, ledger) = import(&[("claims.ttl", CLAIMS)], alias).await; + fluree + .upsert_turtle( + ledger, + "VERSION \"1.2\"\n@prefix ex: .\nex:carol ex:age 43 ~ ex:claim2 .\n", + ) + .await + .expect("re-pointing the claim's object through upsert"); + fluree + .reindex(alias, fluree_db_api::ReindexOptions::default()) + .await + .expect("reindex after the re-point"); + let ledger = fluree.ledger(alias).await.expect("reload after reindex"); + let got = links(&fluree, &ledger).await; + assert_eq!(got.len(), 4, "{got:#?}"); + let t2 = &got + .iter() + .find(|r| r[0].ends_with("claim2")) + .unwrap_or_else(|| panic!("no link for claim2: {got:#?}"))[1]; + assert!( + t2.contains("carol") && t2.contains("43") && !t2.contains("42"), + "claim2 must reify the re-pointed edge: {t2}" + ); +} + +/// The incremental twin: the base index holds the reifier's attachment, the +/// next window carries only the changed slot, and the link must still move. +#[tokio::test] +async fn incremental_index_follows_a_partial_repoint() { + use fluree_db_indexer::IndexerConfig; + + let fluree = FlureeBuilder::memory() + .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) + .build_memory(); + let ledger_id = "it/triple-term-links:incremental-repoint"; + let (local, handle) = + support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let turtle = + |body: &str| format!("VERSION \"1.2\"\n@prefix ex: .\n{body}\n"); + + local + .run_until(async { + let ledger0 = support::genesis_ledger(&fluree, ledger_id); + let first = fluree + .upsert_turtle( + ledger0, + &turtle("ex:alice ex:age 42 ~ ex:claim1 {| ex:source ex:hr |} ."), + ) + .await + .expect("first claim"); + support::trigger_index_and_wait(&handle, ledger_id, first.receipt.t).await; + support::wait_for_index_application(&fluree, ledger_id, first.receipt.t).await; + let ledger1 = fluree + .ledger(ledger_id) + .await + .expect("reload after first index"); + let got = links(&fluree, &ledger1).await; + assert_eq!(got.len(), 1, "{got:#?}"); + assert!(got[0][1].contains("42"), "{got:#?}"); + + let second = fluree + .upsert_turtle( + ledger1, + &turtle("ex:alice ex:age 43 ~ ex:claim1 {| ex:source ex:hr |} ."), + ) + .await + .expect("re-pointing the claim's object through upsert"); + support::trigger_index_and_wait(&handle, ledger_id, second.receipt.t).await; + support::wait_for_index_application(&fluree, ledger_id, second.receipt.t).await; + let ledger2 = fluree + .ledger(ledger_id) + .await + .expect("reload after incremental index"); + let got = links(&fluree, &ledger2).await; + assert_eq!(got.len(), 1, "one live link after the re-point: {got:#?}"); + assert!( + got[0][1].contains("43") && !got[0][1].contains("42"), + "the link must follow the re-point: {got:#?}" + ); + }) + .await; +} diff --git a/fluree-db-indexer/src/build/rebuild.rs b/fluree-db-indexer/src/build/rebuild.rs index 6d84710a9b..7c621f32e0 100644 --- a/fluree-db-indexer/src/build/rebuild.rs +++ b/fluree-db-indexer/src/build/rebuild.rs @@ -106,6 +106,7 @@ pub async fn rebuild_index_from_commits_with_store( where C: ContentStore + Clone + Send + Sync + 'static, { + use crate::run_index::resolve::link_synth::AttachmentOp; use futures::stream::StreamExt; use run_index::resolver::{RebuildChunk, SharedResolverState}; use run_index::spool::SortedCommitInfo; @@ -446,13 +447,18 @@ where let mut string_dicts = Vec::with_capacity(chunks.len()); let mut chunk_records: Vec> = Vec::with_capacity(chunks.len()); let mut chunk_terms: Vec> = Vec::with_capacity(chunks.len()); + let mut chunk_attachments: Vec> = Vec::with_capacity(chunks.len()); for chunk in chunks { subject_dicts.push(chunk.subjects); string_dicts.push(chunk.strings); chunk_records.push(chunk.records); chunk_terms.push(chunk.terms); + chunk_attachments.push(chunk.attachments); } + // Every chunk's attachment ops, global ids, for one replay after + // the loop: a reifier's ops span chunks. + let mut all_attachments: Vec = Vec::new(); // Triple-term interning happens here, once ids are global: the // registry gives each term entry its `o_type`, the builder its @@ -618,6 +624,13 @@ where fluree_db_core::value_id::ObjKey::encode_u32_id(global_str).as_u64(); } } + let mut attachments = std::mem::take(&mut chunk_attachments[ci]); + for op in &mut attachments { + op.remap(s_remap, str_remap) + .map_err(|e| IndexerError::StorageWrite(format!("chunk {ci}: {e}")))?; + } + all_attachments.append(&mut attachments); + let triple_term = fluree_db_core::value_id::ObjKind::TRIPLE_TERM.as_u8(); for record in records.iter_mut() { if record.o_kind != triple_term { @@ -696,6 +709,52 @@ where }); } + // Link records: replay every reifier's attachment history from + // nothing and write the result as one more sorted commit file. + let links_emitted = if all_attachments.is_empty() { + 0 + } else { + let link_ids = shared.link_synth.link_ids().ok_or_else(|| { + IndexerError::StorageWrite("attachment ops without link ids".into()) + })?; + let mut links: Vec = Vec::new(); + let emitted = crate::run_index::resolve::link_synth::replay_attachments( + &mut all_attachments, + |_, _| [None; 3], + &term_registry, + link_ids, + &mut |key| term_builder.get_or_insert(key), + &mut links, + ) + .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; + drop(all_attachments); + links.sort_unstable_by(fluree_db_binary_index::format::run_record::cmp_g_spot); + let ci = chunk_records.len(); + let fsc_path = commits_dir.join("links.fsc"); + let mut spool_writer = run_index::spool::SpoolWriter::new(&fsc_path, ci) + .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; + for record in &links { + spool_writer + .push(record) + .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; + } + let spool_info = spool_writer + .finish() + .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; + sorted_commit_infos.push(SortedCommitInfo { + path: fsc_path, + record_count: spool_info.record_count, + byte_len: spool_info.byte_len, + chunk_idx: ci, + subject_count: 0, + string_count: 0, + types_map_path: None, + duplicates_removed: 0, + term_table: None, + }); + emitted + }; + // Records are persisted to .fsc files on disk — free the in-memory // copies immediately. For large datasets (e.g. 60M flakes) this // reclaims ~2-6 GB of heap before the index build phase. @@ -1142,7 +1201,7 @@ where tracing::info!( term_count, predicates = refs.forward_packs.len(), - links = shared.link_synth.links_emitted, + links = links_emitted, "triple-term dictionary uploaded" ); Some(refs) diff --git a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs index 37c1b0a5ad..352fd2c2b2 100644 --- a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs +++ b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs @@ -11,6 +11,7 @@ //! 2. Output records: `Vec` + `Vec` (parallel ops) //! 3. `OTypeRegistry` built from root's `datatype_iris` + `language_tags` +use crate::run_index::resolve::link_synth::SlotValue; use std::collections::HashMap; use std::io; use std::path::Path; @@ -671,6 +672,30 @@ pub async fn resolve_incremental_commits_v6( for term in &mut chunk_terms { remap_record(term, &reconcile.subject_remap, &reconcile.string_remap)?; } + // Attachment ops replay per reifier from the attachment the base index + // holds, so a re-point that touched one slot still moves the link. + let mut attachments = chunk.attachments; + for op in &mut attachments { + op.remap(&reconcile.subject_remap, &reconcile.string_remap) + .map_err(|e| IncrementalResolveError::Io(io::Error::other(e)))?; + } + let prior_attachments = if attachments.is_empty() { + HashMap::new() + } else { + let mut keys: Vec<(u16, u64)> = attachments.iter().map(|a| (a.g_id, a.ann)).collect(); + keys.sort_unstable(); + keys.dedup(); + base_attachment_states( + Arc::clone(&cs), + &root, + config.artifact_cache_dir.as_deref(), + &shared.predicates, + shared.link_synth.slots(), + &keys, + ) + .await + .map_err(IncrementalResolveError::Io)? + }; let (new_terms, term_watermarks) = { let base_refs = root.term_dict.as_ref(); let base_reader = match base_refs { @@ -729,6 +754,40 @@ pub async fn resolve_incremental_commits_v6( None => builder.get_or_insert(key)?, }; } + if !attachments.is_empty() { + let link_ids = shared.link_synth.link_ids().ok_or_else(|| { + IncrementalResolveError::Io(io::Error::other("attachment ops without link ids")) + })?; + let mut handle_for = |key: fluree_db_core::triple_term::TermKey| -> io::Result { + if let Some(h) = builder.get(&key) { + return Ok(h); + } + if let Some(reader) = &base_reader { + if let Some(h) = reader.find_handle(&key)? { + return Ok(h); + } + } + builder.get_or_insert(key) + }; + let emitted = crate::run_index::resolve::link_synth::replay_attachments( + &mut attachments, + |g_id, ann| { + prior_attachments + .get(&(g_id, ann)) + .copied() + .unwrap_or([None; 3]) + }, + &o_type_registry, + link_ids, + &mut handle_for, + &mut v1_records, + )?; + tracing::debug!( + links = emitted, + reifiers = prior_attachments.len(), + "V6 incremental resolve: links replayed" + ); + } let new_terms: Vec<(u32, u32, Vec)> = builder .entries_sorted() .into_iter() @@ -1191,6 +1250,102 @@ async fn seed_vector_fact_handles( Ok(()) } +/// The attachment the base index holds for each `(g_id, reifier)`: its live +/// `f:reifies*` rows, read with one SPOT point lookup per reifier. The +/// predicate slot's row refers to the predicate IRI as a subject; it maps +/// back to the predicate id through the dictionary. +async fn base_attachment_states( + cs: Arc, + root: &IndexRoot, + cache_dir: Option<&Path>, + predicates: &crate::run_index::resolve::global_dict::PredicateDict, + slots: [Option; 3], + keys: &[(u16, u64)], +) -> io::Result; 3]>> { + use crate::run_index::resolve::link_synth::ObjectId; + use fluree_db_binary_index::format::run_record::RunSortOrder; + use fluree_db_binary_index::format::run_record_v2::RunRecordV2; + use fluree_db_binary_index::read::binary_cursor::BinaryCursor; + use fluree_db_binary_index::read::binary_index_store::BinaryIndexStore; + use fluree_db_binary_index::read::column_types::{BinaryFilter, ColumnProjection, ColumnSet}; + use fluree_db_core::o_type::OType; + use fluree_db_core::subject_id::SubjectId; + + let mut states = HashMap::new(); + if keys.is_empty() || slots.iter().all(Option::is_none) { + return Ok(states); + } + let cache_dir = cache_dir + .map(Path::to_path_buf) + .unwrap_or_else(std::env::temp_dir); + let store = Arc::new( + BinaryIndexStore::load_from_root_v6(Arc::clone(&cs), root, &cache_dir, None).await?, + ); + let projection = ColumnProjection::for_scan(ColumnSet::EMPTY, false, RunSortOrder::Spot); + for &(g_id, ann) in keys { + let Some(branch) = store.branch_for_order(g_id, RunSortOrder::Spot) else { + continue; + }; + let min_key = RunRecordV2 { + s_id: SubjectId(ann), + o_key: 0, + p_id: 0, + t: 0, + o_i: 0, + o_type: 0, + g_id, + }; + let max_key = RunRecordV2 { + s_id: SubjectId(ann), + o_key: u64::MAX, + p_id: u32::MAX, + t: u32::MAX, + o_i: u32::MAX, + o_type: u16::MAX, + g_id, + }; + let filter = BinaryFilter { + s_id: Some(ann), + ..Default::default() + }; + let mut cursor = BinaryCursor::new( + Arc::clone(&store), + RunSortOrder::Spot, + Arc::clone(branch), + &min_key, + &max_key, + filter, + projection, + ); + let mut state: [Option; 3] = [None; 3]; + while let Some(batch) = cursor.next_batch()? { + for i in 0..batch.row_count { + if batch.s_id.get(i) != ann { + continue; + } + let p_id = batch.p_id.get(i); + let Some(slot) = slots.iter().position(|s| *s == Some(p_id)) else { + continue; + }; + let o_type = batch.o_type.get_or(i, 0); + let o_key = batch.o_key.get(i); + state[slot] = match slot { + 0 if o_type == OType::IRI_REF.as_u16() => Some(SlotValue::Subject(o_key)), + 1 if o_type == OType::IRI_REF.as_u16() => store + .resolve_subject_iri(o_key) + .ok() + .and_then(|iri| predicates.get(&iri)) + .map(SlotValue::Predicate), + 2 => Some(SlotValue::Object(ObjectId::Typed { o_type, o_key })), + _ => None, + }; + } + } + states.insert((g_id, ann), state); + } + Ok(states) +} + /// Fetch one commit blob, honoring the optional artifact cache. async fn fetch_commit_bytes( cs: &dyn ContentStore, diff --git a/fluree-db-indexer/src/run_index/resolve/link_synth.rs b/fluree-db-indexer/src/run_index/resolve/link_synth.rs index d15eb3d389..652ccc4d8f 100644 --- a/fluree-db-indexer/src/run_index/resolve/link_synth.rs +++ b/fluree-db-indexer/src/run_index/resolve/link_synth.rs @@ -1,29 +1,154 @@ -//! Synthesizes the RDF 1.2 link record for every `f:reifies*` bundle the -//! resolver sees, so rebuilds carry the same `_:r rdf:reifies ` flakes +//! Synthesizes the RDF 1.2 link record, `_:r rdf:reifies `, for every +//! reifier a rebuild or incremental build sees, so both carry the flakes //! bulk import writes. //! -//! A bundle's three required slots arrive as consecutive records about the -//! reifier subject (the writer emits them together, and a SPOT-sorted commit -//! keeps one subject's records adjacent). The assembler keys a pending bundle -//! on `(g_id, reifier, t, op)`, fills the subject, predicate and object slots -//! as their records pass, and on completion appends the base edge as a -//! pseudo-record to the chunk's term table and a link record, whose `o_key` -//! is that entry's ordinal, to the chunk's records. The build remaps the -//! entry to global ids and interns it, replacing the ordinal with the handle. +//! A commit's `f:reifies*` ops are not a bundle. A re-point that changes one +//! slot writes only that slot's retract and assert (sync and upsert cancel +//! the unchanged slots), and a full re-point's six ops interleave under the +//! per-commit sort, where predicate and object precede `op`. So the resolver +//! only *collects* attachment ops ([`LinkSynth::observe`]), and the build +//! replays them per reifier once ids are global ([`replay_attachments`]): +//! the reifier's ops in `t` order, emitting a link retract and assert +//! whenever its attachment changes from or to a complete edge. An +//! incremental build seeds each reifier with the attachment the base index +//! holds for it; a rebuild replays the whole history from nothing. //! -//! Disabled unless a build path opts in: a path that has not learned to -//! resolve the ordinals must not see link records. +//! Disabled unless a build path opts in: a path that does not replay must +//! not collect. use super::global_dict::PredicateDict; use super::resolver::RebuildChunk; use fluree_db_binary_index::format::run_record::{RunRecord, LIST_INDEX_NONE}; use fluree_db_core::commit::codec::raw_reader::{RawObject, RawOp}; +use fluree_db_core::o_type::{DecodeKind, OType}; +use fluree_db_core::o_type_registry::OTypeRegistry; use fluree_db_core::subject_id::SubjectId; -use fluree_db_core::value_id::ObjKind; +use fluree_db_core::triple_term::TermKey; +use fluree_db_core::value_id::{ObjKey, ObjKind}; +use fluree_db_core::DatatypeDictId; use fluree_vocab::{db, fluree}; use std::collections::HashMap; +use std::io; -/// Per-build state for link synthesis. +/// The base edge's object, as the resolver saw it or as the base index +/// stores it. The two compare through [`ObjectId::typed`]. +#[derive(Debug, Clone, Copy)] +pub enum ObjectId { + /// Kind, key, datatype and tag of a resolved op; the `o_type` needs the + /// registry, which knows every custom datatype only after resolving. + Raw { + o_kind: u8, + o_key: u64, + dt: u16, + lang_id: u16, + }, + /// An index row's `o_type` and key. + Typed { o_type: u16, o_key: u64 }, +} + +impl ObjectId { + fn typed(self, registry: &OTypeRegistry) -> (u16, u64) { + match self { + ObjectId::Raw { + o_kind, + o_key, + dt, + lang_id, + } => ( + registry + .resolve( + ObjKind::from_u8(o_kind), + DatatypeDictId::from_u16(dt), + lang_id, + ) + .as_u16(), + o_key, + ), + ObjectId::Typed { o_type, o_key } => (o_type, o_key), + } + } +} + +/// One slot of an attachment. +#[derive(Debug, Clone, Copy)] +pub enum SlotValue { + /// `f:reifiesSubject`: the base edge's subject id. + Subject(u64), + /// `f:reifiesPredicate`: the base edge's predicate id. + Predicate(u32), + /// `f:reifiesObject`. + Object(ObjectId), +} + +impl SlotValue { + fn slot(self) -> usize { + match self { + SlotValue::Subject(_) => 0, + SlotValue::Predicate(_) => 1, + SlotValue::Object(_) => 2, + } + } + + fn same(self, other: SlotValue, registry: &OTypeRegistry) -> bool { + match (self, other) { + (SlotValue::Subject(a), SlotValue::Subject(b)) => a == b, + (SlotValue::Predicate(a), SlotValue::Predicate(b)) => a == b, + (SlotValue::Object(a), SlotValue::Object(b)) => a.typed(registry) == b.typed(registry), + _ => false, + } + } +} + +/// An `f:reifies*` op as the resolver saw it. Subject and string ids are +/// chunk-local until [`AttachmentOp::remap`]; predicate ids are global. +#[derive(Debug, Clone, Copy)] +pub struct AttachmentOp { + pub g_id: u16, + pub ann: u64, + pub t: u32, + pub op: u8, + pub value: SlotValue, +} + +impl AttachmentOp { + /// Chunk-local subject and string ids → global, with the build's remap + /// tables. + pub fn remap(&mut self, s_remap: &[u64], str_remap: &[u32]) -> Result<(), String> { + let subject = |local: u64| -> Result { + s_remap + .get(local as usize) + .copied() + .ok_or_else(|| format!("attachment subject remap miss: local {local}")) + }; + self.ann = subject(self.ann)?; + match &mut self.value { + SlotValue::Subject(s) => *s = subject(*s)?, + SlotValue::Predicate(_) | SlotValue::Object(ObjectId::Typed { .. }) => {} + SlotValue::Object(ObjectId::Raw { o_kind, o_key, .. }) => { + let kind = ObjKind::from_u8(*o_kind); + if kind == ObjKind::REF_ID { + *o_key = subject(*o_key)?; + } else if kind == ObjKind::LEX_ID || kind == ObjKind::JSON_ID { + let local = ObjKey::from_u64(*o_key).decode_u32_id() as usize; + let global = *str_remap + .get(local) + .ok_or_else(|| format!("attachment string remap miss: local {local}"))?; + *o_key = ObjKey::encode_u32_id(global).as_u64(); + } + } + } + Ok(()) + } +} + +/// The link flake's predicate and datatype ids. +#[derive(Debug, Clone, Copy)] +pub struct LinkIds { + pub p_id: u32, + pub dt: u16, +} + +/// Per-build collector for attachment ops. #[derive(Debug, Default)] pub struct LinkSynth { enabled: bool, @@ -35,37 +160,31 @@ pub struct LinkSynth { /// Predicate-dictionary length at the last slot lookup, so the lookup /// repeats only when new predicates appeared. slots_checked_at: u32, - rdf_reifies: Option, - triple_term_dt: Option, - pending: Option, - /// Link records emitted so far. - pub links_emitted: u64, -} - -#[derive(Debug)] -struct Pending { - g_id: u16, - ann: u64, - t: u32, - op: u8, - s: Option, - p: Option, - /// `(o_kind, o_key, dt, lang_id)` of the base edge's object. - o: Option<(u8, u64, u16, u16)>, + link: Option, } impl LinkSynth { - /// A disabled assembler; see [`Self::enable`]. + /// A disabled collector; see [`Self::enable`]. pub fn new() -> Self { Self::default() } - /// Turn synthesis on. Only a build path that resolves term ordinals may - /// do this. + /// Turn collection on. Only a build path that replays may do this. pub fn enable(&mut self) { self.enabled = true; } + /// The reserved-slot predicate ids, in slot order. + pub fn slots(&self) -> [Option; 3] { + self.slots + } + + /// The link's predicate and datatype ids, allocated on the first + /// attachment op seen so they enter the dictionaries with the commits. + pub fn link_ids(&self) -> Option { + self.link + } + /// Refresh the reserved-slot predicate ids when the dictionary grew. fn refresh_slots(&mut self, predicates: &PredicateDict) { if self.slots.iter().all(Option::is_some) || predicates.len() == self.slots_checked_at { @@ -84,8 +203,8 @@ impl LinkSynth { } } - /// Feed one resolved record (with its raw op, for the predicate slot's - /// IRI). Must be called for every record in commit order. + /// Record one resolved op if it is an attachment slot (with its raw op, + /// for the predicate slot's IRI). pub fn observe( &mut self, raw: &RawOp<'_>, @@ -102,112 +221,144 @@ impl LinkSynth { let Some(slot) = self.slots.iter().position(|s| *s == Some(record.p_id)) else { return; }; - - let key = (record.g_id, record.s_id.as_u64(), record.t, record.op); - let same = self - .pending - .as_ref() - .is_some_and(|p| (p.g_id, p.ann, p.t, p.op) == key); - if !same { - self.flush(chunk, predicates, datatypes); - self.pending = Some(Pending { - g_id: record.g_id, - ann: record.s_id.as_u64(), - t: record.t, - op: record.op, - s: None, - p: None, - o: None, - }); - } - let pending = self.pending.as_mut().expect("pending bundle set above"); - match slot { + let value = match slot { 0 => { - if ObjKind::from_u8(record.o_kind) == ObjKind::REF_ID { - pending.s = Some(record.o_key); + if ObjKind::from_u8(record.o_kind) != ObjKind::REF_ID { + return; } + SlotValue::Subject(record.o_key) } 1 => { - if let RawObject::Ref { ns_code, name } = raw.o { - let prefix = ns_prefixes - .get(&ns_code) - .map(std::string::String::as_str) - .unwrap_or(""); - pending.p = Some(predicates.get_or_insert_parts(prefix, name)); - } - } - _ => { - // Resolved under `f:reifiesObject`, which matches the base - // edge's encoding for every kind except the per-predicate - // arenas; those bundles are left to the bundle path. - let kind = ObjKind::from_u8(record.o_kind); - if kind != ObjKind::NUM_BIG && kind != ObjKind::VECTOR_ID { - pending.o = Some((record.o_kind, record.o_key, record.dt, record.lang_id)); - } - } - } - } - - /// Emit the pending bundle if complete. Called on a key change and at - /// the end of each commit. - pub fn flush( - &mut self, - chunk: &mut RebuildChunk, - predicates: &mut PredicateDict, - datatypes: &mut PredicateDict, - ) { - let Some(pending) = self.pending.take() else { - return; - }; - let (Some(s), Some(p), Some((o_kind, o_key, dt, lang_id))) = - (pending.s, pending.p, pending.o) - else { - return; - }; - let link_p = *self - .rdf_reifies - .get_or_insert_with(|| predicates.get_or_insert(fluree_vocab::rdf::REIFIES)); - let link_dt = match self.triple_term_dt { - Some(d) => d, - None => { - let raw = datatypes.get_or_insert(fluree::TRIPLE_TERM); - let Ok(d) = u16::try_from(raw) else { - tracing::warn!( - dt_id = raw, - "f:tripleTerm datatype id exceeds u16; link skipped" - ); + let RawObject::Ref { ns_code, name } = raw.o else { return; }; - self.triple_term_dt = Some(d); - d + let prefix = ns_prefixes + .get(&ns_code) + .map(std::string::String::as_str) + .unwrap_or(""); + SlotValue::Predicate(predicates.get_or_insert_parts(prefix, name)) } + _ => SlotValue::Object(ObjectId::Raw { + o_kind: record.o_kind, + o_key: record.o_key, + dt: record.dt, + lang_id: record.lang_id, + }), }; - let ordinal = chunk.terms.len() as u64; - chunk.terms.push(RunRecord { - g_id: pending.g_id, - s_id: SubjectId::from_u64(s), - p_id: p, - dt, - o_kind, - op: 1, - o_key, - t: pending.t, - lang_id, - i: LIST_INDEX_NONE, + if self.link.is_none() { + let p_id = predicates.get_or_insert(fluree_vocab::rdf::REIFIES); + let raw_dt = datatypes.get_or_insert(fluree::TRIPLE_TERM); + let Ok(dt) = u16::try_from(raw_dt) else { + tracing::warn!( + dt_id = raw_dt, + "f:tripleTerm datatype id exceeds u16; links skipped" + ); + return; + }; + self.link = Some(LinkIds { p_id, dt }); + } + chunk.attachments.push(AttachmentOp { + g_id: record.g_id, + ann: record.s_id.as_u64(), + t: record.t, + op: record.op, + value, }); - chunk.records.push(RunRecord { - g_id: pending.g_id, - s_id: SubjectId::from_u64(pending.ann), - p_id: link_p, - dt: link_dt, + } +} + +/// The term a complete attachment names, or `None` while a slot is missing. +/// An object in a per-(graph, predicate) arena has no graph-independent +/// identity and gets no term; import skips those edges too. +fn term_of(state: &[Option; 3], registry: &OTypeRegistry) -> Option { + let ( + Some(SlotValue::Subject(s_id)), + Some(SlotValue::Predicate(p_id)), + Some(SlotValue::Object(o)), + ) = (state[0], state[1], state[2]) + else { + return None; + }; + let (o_type, o_key) = o.typed(registry); + let o_type = OType::from_u16(o_type); + if matches!( + o_type.decode_kind(), + DecodeKind::NumBigArena | DecodeKind::VectorArena + ) { + return None; + } + Some(TermKey { + s_id, + p_id, + o_type, + o_key, + }) +} + +/// Replay every reifier's attachment ops (ids global) and emit its link +/// records: at each `t` where the attachment changes from or to a complete +/// edge, a retract of the old term's link and an assert of the new one. +/// `prior` is the attachment the base index holds for `(g_id, reifier)` +/// before these ops; `handle_for` interns a term. Returns the number of +/// link records emitted. +pub fn replay_attachments( + ops: &mut [AttachmentOp], + mut prior: impl FnMut(u16, u64) -> [Option; 3], + registry: &OTypeRegistry, + link: LinkIds, + handle_for: &mut dyn FnMut(TermKey) -> io::Result, + out: &mut Vec, +) -> io::Result { + // Within one `t`, retracts apply before asserts so a re-pointed slot + // passes through its old value on the way to the new one. + ops.sort_by_key(|o| (o.g_id, o.ann, o.t, o.op, o.value.slot())); + let mut emitted = 0u64; + let mut link_record = |g_id: u16, ann: u64, key: TermKey, t: u32, op: u8| -> io::Result<()> { + let handle = handle_for(key)?; + out.push(RunRecord { + g_id, + s_id: SubjectId::from_u64(ann), + p_id: link.p_id, + dt: link.dt, o_kind: ObjKind::TRIPLE_TERM.as_u8(), - op: pending.op, - o_key: ordinal, - t: pending.t, + op, + o_key: handle, + t, lang_id: 0, i: LIST_INDEX_NONE, }); - chunk.flake_count += 1; - self.links_emitted += 1; + emitted += 1; + Ok(()) + }; + let mut i = 0; + while i < ops.len() { + let (g_id, ann) = (ops[i].g_id, ops[i].ann); + let mut state = prior(g_id, ann); + while i < ops.len() && ops[i].g_id == g_id && ops[i].ann == ann { + let t = ops[i].t; + let before = term_of(&state, registry); + while i < ops.len() && ops[i].g_id == g_id && ops[i].ann == ann && ops[i].t == t { + let op = ops[i]; + let slot = op.value.slot(); + if op.op == 0 { + if state[slot].is_some_and(|v| v.same(op.value, registry)) { + state[slot] = None; + } + } else { + state[slot] = Some(op.value); + } + i += 1; + } + let after = term_of(&state, registry); + if before != after { + if let Some(old) = before { + link_record(g_id, ann, old, t, 0)?; + } + if let Some(new) = after { + link_record(g_id, ann, new, t, 1)?; + } + } + } } + Ok(emitted) } diff --git a/fluree-db-indexer/src/run_index/resolve/resolver.rs b/fluree-db-indexer/src/run_index/resolve/resolver.rs index 676066e0a6..f8a0f9355e 100644 --- a/fluree-db-indexer/src/run_index/resolve/resolver.rs +++ b/fluree-db-indexer/src/run_index/resolve/resolver.rs @@ -1502,9 +1502,6 @@ impl SharedResolverState { Ok(()) })?; - self.link_synth - .flush(chunk, &mut self.predicates, &mut self.datatypes); - // Emit txn-meta records into the same chunk. let meta_count = self.emit_txn_meta_chunk( commit_hash_hex, @@ -2141,8 +2138,11 @@ pub struct RebuildChunk { /// Buffered RunRecords (with chunk-local subject/string IDs). pub records: Vec, /// Reified base edges as pseudo-records, addressed by ordinal from the - /// link records' `o_key` (see `link_synth`). + /// link records' `o_key` (the bulk-import sink's form; see `link_synth`). pub terms: Vec, + /// `f:reifies*` ops for the build to replay into link records once ids + /// are global (see `link_synth`). + pub attachments: Vec, /// Running count of flakes (records) in this chunk. pub flake_count: u64, } @@ -2154,6 +2154,7 @@ impl RebuildChunk { strings: super::chunk_dict::ChunkStringDict::new(), records: Vec::new(), terms: Vec::new(), + attachments: Vec::new(), flake_count: 0, } } From 9b3980b8f7412bec48fff391ef3475f9242b7b0a Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 14:53:14 -0400 Subject: [PATCH 08/92] perf(query): elide unread reified-edge positions before the fast paths look The count planner and the other fused fast paths classify the query's pattern list as lowered, so a `COUNT(*)` over `<< ?s ?p ?o >>` still carried three component BINDs at that point and fell through to the generic join, ten times slower than the same two-triple join written by hand. Run the unread-position elision on the query before any fast path sees it; the WHERE planner's own pass stays for the seeded and nested cases. --- fluree-db-query/src/execute/operator_tree.rs | 39 ++++++++++++++++++++ fluree-db-query/src/execute/where_plan.rs | 2 +- 2 files changed, 40 insertions(+), 1 deletion(-) diff --git a/fluree-db-query/src/execute/operator_tree.rs b/fluree-db-query/src/execute/operator_tree.rs index d6ea2395fb..bcf718e7c1 100644 --- a/fluree-db-query/src/execute/operator_tree.rs +++ b/fluree-db-query/src/execute/operator_tree.rs @@ -2415,6 +2415,33 @@ fn result_is_multiplicity_blind(query: &Query) -> bool { } } +/// The query with the `BIND(SUBJECT|PREDICATE|OBJECT(?term) AS ?v)` patterns +/// nothing reads removed (see `elide_unread_term_binds`), or `None` when it +/// has none to drop. +fn elide_unread_term_binds_in_query(query: &Query) -> Option { + let deps = compute_variable_deps(query); + let required = deps.as_ref().map(|d| d.required_where_vars.as_slice()); + let needed: HashSet = match required { + Some(req) => req.iter().copied().collect(), + None => { + let mut counts: HashMap = HashMap::new(); + let mut vars: HashSet = HashSet::new(); + collect_var_stats(&query.patterns, &mut counts, &mut vars); + vars.extend(counts.keys().copied()); + vars + } + }; + let patterns = crate::execute::where_plan::elide_unread_term_binds( + &query.patterns, + &needed, + required, + &HashSet::new(), + )?; + let mut query = query.clone(); + query.patterns = patterns; + Some(query) +} + fn build_operator_tree_inner( query: &Query, stats: Option>, @@ -2453,6 +2480,18 @@ fn build_operator_tree_inner( // general pipeline run over the deduplicating `DatasetOperator`. let enable_fused_fast_paths = enable_fused_fast_paths && !planning.multi_default_graph; + // Drop the reified-edge positions nothing reads before any fast path + // looks at the pattern list, so a `COUNT(*)` over `<< ?s ?p ?o >>` is the + // plain two-triple join the count planner handles. + let elided_query; + let query = match elide_unread_term_binds_in_query(query) { + Some(q) => { + elided_query = q; + &elided_query + } + None => query, + }; + // Expression-based ORDER BY (`query.order_binds`) is materialized only by the // generic pipeline's dedicated post-grouping bind stage. No fast path runs // that stage, so any fast path that returns early would sort on a synthetic diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index e4d2a0a8b5..9cdc10b1f5 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -2316,7 +2316,7 @@ fn values_cell_as_ref_term(binding: &crate::binding::Binding) -> Option { /// projection. A variable the seed binds keeps its BIND, which is then the /// equality check on that position. The `f:reifies*` twin of this rule is /// `elide_redundant_chain`. -fn elide_unread_term_binds( +pub(crate) fn elide_unread_term_binds( patterns: &[Pattern], needed_vars: &HashSet, required_where_vars: Option<&[VarId]>, From 203cc3e94839caeea7c1a3a0b3fb5ec8d6183f94 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 17:06:22 -0400 Subject: [PATCH 09/92] fix(query): term accessors keep the object's datatype on every path OBJECT() of a materialized term, and DATATYPE/LANG over any accessor, went through the numeric comparable, which carries no datatype: an indexed "5"^^xsd:int read as xsd:integer as soon as an unrelated unindexed commit turned late materialization off, and DATATYPE(OBJECT(?t)) lost the subtype even on a quiescent ledger. term_component_binding now yields the component as a binding on both the encoded path (dictionary key) and the materialized path (the term itself), with datatype and tag, and DATATYPE and LANG read an accessor argument through it exactly as they read a bound variable. Pinned on both paths. --- fluree-db-api/tests/it_triple_term_links.rs | 66 ++++++++++ fluree-db-query/src/eval.rs | 2 +- fluree-db-query/src/eval/rdf.rs | 136 ++++++++++++-------- fluree-db-query/src/eval/string.rs | 64 ++++++--- 4 files changed, 194 insertions(+), 74 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 6417860812..c33fd69543 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -606,3 +606,69 @@ async fn incremental_index_follows_a_partial_repoint() { }) .await; } + +async fn assert_object_types_survive(fluree: &fluree_db_api::Fluree, ledger: &LedgerState) { + let got = run_link_query( + fluree, + ledger, + "SELECT (DATATYPE(OBJECT(?t)) AS ?dt) WHERE { ?r rdf:reifies ?t . \ + FILTER(PREDICATE(?t) = ex:size) } ORDER BY ?dt" + .to_string(), + ) + .await; + assert_eq!(got.len(), 2, "{got:#?}"); + assert!( + got[0][0].ends_with("int") && got[1][0].ends_with("integer"), + "{got:#?}" + ); + let got = run_link_query( + fluree, + ledger, + "SELECT (LANG(OBJECT(?t)) AS ?l) WHERE { ?r rdf:reifies ?t . \ + FILTER(PREDICATE(?t) = ex:title) } ORDER BY ?l" + .to_string(), + ) + .await; + let langs: Vec<&str> = got.iter().map(|r| r[0].as_str()).collect(); + assert_eq!(langs, ["", "en", "fr"], "{got:#?}"); + let got = run_link_query( + fluree, + ledger, + "SELECT (DATATYPE(?o) AS ?dt) WHERE { ?r rdf:reifies <<( ex:doc ex:size ?o )>> } \ + ORDER BY ?dt" + .to_string(), + ) + .await; + assert_eq!(got.len(), 2, "{got:#?}"); + assert!( + got[0][0].ends_with("int") && got[1][0].ends_with("integer"), + "{got:#?}" + ); +} + +/// `DATATYPE` and `LANG` over an accessor read the component the way a +/// bound variable is read: on the indexed path, and on the materialized +/// path once an unrelated unindexed commit turns late materialization off. +#[tokio::test] +async fn link_lowering_keeps_object_types_through_accessors_and_novelty() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let (fluree, ledger) = import( + &[("literals.ttl", LITERAL_CLAIMS)], + "it/triple-term-links:accessor-types", + ) + .await; + assert_object_types_survive(&fluree, &ledger).await; + + let after = fluree + .insert( + ledger, + &json!({ + "@context": { "ex": "http://example.org/" }, + "@id": "ex:other", + "ex:note": "unrelated" + }), + ) + .await + .expect("unrelated insert"); + assert_object_types_survive(&fluree, &after.ledger).await; +} diff --git a/fluree-db-query/src/eval.rs b/fluree-db-query/src/eval.rs index b4bde059c3..1c6e105f16 100644 --- a/fluree-db-query/src/eval.rs +++ b/fluree-db-query/src/eval.rs @@ -300,7 +300,7 @@ impl Expression { args, } = self { - if let Some(binding) = rdf::encoded_term_component(func, args, row, ctx)? { + if let Some(binding) = rdf::term_component_binding(func, args, row, ctx)? { return Ok(binding); } } diff --git a/fluree-db-query/src/eval/rdf.rs b/fluree-db-query/src/eval/rdf.rs index 2bb9a043e4..435241e463 100644 --- a/fluree-db-query/src/eval/rdf.rs +++ b/fluree-db-query/src/eval/rdf.rs @@ -73,6 +73,18 @@ pub fn eval_datatype( None => Ok(None), // unbound variable }; } + // A term accessor binds its component the way a variable is bound, so + // `DATATYPE(OBJECT(?t))` keeps `xsd:int` where the value path would say + // `xsd:integer`. + if let Expression::Call { + func: func @ (Function::TripleSubject | Function::TriplePredicate | Function::TripleObject), + args: inner, + } = &args[0] + { + if let Some(binding) = term_component_binding(func, inner, row, ctx)? { + return datatype_of_binding(&binding, ctx, strict); + } + } // General case: evaluate the argument expression (arithmetic, casts, // constants, nested builtins) and classify the resulting value. match args[0].eval_to_comparable(row, ctx)? { @@ -545,41 +557,89 @@ fn term_object_binding( )) } -/// `BIND(SUBJECT|PREDICATE|OBJECT(?term) AS ?v)` on an encoded handle: the -/// component in the encoded form a scan would bind it, so the BIND is one -/// dictionary lookup and no materialization. `None` takes the value path. -pub(crate) fn encoded_term_component( +/// The component `SUBJECT|PREDICATE|OBJECT(?term)` names, as a binding: from +/// the dictionary key when the term is a late-materialized handle (one +/// lookup, no materialization), or from the term itself when it is +/// materialized. Either way the object keeps its datatype or tag, so a +/// `BIND`, `DATATYPE` or `LANG` over it sees what a bound variable would. +/// `None` when the argument is not a variable bound to a term. +pub(crate) fn term_component_binding( func: &Function, args: &[Expression], row: &R, ctx: Option<&ExecutionContext<'_>>, ) -> Result> { - let Some((key, t, ctx)) = encoded_term(args, row, ctx)? else { + if let Some((key, t, ctx)) = encoded_term(args, row, ctx)? { + let binding = match func { + Function::TripleSubject => Binding::encoded_sid(key.s_id), + Function::TriplePredicate => Binding::EncodedPid { p_id: key.p_id }, + Function::TripleObject => term_object_binding(&key, t, ctx)?, + _ => return Ok(None), + }; + return Ok(Some(binding)); + } + let [Expression::Var(v)] = args else { return Ok(None); }; - let binding = match func { - Function::TripleSubject => Binding::encoded_sid(key.s_id), - Function::TriplePredicate => Binding::EncodedPid { p_id: key.p_id }, - Function::TripleObject => term_object_binding(&key, t, ctx)?, - _ => return Ok(None), + let Some(Binding::Lit { + val: fluree_db_core::FlakeValue::TripleTerm(term), + .. + }) = row.get(*v) + else { + return Ok(None); }; - Ok(Some(binding)) + Ok(Some(match func { + Function::TripleSubject => Binding::sid(term.s.clone()), + Function::TriplePredicate => Binding::sid(term.p.clone()), + Function::TripleObject => materialized_term_object(term), + _ => return Ok(None), + })) } /// A materialized term's object with its datatype or language tag, which -/// `OBJECT()` must keep: `"chat"@fr` is not `"chat"`. -fn term_object_comparable( - term: &fluree_db_core::TripleTermValue, +/// `OBJECT()` must keep: `"chat"@fr` is not `"chat"`, `"5"^^xsd:int` is not +/// `5`. +fn materialized_term_object(term: &fluree_db_core::TripleTermValue) -> Binding { + match &term.o { + fluree_db_core::FlakeValue::Ref(sid) => Binding::sid(sid.clone()), + other => Binding::Lit { + val: other.clone(), + dtc: match &term.lang { + Some(lang) => fluree_db_core::DatatypeConstraint::LangTag(Arc::from(lang.as_str())), + None => fluree_db_core::DatatypeConstraint::Explicit(term.dt.clone()), + }, + t: None, + op: None, + p_id: None, + }, + } +} + +/// One accessor: the component's binding converted exactly as a bound +/// variable is, else the value path for a non-variable argument. +fn eval_term_accessor( + func: Function, + args: &[Expression], + row: &R, ctx: Option<&ExecutionContext<'_>>, -) -> Option { - if let fluree_db_core::FlakeValue::Ref(sid) = &term.o { - return Some(ComparableValue::Sid(sid.clone())); + name: &str, +) -> Result> { + check_arity(args, 1, name)?; + if let Some(binding) = term_component_binding(&func, args, row, ctx)? { + return super::binding_to_comparable(Some(&binding), ctx); + } + if matches!(args[0], Expression::Var(_)) { + return Ok(None); } - let dtc = match &term.lang { - Some(lang) => fluree_db_core::DatatypeConstraint::LangTag(Arc::from(lang.as_str())), - None => fluree_db_core::DatatypeConstraint::Explicit(term.dt.clone()), + let Some(term) = triple_term_arg(args, row, ctx, name)? else { + return Ok(None); + }; + let binding = match func { + Function::TripleSubject => Binding::sid(term.s), + Function::TriplePredicate => Binding::sid(term.p), + _ => materialized_term_object(&term), }; - super::lit_to_comparable(&term.o, &dtc, ctx) + super::binding_to_comparable(Some(&binding), ctx) } pub fn eval_triple_subject( @@ -587,17 +647,7 @@ pub fn eval_triple_subject( row: &R, ctx: Option<&ExecutionContext<'_>>, ) -> Result> { - if let Some((key, _, ctx)) = encoded_term(args, row, ctx)? { - let store = ctx - .binary_store - .as_deref() - .expect("encoded term has a store"); - let iri = store - .resolve_subject_iri(key.s_id) - .map_err(|e| QueryError::from_io("resolve_subject_iri", e))?; - return Ok(Some(ComparableValue::Sid(store.encode_iri(&iri)))); - } - Ok(triple_term_arg(args, row, ctx, "SUBJECT")?.map(|t| ComparableValue::Sid(t.s))) + eval_term_accessor(Function::TripleSubject, args, row, ctx, "SUBJECT") } pub fn eval_triple_predicate( @@ -605,21 +655,7 @@ pub fn eval_triple_predicate( row: &R, ctx: Option<&ExecutionContext<'_>>, ) -> Result> { - if let Some((key, _, ctx)) = encoded_term(args, row, ctx)? { - let store = ctx - .binary_store - .as_deref() - .expect("encoded term has a store"); - let sid = store - .p_sid_table() - .get(key.p_id as usize) - .cloned() - .ok_or_else(|| { - QueryError::Internal(format!("triple-term predicate {} is not known", key.p_id)) - })?; - return Ok(Some(ComparableValue::Sid(sid))); - } - Ok(triple_term_arg(args, row, ctx, "PREDICATE")?.map(|t| ComparableValue::Sid(t.p))) + eval_term_accessor(Function::TriplePredicate, args, row, ctx, "PREDICATE") } pub fn eval_triple_object( @@ -627,11 +663,7 @@ pub fn eval_triple_object( row: &R, ctx: Option<&ExecutionContext<'_>>, ) -> Result> { - if let Some((key, t, ctx)) = encoded_term(args, row, ctx)? { - let object = term_object_binding(&key, t, ctx)?; - return super::binding_to_comparable(Some(&object), Some(ctx)); - } - Ok(triple_term_arg(args, row, ctx, "OBJECT")?.and_then(|t| term_object_comparable(&t, ctx))) + eval_term_accessor(Function::TripleObject, args, row, ctx, "OBJECT") } pub fn eval_is_triple( diff --git a/fluree-db-query/src/eval/string.rs b/fluree-db-query/src/eval/string.rs index 2bb9e3bcf9..cef96c1983 100644 --- a/fluree-db-query/src/eval/string.rs +++ b/fluree-db-query/src/eval/string.rs @@ -151,6 +151,35 @@ pub fn eval_str( })) } +/// `LANG` of a binding, read from its tag without materializing the value. +fn lang_of_binding( + binding: Option<&Binding>, + ctx: Option<&ExecutionContext<'_>>, + strict: bool, +) -> Result> { + let tag = match binding { + Some(Binding::Lit { dtc, .. }) => dtc + .lang_tag() + .map(std::string::ToString::to_string) + .unwrap_or_default(), + Some(Binding::EncodedLit { lang_id, .. }) => { + if let Some(store) = ctx.and_then(|c| c.binary_store.as_deref()) { + store + .decode_meta(*lang_id, i32::MIN) + .and_then(|m| m.lang) + .unwrap_or_default() + } else { + String::new() + } + } + // SPARQL §17.4.2.2: LANG of a non-literal (IRI/ref, bnode) or an + // unbound variable is a type error → no value. + _ if strict => return Ok(None), + _ => String::new(), + }; + Ok(Some(ComparableValue::String(Arc::from(tag)))) +} + /// `strict` selects the SPARQL behavior for non-literal arguments: a type /// error, evaluated as "no value" (`Ok(None)`) so a FILTER excludes the row /// and a project-expression/BIND leaves the variable unbound (§17.2/§18.5). @@ -166,27 +195,20 @@ pub fn eval_lang( // Fast path: a bare variable reads the binding's language tag directly, // without materializing the value. if let Expression::Var(var_id) = &args[0] { - let tag = match row.get(*var_id) { - Some(Binding::Lit { dtc, .. }) => dtc - .lang_tag() - .map(std::string::ToString::to_string) - .unwrap_or_default(), - Some(Binding::EncodedLit { lang_id, .. }) => { - if let Some(store) = ctx.and_then(|c| c.binary_store.as_deref()) { - store - .decode_meta(*lang_id, i32::MIN) - .and_then(|m| m.lang) - .unwrap_or_default() - } else { - String::new() - } - } - // SPARQL §17.4.2.2: LANG of a non-literal (IRI/ref, bnode) or an - // unbound variable is a type error → no value. - _ if strict => return Ok(None), - _ => String::new(), - }; - return Ok(Some(ComparableValue::String(Arc::from(tag)))); + return lang_of_binding(row.get(*var_id), ctx, strict); + } + // A term accessor binds its component the way a variable is bound. + if let Expression::Call { + func: + func @ (crate::ir::Function::TripleSubject + | crate::ir::Function::TriplePredicate + | crate::ir::Function::TripleObject), + args: inner, + } = &args[0] + { + if let Some(binding) = crate::eval::rdf::term_component_binding(func, inner, row, ctx)? { + return lang_of_binding(Some(&binding), ctx, strict); + } } // General case: evaluate the argument expression and read the tag off the // resulting value (only lang-tagged strings carry one). From 8e6daf0c49fb19f47dc34009f130ec4c5a348386 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 17:06:22 -0400 Subject: [PATCH 10/92] fix(index): fail the build on a base attachment read error; stream link replay The incremental replay read a reifier's base predicate slot with a swallowed error, so a failed dictionary fetch looked like a missing predicate, the replay saw an incomplete attachment before and after an object-only re-point, and the old link survived. The read error and a predicate absent from the dictionary now fail the build. Rebuild frees each chunk's record and term buffers as soon as its spool file is written instead of after the loop, and the replay hands link records to the spool one reifier at a time, in (graph, SPOT) order, rather than accumulating them beside the attachment table. --- fluree-db-indexer/src/build/rebuild.rs | 53 +++++++++-------- .../run_index/build/incremental_resolve.rs | 27 +++++++-- .../src/run_index/resolve/link_synth.rs | 58 +++++++++++-------- 3 files changed, 83 insertions(+), 55 deletions(-) diff --git a/fluree-db-indexer/src/build/rebuild.rs b/fluree-db-indexer/src/build/rebuild.rs index 7c621f32e0..fb40230f54 100644 --- a/fluree-db-indexer/src/build/rebuild.rs +++ b/fluree-db-indexer/src/build/rebuild.rs @@ -696,6 +696,14 @@ where .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; } + // The chunk is on disk; release its buffers now rather than + // after the loop, so the peak is one chunk plus the link + // replay, not every chunk. + records.clear(); + records.shrink_to_fit(); + terms.clear(); + terms.shrink_to_fit(); + sorted_commit_infos.push(SortedCommitInfo { path: fsc_path, record_count: spool_info.record_count, @@ -710,48 +718,45 @@ where } // Link records: replay every reifier's attachment history from - // nothing and write the result as one more sorted commit file. + // nothing, streamed straight into one more sorted commit file. let links_emitted = if all_attachments.is_empty() { 0 } else { let link_ids = shared.link_synth.link_ids().ok_or_else(|| { IndexerError::StorageWrite("attachment ops without link ids".into()) })?; - let mut links: Vec = Vec::new(); + let ci = chunk_records.len(); + let fsc_path = commits_dir.join("links.fsc"); + let mut spool_writer = run_index::spool::SpoolWriter::new(&fsc_path, ci) + .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; let emitted = crate::run_index::resolve::link_synth::replay_attachments( &mut all_attachments, |_, _| [None; 3], &term_registry, link_ids, &mut |key| term_builder.get_or_insert(key), - &mut links, + &mut |record| spool_writer.push(&record), ) .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; drop(all_attachments); - links.sort_unstable_by(fluree_db_binary_index::format::run_record::cmp_g_spot); - let ci = chunk_records.len(); - let fsc_path = commits_dir.join("links.fsc"); - let mut spool_writer = run_index::spool::SpoolWriter::new(&fsc_path, ci) - .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; - for record in &links { - spool_writer - .push(record) - .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; - } let spool_info = spool_writer .finish() .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; - sorted_commit_infos.push(SortedCommitInfo { - path: fsc_path, - record_count: spool_info.record_count, - byte_len: spool_info.byte_len, - chunk_idx: ci, - subject_count: 0, - string_count: 0, - types_map_path: None, - duplicates_removed: 0, - term_table: None, - }); + if emitted > 0 { + sorted_commit_infos.push(SortedCommitInfo { + path: fsc_path, + record_count: spool_info.record_count, + byte_len: spool_info.byte_len, + chunk_idx: ci, + subject_count: 0, + string_count: 0, + types_map_path: None, + duplicates_removed: 0, + term_table: None, + }); + } else { + let _ = std::fs::remove_file(&fsc_path); + } emitted }; diff --git a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs index 352fd2c2b2..95352e5060 100644 --- a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs +++ b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs @@ -780,7 +780,10 @@ pub async fn resolve_incremental_commits_v6( &o_type_registry, link_ids, &mut handle_for, - &mut v1_records, + &mut |record| { + v1_records.push(record); + Ok(()) + }, )?; tracing::debug!( links = emitted, @@ -1331,11 +1334,23 @@ async fn base_attachment_states( let o_key = batch.o_key.get(i); state[slot] = match slot { 0 if o_type == OType::IRI_REF.as_u16() => Some(SlotValue::Subject(o_key)), - 1 if o_type == OType::IRI_REF.as_u16() => store - .resolve_subject_iri(o_key) - .ok() - .and_then(|iri| predicates.get(&iri)) - .map(SlotValue::Predicate), + // A read failure or a predicate the dictionary does not + // hold must fail the build: treated as "no predicate", an + // object-only re-point would replay from and to an + // incomplete attachment and leave the old link in place. + 1 if o_type == OType::IRI_REF.as_u16() => { + let iri = store.resolve_subject_iri(o_key)?; + let p_id = predicates.get(&iri).ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + format!( + "reifier {ann} is attached to predicate {iri}, which the \ + dictionary does not hold" + ), + ) + })?; + Some(SlotValue::Predicate(p_id)) + } 2 => Some(SlotValue::Object(ObjectId::Typed { o_type, o_key })), _ => None, }; diff --git a/fluree-db-indexer/src/run_index/resolve/link_synth.rs b/fluree-db-indexer/src/run_index/resolve/link_synth.rs index 652ccc4d8f..ff93d12c4f 100644 --- a/fluree-db-indexer/src/run_index/resolve/link_synth.rs +++ b/fluree-db-indexer/src/run_index/resolve/link_synth.rs @@ -295,11 +295,13 @@ fn term_of(state: &[Option; 3], registry: &OTypeRegistry) -> Option io::Result, - out: &mut Vec, + sink: &mut dyn FnMut(RunRecord) -> io::Result<()>, ) -> io::Result { // Within one `t`, retracts apply before asserts so a re-pointed slot // passes through its old value on the way to the new one. ops.sort_by_key(|o| (o.g_id, o.ann, o.t, o.op, o.value.slot())); let mut emitted = 0u64; - let mut link_record = |g_id: u16, ann: u64, key: TermKey, t: u32, op: u8| -> io::Result<()> { - let handle = handle_for(key)?; - out.push(RunRecord { - g_id, - s_id: SubjectId::from_u64(ann), - p_id: link.p_id, - dt: link.dt, - o_kind: ObjKind::TRIPLE_TERM.as_u8(), - op, - o_key: handle, - t, - lang_id: 0, - i: LIST_INDEX_NONE, - }); - emitted += 1; - Ok(()) - }; + let mut group: Vec = Vec::new(); let mut i = 0; while i < ops.len() { let (g_id, ann) = (ops[i].g_id, ops[i].ann); @@ -352,13 +338,35 @@ pub fn replay_attachments( let after = term_of(&state, registry); if before != after { if let Some(old) = before { - link_record(g_id, ann, old, t, 0)?; + group.push(link_record(g_id, ann, link, handle_for(old)?, t, 0)); } if let Some(new) = after { - link_record(g_id, ann, new, t, 1)?; + group.push(link_record(g_id, ann, link, handle_for(new)?, t, 1)); } } } + // One reifier's records share `g_id` and subject; sorting them puts + // the stream as a whole in `(g_id, SPOT)` order. + group.sort_unstable_by(fluree_db_binary_index::format::run_record::cmp_g_spot); + emitted += group.len() as u64; + for record in group.drain(..) { + sink(record)?; + } } Ok(emitted) } + +fn link_record(g_id: u16, ann: u64, link: LinkIds, handle: u64, t: u32, op: u8) -> RunRecord { + RunRecord { + g_id, + s_id: SubjectId::from_u64(ann), + p_id: link.p_id, + dt: link.dt, + o_kind: ObjKind::TRIPLE_TERM.as_u8(), + op, + o_key: handle, + t, + lang_id: 0, + i: LIST_INDEX_NONE, + } +} From a2d1f999226bbbb71646bb539f9c8f027698e467 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 17:06:27 -0400 Subject: [PATCH 11/92] fix(query): rdf:reifies takes a triple term, on write and on scan A reified-triple pattern lowers to `?r rdf:reifies ?t`, and a row such as `ex:r rdf:reifies ex:x` satisfied it: the accessor BINDs failed and kept the row, and a count elided them altogether. Neither a filter nor a datatype constraint on the link triple is an option, since both take the shape off the count plan and the planner paths that gate on dtc.is_none(), and the count planner counts from leaf-directory metadata that cannot see object types. So the predicate carries the rule. The write path refuses rdf:reifies with a non-term object on every surface (JSON-LD, SPARQL UPDATE, Turtle transact, the template path, bulk import), as it refuses user-authored f:reifies*; the reified-triple forms never reach those checks. The scan treats an rdf:reifies pattern with a variable object as term-only on the encoded object type, on novelty flakes too, and narrows a POST seek to the term segment. Rows written as a plain predicate before this rule are dropped by the scan and still counted by the count plan. --- fluree-db-api/src/tx.rs | 18 ++++++ fluree-db-api/tests/it_triple_term_links.rs | 57 +++++++++++++++++++ fluree-db-query/src/binary_scan.rs | 19 +++++++ fluree-db-transact/src/flake_sink.rs | 17 ++++++ fluree-db-transact/src/import_sink.rs | 16 ++++++ fluree-db-transact/src/lower_sparql_update.rs | 18 ++++++ .../src/parse/edge_annotations.rs | 6 +- 7 files changed, 149 insertions(+), 2 deletions(-) diff --git a/fluree-db-api/src/tx.rs b/fluree-db-api/src/tx.rs index f84ffa0309..ce174e346e 100644 --- a/fluree-db-api/src/tx.rs +++ b/fluree-db-api/src/tx.rs @@ -2504,6 +2504,24 @@ fn convert_named_graphs_to_templates( for obj in &triple.objects { let (object_term, dtc) = convert_object(obj, &block.prefixes, ns_registry)?; + // `rdf:reifies` names a triple term; the reified forms come + // through `block.reified` below, so an ordinary object here + // is a data error the link lowering would read as a link. + if matches!(&predicate_term, TemplateTerm::Sid(p) if fluree_db_core::is_rdf_reifies(p)) + && !matches!( + &object_term, + TemplateTerm::Value(fluree_db_core::FlakeValue::TripleTerm(_)) + ) + { + return Err(ApiError::Transact( + fluree_db_transact::TransactError::UnsupportedFeature( + "'rdf:reifies' takes a triple term as its object; write the \ + reified-triple form (`<< s p o >>` or `~ `) rather \ + than an ordinary object" + .to_string(), + ), + )); + } let mut template = TripleTemplate::new(subject_term.clone(), predicate_term.clone(), object_term); template = template.in_graph(std::sync::Arc::clone(&graph)); diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index c33fd69543..5598b552ef 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -672,3 +672,60 @@ async fn link_lowering_keeps_object_types_through_accessors_and_novelty() { .expect("unrelated insert"); assert_object_types_survive(&fluree, &after.ledger).await; } + +/// `rdf:reifies` names a triple term. An ordinary object is refused on the +/// transactional path and on import, so every `rdf:reifies` row is a link +/// and a scan of the predicate never has to second-guess its object. +#[tokio::test] +async fn rdf_reifies_with_an_ordinary_object_is_refused() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger0 = support::genesis_ledger(&fluree, "it/triple-term-links:reifies-firewall"); + let bad = "@prefix ex: .\n\ + @prefix rdf: .\n\ + ex:r rdf:reifies ex:x ; ex:source ex:hr .\n"; + let err = fluree + .upsert_turtle(ledger0, bad) + .await + .expect_err("rdf:reifies with an IRI object must be refused"); + assert!(format!("{err}").contains("reifies"), "{err}"); + + fluree + .create_ledger("it/triple-term-links:reifies-firewall-sparql") + .await + .expect("create ledger"); + let handle = fluree + .ledger_cached("it/triple-term-links:reifies-firewall-sparql") + .await + .expect("ledger handle"); + let err = fluree + .stage(&handle) + .sparql_update( + "PREFIX ex: \n\ + PREFIX rdf: \n\ + INSERT DATA { ex:r rdf:reifies ex:x ; ex:source ex:hr . }", + ) + .execute() + .await + .expect_err("SPARQL UPDATE of rdf:reifies with an IRI object must be refused"); + assert!(format!("{err}").contains("reifies"), "{err}"); + + let db_dir = tempfile::tempdir().expect("db tmpdir"); + let data_dir = tempfile::tempdir().expect("data tmpdir"); + std::fs::write(data_dir.path().join("bad.ttl"), bad).expect("write fixture"); + let fluree = FlureeBuilder::file(db_dir.path().to_string_lossy().to_string()) + .build() + .expect("build file-backed Fluree"); + let result = fluree + .create("it/triple-term-links:reifies-firewall-import") + .import(data_dir.path()) + .threads(1) + .memory_budget_mb(256) + .cleanup(false) + .execute() + .await; + let err = match result { + Ok(_) => panic!("import of rdf:reifies with an IRI object must fail"), + Err(e) => e, + }; + assert!(format!("{err}").contains("reifies"), "{err}"); +} diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index fea5288306..eb24c28a7d 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -232,6 +232,12 @@ pub struct BinaryScanOperator { /// object and the store encodes them: every emitted term handle's /// dictionary key must match, checked without decoding the term. term_key_filter: Option, + /// An `rdf:reifies` scan with a variable object yields only triple-term + /// rows: the predicate is the link, and the write path admits no other + /// object, so this only ever drops rows written before that rule. + /// Checked on the encoded object type; it also narrows a POST seek to + /// the term segment. + term_only: bool, /// Bound object value, if the triple pattern's object is a constant. bound_o: Option, /// `bound_o` as its persisted `(o_type, o_key)` when it is an IRI the @@ -765,6 +771,7 @@ impl BinaryScanOperator { object_bounds, term_o_key_range: None, term_key_filter: None, + term_only: false, bound_o: None, bound_o_encoded: None, check_s_eq_o, @@ -871,6 +878,10 @@ impl BinaryScanOperator { } } + if self.term_only && !matches!(flake.o, FlakeValue::TripleTerm(_)) { + continue; + } + // Datatype / language constraint checks (range fallback path). if let Some(dtc) = &self.pattern.dtc { if dtc.datatype() != &flake.dt { @@ -1459,6 +1470,9 @@ impl BinaryScanOperator { continue; } } + if self.term_only && o_type != OType::TRIPLE_TERM.as_u16() { + continue; + } if let Some((lo, hi)) = self.term_o_key_range { if o_key < lo || o_key > hi { continue; @@ -2150,6 +2164,8 @@ impl Operator for BinaryScanOperator { Self::extract_bound_terms_snapshot(ctx.active_snapshot, &self.pattern); self.bound_o = o_val; self.bound_o_encoded = None; + self.term_only = + self.bound_o.is_none() && p_sid.as_ref().is_some_and(fluree_db_core::is_rdf_reifies); let mut filter = Self::build_filter_from_snapshot_sids( ctx.active_snapshot, &self.pattern, @@ -2492,6 +2508,9 @@ impl Operator for BinaryScanOperator { let mut range_o_type: Option = None; let mut term_o_key_range: Option<(u64, u64)> = None; if order == RunSortOrder::Post && filter.p_id.is_some() && self.bound_o.is_none() { + if self.term_only && filter.o_type.is_none() { + filter.o_type = Some(OType::TRIPLE_TERM.as_u16()); + } if let Some(bounds) = self.object_bounds.as_ref() { // A term-predicate bound is one contiguous handle interval: // handles are `(inner p_id << 32) | seq`. diff --git a/fluree-db-transact/src/flake_sink.rs b/fluree-db-transact/src/flake_sink.rs index 14c2985cf1..f839fecb76 100644 --- a/fluree-db-transact/src/flake_sink.rs +++ b/fluree-db-transact/src/flake_sink.rs @@ -180,6 +180,23 @@ impl<'a> FlakeSink<'a> { let (o, dtc) = self.resolve_object(object)?; + // `rdf:reifies` names a triple term; any other object is a data + // error, and one the link lowering would read as a reifier of + // nothing. The reified-triple forms never reach here: the parser + // hands them to `emit_reified_triple`. + if fluree_db_core::is_rdf_reifies(&p) && !matches!(o, FlakeValue::TripleTerm(_)) { + let e = TransactError::UnsupportedFeature( + "'rdf:reifies' takes a triple term as its object; write the reified-triple \ + form (`<< s p o >>` or `~ `) rather than an ordinary object" + .to_string(), + ); + tracing::error!("FlakeSink: rdf:reifies with a non-term object, aborting — {e}"); + if self.invariant_error.is_none() { + self.invariant_error = Some(e); + } + return None; + } + let dt = dtc.datatype().clone(); let lang = dtc.lang_tag().map(std::string::ToString::to_string); diff --git a/fluree-db-transact/src/import_sink.rs b/fluree-db-transact/src/import_sink.rs index eb166c4da2..ee19ef313f 100644 --- a/fluree-db-transact/src/import_sink.rs +++ b/fluree-db-transact/src/import_sink.rs @@ -881,6 +881,22 @@ mod inner { return; }; + // `rdf:reifies` names a triple term; the reified-triple forms + // arrive through `emit_reified_triple`, so an ordinary object + // here is a data error the link lowering would read as a + // reifier of nothing. Same rule as the transactional sink. + if fluree_db_core::is_rdf_reifies(&p) && !matches!(o, FlakeValue::TripleTerm(_)) { + if self.encode_error.is_none() { + let msg = "'rdf:reifies' takes a triple term as its object; write the \ + reified-triple form (`<< s p o >>` or `~ `) rather \ + than an ordinary object" + .to_string(); + tracing::error!("ImportSink: {msg}"); + self.encode_error = Some(CommitCodecError::InvalidOp(msg)); + } + return; + } + let dt = dtc.datatype().clone(); let lang = dtc.lang_tag().map(std::string::ToString::to_string); diff --git a/fluree-db-transact/src/lower_sparql_update.rs b/fluree-db-transact/src/lower_sparql_update.rs index a7041dfacc..d1eae2d916 100644 --- a/fluree-db-transact/src/lower_sparql_update.rs +++ b/fluree-db-transact/src/lower_sparql_update.rs @@ -557,6 +557,24 @@ fn reject_user_authored_reifies( for tp in triples { check_predicate(&tp.predicate, prologue)?; + // `rdf:reifies` names a triple term; the reified-triple object forms + // lower to the bundle, so any other object is a data error the link + // lowering would read as a link. + if let PredicateTerm::Iri(iri) = &tp.predicate { + if expand_iri(iri, prologue)? == fluree_vocab::rdf::REIFIES + && !matches!( + tp.object, + fluree_db_sparql::ast::Term::QuotedTriple(_) + | fluree_db_sparql::ast::Term::TripleTerm(_) + ) + { + return Err(LowerError::UnsupportedFeature { + feature: "rdf:reifies with an ordinary object in SPARQL UPDATE (it takes \ + a triple term — write `<< s p o >>` or `~ `)", + span: iri.span, + }); + } + } if let Some(ann) = &tp.annotation { for unit in &ann.units { if let Some(block) = &unit.block { diff --git a/fluree-db-transact/src/parse/edge_annotations.rs b/fluree-db-transact/src/parse/edge_annotations.rs index ba1e7f0418..053a711657 100644 --- a/fluree-db-transact/src/parse/edge_annotations.rs +++ b/fluree-db-transact/src/parse/edge_annotations.rs @@ -437,7 +437,9 @@ fn scan_user_authored_reifies_iris(value: &Value, context: &ParsedContext) -> Re } else { expand_iri(k, effective) }; - if reifies_iris::ALL.iter().any(|iri| *iri == expanded_key) { + if reifies_iris::ALL.iter().any(|iri| *iri == expanded_key) + || expanded_key == fluree_vocab::rdf::REIFIES + { return Err(TransactError::UnsupportedFeature(format!( "'{k}' resolves to a system-controlled predicate '{expanded_key}'; \ use @annotation or @reifies instead" @@ -2054,7 +2056,7 @@ fn intercept_annotations_for_predicate( ReifiedObjectShape::Iri(object_id) }; - if reifies_iris::ALL.contains(&predicate) { + if reifies_iris::ALL.contains(&predicate) || predicate == fluree_vocab::rdf::REIFIES { return Err(TransactError::UnsupportedFeature(format!( "'{predicate}' is a system-controlled predicate; use @annotation instead" ))); From 9f38bf20f12ba62bfa59ae973c68273acd45ff89 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 19:45:31 -0400 Subject: [PATCH 12/92] perf(index): batch base lookups when replaying attachments incrementally An incremental build replays each touched reifier from the attachment the base index holds, and a re-point retracts the old term's link. Both lookups ran once per reifier: a fresh SPOT cursor for the base attachment, and a reverse-tree lookup for each term, which rereads a whole ~2.3 MB leaf when no leaf cache is attached. The base attachments now come from one batched SPOT pass per graph (batched_lookup_subject_properties), with each predicate IRI resolved once. Term handles are resolved together: a dry replay names the terms the real replay will ask for, and TermDictReader::find_handles reads each leaf once. On the 10% annotation benchmark slice, an incremental index over 100k partial re-points drops from 72.7 s to 30.2 s (the bundle path on main: 27.0 s), at the same peak memory. --- fluree-db-binary-index/src/dict/term_dict.rs | 12 ++ .../run_index/build/incremental_resolve.rs | 180 +++++++++--------- 2 files changed, 100 insertions(+), 92 deletions(-) diff --git a/fluree-db-binary-index/src/dict/term_dict.rs b/fluree-db-binary-index/src/dict/term_dict.rs index eeb9540d93..d36ea8dcaf 100644 --- a/fluree-db-binary-index/src/dict/term_dict.rs +++ b/fluree-db-binary-index/src/dict/term_dict.rs @@ -106,6 +106,18 @@ impl TermDictReader { } } + /// [`Self::find_handle`] for many keys, reading each reverse-tree leaf once. + pub fn find_handles(&self, keys: &[TermKey]) -> io::Result>> { + match &self.reverse { + Some(tree) => { + let bytes: Vec<[u8; TermKey::LEN]> = + keys.iter().map(TermKey::to_be_bytes).collect(); + tree.reverse_lookup_many(bytes.iter().map(|b| &b[..])) + } + None => Ok(vec![None; keys.len()]), + } + } + /// The encoded base edge behind `handle`, or `None` for an unknown handle. pub fn resolve(&self, handle: u64) -> io::Result> { let Some(reader) = self.forward.get(&term_handle_p_id(handle)) else { diff --git a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs index 95352e5060..efa2128f55 100644 --- a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs +++ b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs @@ -724,7 +724,8 @@ pub async fn resolve_incremental_commits_v6( let mut builder = fluree_db_binary_index::dict::TermDictBuilder::above_watermarks(&base_wms); let triple_term = ObjKind::TRIPLE_TERM.as_u8(); - for record in &mut v1_records { + let mut record_keys = Vec::new(); + for (idx, record) in v1_records.iter().enumerate() { if record.o_kind != triple_term { continue; } @@ -744,39 +745,61 @@ pub async fn resolve_incremental_commits_v6( ), o_key: term.o_key, }; - let existing = match (&base_reader, builder.get(&key)) { - (_, Some(h)) => Some(h), - (Some(reader), None) => reader.find_handle(&key)?, - (None, None) => None, - }; - record.o_key = match existing { - Some(h) => h, - None => builder.get_or_insert(key)?, - }; + record_keys.push((idx, key)); } - if !attachments.is_empty() { - let link_ids = shared.link_synth.link_ids().ok_or_else(|| { + let prior = |g_id: u16, ann: u64| { + prior_attachments + .get(&(g_id, ann)) + .copied() + .unwrap_or([None; 3]) + }; + let link_ids = if attachments.is_empty() { + None + } else { + Some(shared.link_synth.link_ids().ok_or_else(|| { IncrementalResolveError::Io(io::Error::other("attachment ops without link ids")) - })?; - let mut handle_for = |key: fluree_db_core::triple_term::TermKey| -> io::Result { - if let Some(h) = builder.get(&key) { - return Ok(h); - } - if let Some(reader) = &base_reader { - if let Some(h) = reader.find_handle(&key)? { - return Ok(h); - } + })?) + }; + // Resolve every base handle in one batched pass: a reverse lookup per + // term reads a whole tree leaf, and a re-point needs one for the term + // it retracts. A dry replay names the terms the real one will ask for. + let mut base_handles = HashMap::new(); + if let Some(reader) = &base_reader { + let mut wanted: Vec<_> = record_keys.iter().map(|&(_, key)| key).collect(); + if let Some(link_ids) = link_ids { + crate::run_index::resolve::link_synth::replay_attachments( + &mut attachments, + prior, + &o_type_registry, + link_ids, + &mut |key| { + wanted.push(key); + Ok(0) + }, + &mut |_| Ok(()), + )?; + } + wanted.sort_unstable(); + wanted.dedup(); + for (key, handle) in wanted.iter().zip(reader.find_handles(&wanted)?) { + if let Some(handle) = handle { + base_handles.insert(*key, handle); } - builder.get_or_insert(key) - }; + } + } + let mut handle_for = |key: fluree_db_core::triple_term::TermKey| -> io::Result { + match base_handles.get(&key) { + Some(&handle) => Ok(handle), + None => builder.get_or_insert(key), + } + }; + for (idx, key) in record_keys { + v1_records[idx].o_key = handle_for(key)?; + } + if let Some(link_ids) = link_ids { let emitted = crate::run_index::resolve::link_synth::replay_attachments( &mut attachments, - |g_id, ann| { - prior_attachments - .get(&(g_id, ann)) - .copied() - .unwrap_or([None; 3]) - }, + prior, &o_type_registry, link_ids, &mut handle_for, @@ -1254,7 +1277,7 @@ async fn seed_vector_fact_handles( } /// The attachment the base index holds for each `(g_id, reifier)`: its live -/// `f:reifies*` rows, read with one SPOT point lookup per reifier. The +/// `f:reifies*` rows, read in one batched SPOT pass per graph. The /// predicate slot's row refers to the predicate IRI as a subject; it maps /// back to the predicate id through the dictionary. async fn base_attachment_states( @@ -1266,13 +1289,8 @@ async fn base_attachment_states( keys: &[(u16, u64)], ) -> io::Result; 3]>> { use crate::run_index::resolve::link_synth::ObjectId; - use fluree_db_binary_index::format::run_record::RunSortOrder; - use fluree_db_binary_index::format::run_record_v2::RunRecordV2; - use fluree_db_binary_index::read::binary_cursor::BinaryCursor; use fluree_db_binary_index::read::binary_index_store::BinaryIndexStore; - use fluree_db_binary_index::read::column_types::{BinaryFilter, ColumnProjection, ColumnSet}; use fluree_db_core::o_type::OType; - use fluree_db_core::subject_id::SubjectId; let mut states = HashMap::new(); if keys.is_empty() || slots.iter().all(Option::is_none) { @@ -1284,79 +1302,57 @@ async fn base_attachment_states( let store = Arc::new( BinaryIndexStore::load_from_root_v6(Arc::clone(&cs), root, &cache_dir, None).await?, ); - let projection = ColumnProjection::for_scan(ColumnSet::EMPTY, false, RunSortOrder::Spot); - for &(g_id, ann) in keys { - let Some(branch) = store.branch_for_order(g_id, RunSortOrder::Spot) else { - continue; - }; - let min_key = RunRecordV2 { - s_id: SubjectId(ann), - o_key: 0, - p_id: 0, - t: 0, - o_i: 0, - o_type: 0, - g_id, - }; - let max_key = RunRecordV2 { - s_id: SubjectId(ann), - o_key: u64::MAX, - p_id: u32::MAX, - t: u32::MAX, - o_i: u32::MAX, - o_type: u16::MAX, + let iri_ref = OType::IRI_REF.as_u16(); + // Attachments name few distinct predicates; resolve each sid once. + let mut predicate_ids: HashMap = HashMap::new(); + // `keys` is sorted, so each graph's reifiers are one contiguous run. + for run in keys.chunk_by(|a, b| a.0 == b.0) { + let g_id = run[0].0; + let anns: Vec = run.iter().map(|&(_, ann)| ann).collect(); + let rows = fluree_db_binary_index::batched_lookup_subject_properties( + &store, g_id, - }; - let filter = BinaryFilter { - s_id: Some(ann), - ..Default::default() - }; - let mut cursor = BinaryCursor::new( - Arc::clone(&store), - RunSortOrder::Spot, - Arc::clone(branch), - &min_key, - &max_key, - filter, - projection, - ); - let mut state: [Option; 3] = [None; 3]; - while let Some(batch) = cursor.next_batch()? { - for i in 0..batch.row_count { - if batch.s_id.get(i) != ann { - continue; - } - let p_id = batch.p_id.get(i); + &anns, + root.index_t, + )?; + for ann in anns { + let mut state: [Option; 3] = [None; 3]; + for &(p_id, o_type, o_key) in rows.get(&ann).map_or(&[][..], Vec::as_slice) { let Some(slot) = slots.iter().position(|s| *s == Some(p_id)) else { continue; }; - let o_type = batch.o_type.get_or(i, 0); - let o_key = batch.o_key.get(i); state[slot] = match slot { - 0 if o_type == OType::IRI_REF.as_u16() => Some(SlotValue::Subject(o_key)), + 0 if o_type == iri_ref => Some(SlotValue::Subject(o_key)), // A read failure or a predicate the dictionary does not // hold must fail the build: treated as "no predicate", an // object-only re-point would replay from and to an // incomplete attachment and leave the old link in place. - 1 if o_type == OType::IRI_REF.as_u16() => { - let iri = store.resolve_subject_iri(o_key)?; - let p_id = predicates.get(&iri).ok_or_else(|| { - io::Error::new( - io::ErrorKind::InvalidData, - format!( - "reifier {ann} is attached to predicate {iri}, which the \ - dictionary does not hold" - ), - ) - })?; + 1 if o_type == iri_ref => { + let p_id = match predicate_ids.get(&o_key) { + Some(&p_id) => p_id, + None => { + let iri = store.resolve_subject_iri(o_key)?; + let p_id = predicates.get(&iri).ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + format!( + "reifier {ann} is attached to predicate {iri}, \ + which the dictionary does not hold" + ), + ) + })?; + predicate_ids.insert(o_key, p_id); + p_id + } + }; Some(SlotValue::Predicate(p_id)) } 2 => Some(SlotValue::Object(ObjectId::Typed { o_type, o_key })), _ => None, }; } + states.insert((g_id, ann), state); } - states.insert((g_id, ann), state); } Ok(states) } From 5014bf5f3eb7565c91b420184fa5c77574464940 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 19:54:14 -0400 Subject: [PATCH 13/92] chore(core): artifact read accounting behind FLUREE_IO_STATS `fluree_db_core::io_stats` counts artifact reads by kind and source, and the CLI prints the table on exit when FLUREE_IO_STATS is set. Remote fetches are counted once, at the storage bridge (`store` / `store-range`), with dictionary blobs told apart by format (string, subject and term packs, reverse-tree leaves and branches); readers count what a local path or the disk artifact cache served. FLUREE_FORCE_REMOTE_READS makes a local store answer as a remote one: no resolved local paths and no resident bytes, so every artifact goes through `get` / `get_range` and the disk cache, and the counts are what S3 would be asked for. Off, each hook is one relaxed atomic load and builds no label. --- .../src/dict/pack_reader.rs | 2 + .../src/read/binary_index_store.rs | 12 ++- fluree-db-cli/src/main.rs | 3 + fluree-db-core/src/disk_cache.rs | 12 +++ fluree-db-core/src/io_stats.rs | 79 +++++++++++++++++++ fluree-db-core/src/lib.rs | 1 + fluree-db-core/src/storage.rs | 55 ++++++++++++- 7 files changed, 159 insertions(+), 5 deletions(-) create mode 100644 fluree-db-core/src/io_stats.rs diff --git a/fluree-db-binary-index/src/dict/pack_reader.rs b/fluree-db-binary-index/src/dict/pack_reader.rs index 41126bdaff..6acd62dec7 100644 --- a/fluree-db-binary-index/src/dict/pack_reader.rs +++ b/fluree-db-binary-index/src/dict/pack_reader.rs @@ -478,6 +478,7 @@ fn fetch_and_load( // Fast paths: check if something appeared since construction. if let Some(path) = ctx.cs.resolve_local_path(pack_cid) { let backing = load_pack_backing(&path)?; + fluree_db_core::io_stats::record(|| "pack", "local", backing.bytes().len()); let meta = parse_pack_meta(backing.bytes())?; validate_lazy_meta(&meta, expected_first_id, expected_last_id, ctx)?; return Ok(LazyLoaded { meta, backing }); @@ -487,6 +488,7 @@ fn fetch_and_load( let disk_cache = ctx.cs.permits_plaintext_cache(); if disk_cache && cache_path.exists() { let backing = load_pack_backing(cache_path)?; + fluree_db_core::io_stats::record(|| "pack", "cache", backing.bytes().len()); let meta = parse_pack_meta(backing.bytes())?; validate_lazy_meta(&meta, expected_first_id, expected_last_id, ctx)?; return Ok(LazyLoaded { meta, backing }); diff --git a/fluree-db-binary-index/src/read/binary_index_store.rs b/fluree-db-binary-index/src/read/binary_index_store.rs index 32699db4cc..400801e979 100644 --- a/fluree-db-binary-index/src/read/binary_index_store.rs +++ b/fluree-db-binary-index/src/read/binary_index_store.rs @@ -3679,10 +3679,19 @@ impl ContentStoreRangeFetcher { Ok(Some(buf)) } + let kind = || { + id.content_kind().map_or_else( + || "unknown".to_string(), + |k| format!("{k:?}").to_lowercase(), + ) + }; // Try local path first — positional read. if let Some(local_path) = self.cs.resolve_local_path(id) { match read_range_from_file(&local_path, range.clone())? { - Some(buf) => return Ok(buf), + Some(buf) => { + fluree_db_core::io_stats::record(kind, "local-range", buf.len()); + return Ok(buf); + } None => { tracing::debug!( path = %local_path.display(), @@ -3697,6 +3706,7 @@ impl ContentStoreRangeFetcher { if self.cs.permits_plaintext_cache() { let cache_path = self.cache_dir.join(id.to_string()); if let Some(buf) = read_range_from_file(&cache_path, range.clone())? { + fluree_db_core::io_stats::record(kind, "cache-range", buf.len()); return Ok(buf); } } diff --git a/fluree-db-cli/src/main.rs b/fluree-db-cli/src/main.rs index 41f95993c2..f65d51c1e3 100644 --- a/fluree-db-cli/src/main.rs +++ b/fluree-db-cli/src/main.rs @@ -249,6 +249,9 @@ async fn async_main() { fluree_db_core::fd_limit::log_raise_outcome(&fd_raise); let result = fluree_db_cli::run(cli).await; + if let Some(report) = fluree_db_core::io_stats::report() { + eprintln!("{report}"); + } shutdown_tracer().await; if let Err(e) = result { diff --git a/fluree-db-core/src/disk_cache.rs b/fluree-db-core/src/disk_cache.rs index ed356f874c..d0d232343a 100644 --- a/fluree-db-core/src/disk_cache.rs +++ b/fluree-db-core/src/disk_cache.rs @@ -762,6 +762,14 @@ async fn fetch_uncached( .await } +/// The accounting label for an artifact fetched by CID alone. +fn kind_name(id: &ContentId) -> String { + id.content_kind().map_or_else( + || "unknown".to_string(), + |k| format!("{k:?}").to_lowercase(), + ) +} + pub async fn fetch_cached_bytes( cs: &dyn ContentStore, id: &ContentId, @@ -776,6 +784,7 @@ pub async fn fetch_cached_bytes( if let Some(local_path) = cs.resolve_local_path(id) { if let Some(bytes) = try_read_cached_bytes(&local_path)? { + crate::io_stats::record(|| ext, "local", bytes.len()); return Ok(bytes); } tracing::debug!( @@ -790,6 +799,7 @@ pub async fn fetch_cached_bytes( } if let Some(bytes) = try_read_cached_bytes(&cached)? { + crate::io_stats::record(|| ext, "cache", bytes.len()); return Ok(bytes); } cache @@ -812,6 +822,7 @@ pub async fn fetch_cached_bytes_cid( if let Some(local_path) = cs.resolve_local_path(id) { if let Some(bytes) = try_read_cached_bytes(&local_path)? { + crate::io_stats::record(|| kind_name(id), "local", bytes.len()); return Ok(bytes); } tracing::debug!( @@ -826,6 +837,7 @@ pub async fn fetch_cached_bytes_cid( } if let Some(bytes) = try_read_cached_bytes(&cached)? { + crate::io_stats::record(|| kind_name(id), "cache", bytes.len()); return Ok(bytes); } cache diff --git a/fluree-db-core/src/io_stats.rs b/fluree-db-core/src/io_stats.rs new file mode 100644 index 0000000000..89edadbc96 --- /dev/null +++ b/fluree-db-core/src/io_stats.rs @@ -0,0 +1,79 @@ +//! Process-wide artifact read accounting, on when `FLUREE_IO_STATS` is set. +//! +//! Every artifact read records its kind, where the bytes came from and the +//! byte count. Remote fetches are counted once, at the storage bridge +//! (`store`: a whole-object get, `store-range`: a ranged get), with dictionary +//! blobs told apart by format; readers count what a local path (`local`, +//! `local-range`) or the disk artifact cache (`cache`, `cache-range`) served. +//! `FLUREE_FORCE_REMOTE_READS` makes a local store report what S3 would be +//! asked for. Off, every hook is one relaxed atomic load and builds no label. + +use std::collections::BTreeMap; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Mutex, OnceLock}; + +static ENABLED: AtomicBool = AtomicBool::new(false); +static INIT: OnceLock<()> = OnceLock::new(); +/// `(kind, source)` → `(count, bytes)`. +type Table = BTreeMap<(String, &'static str), (u64, u64)>; +static TABLE: Mutex = Mutex::new(BTreeMap::new()); + +/// Whether accounting is on (`FLUREE_IO_STATS` set to anything but `0`). +#[inline] +pub fn enabled() -> bool { + INIT.get_or_init(|| { + let on = std::env::var("FLUREE_IO_STATS").is_ok_and(|v| v != "0"); + ENABLED.store(on, Ordering::Relaxed); + }); + ENABLED.load(Ordering::Relaxed) +} + +/// Record one read of `bytes` bytes of a `kind()` artifact from `source`. +/// `kind` runs only when accounting is on. +pub fn record>(kind: impl FnOnce() -> K, source: &'static str, bytes: usize) { + if !enabled() { + return; + } + let mut table = TABLE + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + let entry = table.entry((kind().into(), source)).or_insert((0, 0)); + entry.0 += 1; + entry.1 += bytes as u64; +} + +/// Forget everything recorded so far. +pub fn reset() { + if enabled() { + TABLE + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .clear(); + } +} + +/// The recorded reads as one line per `(kind, source)` plus a total, or +/// `None` when accounting is off or nothing was read. +pub fn report() -> Option { + if !enabled() { + return None; + } + let table = TABLE + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + if table.is_empty() { + return None; + } + let mut out = String::from("io reads (kind, source, count, bytes):\n"); + let (mut count, mut bytes) = (0u64, 0u64); + for ((kind, source), (n, b)) in table.iter() { + out.push_str(&format!(" {kind:<10} {source:<12} {n:>8} {b:>14}\n")); + count += n; + bytes += b; + } + out.push_str(&format!( + " {:<10} {:<12} {count:>8} {bytes:>14}", + "total", "" + )); + Some(out) +} diff --git a/fluree-db-core/src/lib.rs b/fluree-db-core/src/lib.rs index 78d6aa4f31..5b2bf7d108 100644 --- a/fluree-db-core/src/lib.rs +++ b/fluree-db-core/src/lib.rs @@ -55,6 +55,7 @@ pub mod graph_registry; pub mod ids; pub mod index_schema; pub mod index_stats; +pub mod io_stats; pub mod ledger_config; pub mod ledger_id; pub mod namespaces; diff --git a/fluree-db-core/src/storage.rs b/fluree-db-core/src/storage.rs index 6e1dd89356..a3776723c1 100644 --- a/fluree-db-core/src/storage.rs +++ b/fluree-db-core/src/storage.rs @@ -814,6 +814,37 @@ impl ContentStore for Arc { // StorageContentStore (bridge adapter: existing Storage → ContentStore) // ============================================================================ +/// `FLUREE_FORCE_REMOTE_READS=1` makes a local store behave like a remote +/// one for reads: no resolved local paths and no resident bytes, so every +/// artifact goes through `get` / `get_range` and the disk artifact cache, +/// exactly as it would against S3. A measurement switch; see `io_stats`. +fn force_remote_reads() -> bool { + static ON: std::sync::OnceLock = std::sync::OnceLock::new(); + *ON.get_or_init(|| std::env::var("FLUREE_FORCE_REMOTE_READS").is_ok_and(|v| v != "0")) +} + +/// Count one remote read. Dictionary blobs share a codec, so they are told +/// apart by their format magic. +fn record_read(id: &ContentId, source: &'static str, bytes: &[u8]) { + if !crate::io_stats::enabled() { + return; + } + let label = match (id.content_kind(), bytes.get(..4)) { + (Some(ContentKind::DictBlob { .. }), Some(b"FPK1")) => match bytes.get(5) { + Some(0) => "string-pack".to_string(), + Some(1) => "subject-pack".to_string(), + Some(2) => "term-pack".to_string(), + _ => "pack".to_string(), + }, + (Some(ContentKind::DictBlob { .. }), Some(b"DLR1")) => "dict-leaf".to_string(), + (Some(ContentKind::DictBlob { .. }), Some(b"DTB1")) => "dict-branch".to_string(), + (Some(ContentKind::DictBlob { .. }), _) => "dictblob".to_string(), + (Some(k), _) => format!("{k:?}").to_lowercase(), + (None, _) => "unknown".to_string(), + }; + crate::io_stats::record(|| label, source, bytes.len()); +} + /// Bridge adapter that wraps an existing `S: Storage` to provide `ContentStore`. /// /// This is the critical piece for incremental migration: code that already has @@ -953,17 +984,24 @@ impl ContentStore for StorageContentStore { // backend called it absent. Rebuilding a bare `not_found(address)` here // throws both away, and this is the error a caller actually sees. let primary = match self.storage.read_bytes(&address).await { - Ok(bytes) => return Ok(bytes), + Ok(bytes) => { + record_read(id, "store", &bytes); + return Ok(bytes); + } Err(crate::error::Error::NotFound(reason)) => reason, Err(e) => return Err(e), }; // Fallback: dicts moved from per-branch to @shared namespace if let Some(legacy) = self.legacy_dict_address(id) { - return self.storage.read_bytes(&legacy).await; + let bytes = self.storage.read_bytes(&legacy).await?; + record_read(id, "store", &bytes); + return Ok(bytes); } // Fallback: index roots stored with .json before .fir6 rename if let Some(legacy) = self.legacy_index_root_address(id) { - return self.storage.read_bytes(&legacy).await; + let bytes = self.storage.read_bytes(&legacy).await?; + record_read(id, "store", &bytes); + return Ok(bytes); } Err(crate::error::Error::not_found(primary)) } @@ -1062,6 +1100,9 @@ impl ContentStore for StorageContentStore { } fn resolve_cached_bytes(&self, id: &ContentId) -> Option> { + if force_remote_reads() { + return None; + } // CID-keyed straight through — no address formatting on the lookup. // A resident tier indexes by CID regardless of which (current or // legacy) address the bytes were fetched from. @@ -1081,6 +1122,9 @@ impl ContentStore for StorageContentStore { } fn resolve_local_path(&self, id: &ContentId) -> Option { + if force_remote_reads() { + return None; + } let address = self.cid_to_address(id).ok()?; if let Some(path) = self.storage.resolve_local_path(&address) { return Some(path); @@ -1100,7 +1144,10 @@ impl ContentStore for StorageContentStore { let address = self.cid_to_address(id)?; // Same reason-preservation as `get` above. let primary = match self.storage.read_byte_range(&address, range.clone()).await { - Ok(bytes) => return Ok(bytes), + Ok(bytes) => { + record_read(id, "store-range", &bytes); + return Ok(bytes); + } Err(crate::error::Error::NotFound(reason)) => reason, Err(e) => return Err(e), }; From c308c54d59982938961b75fb095372e96ea541b2 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 20:09:20 -0400 Subject: [PATCH 14/92] test(index): count the triple-term entry in the built-in o_type table OType::TRIPLE_TERM joined the Fluree-reserved entries, so the built-in table holds 45 entries, not 44. --- fluree-db-binary-index/src/format/index_root.rs | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/fluree-db-binary-index/src/format/index_root.rs b/fluree-db-binary-index/src/format/index_root.rs index aecbbb4f47..339d2998e8 100644 --- a/fluree-db-binary-index/src/format/index_root.rs +++ b/fluree-db-binary-index/src/format/index_root.rs @@ -1949,8 +1949,8 @@ mod tests { #[test] fn o_type_table_built_in() { let table = IndexRoot::build_o_type_table(&[], &[]); - // Should contain all 31 embedded + 13 Fluree = 44 entries. - assert_eq!(table.len(), 44); + // Should contain all 31 embedded + 14 Fluree = 45 entries. + assert_eq!(table.len(), 45); // Spot-check a few entries. let int_entry = table @@ -1979,8 +1979,8 @@ mod tests { #[test] fn o_type_table_with_langs() { let table = IndexRoot::build_o_type_table(&[], &["en".to_string(), "fr".to_string()]); - // 44 built-in + 2 langString = 46. - assert_eq!(table.len(), 46); + // 45 built-in + 2 langString = 47. + assert_eq!(table.len(), 47); // lang_id is 1-based: first tag "en" gets lang_id=1 let en_entry = table @@ -1994,8 +1994,8 @@ mod tests { #[test] fn o_type_table_with_custom_types() { let table = IndexRoot::build_o_type_table(&["http://example.org/myType".to_string()], &[]); - // 44 built-in + 1 customer = 45. - assert_eq!(table.len(), 45); + // 45 built-in + 1 customer = 46. + assert_eq!(table.len(), 46); let custom = table.last().unwrap(); assert!(OType::from_u16(custom.o_type).is_customer_datatype()); From 326e455d7b7ccb1ba818cd330aecb01f01981340 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 20:09:20 -0400 Subject: [PATCH 15/92] perf(index): resolve base class Sids in one batched pass on incremental builds Phase 3b seeds every class entry of the base stats and resolves each class Sid to its sid64 through the subject reverse tree, one lookup per class. Without a leaf cache each lookup rereads a whole ~2.3 MB leaf, so a ledger with many classes paid it on every incremental pass: 203,269 classes took 21 s on the 10% annotation benchmark slice, whatever the size of the window. `BinaryIndexStore::find_subject_ids_by_parts` resolves them together through `reverse_lookup_many`, reading each leaf once. A failed read now aborts to a full rebuild, as the base PSOT class lookup beside it does, instead of dropping the class from the published stats. --- .../src/read/binary_index_store.rs | 17 ++++++ fluree-db-indexer/src/build/incremental.rs | 54 ++++++++++--------- 2 files changed, 46 insertions(+), 25 deletions(-) diff --git a/fluree-db-binary-index/src/read/binary_index_store.rs b/fluree-db-binary-index/src/read/binary_index_store.rs index 400801e979..e033e53b91 100644 --- a/fluree-db-binary-index/src/read/binary_index_store.rs +++ b/fluree-db-binary-index/src/read/binary_index_store.rs @@ -2332,6 +2332,23 @@ impl BinaryIndexStore { } } + /// [`Self::find_subject_id_by_parts`] for many subjects, reading each + /// reverse-tree leaf once. + pub fn find_subject_ids_by_parts(&self, parts: &[(u16, &str)]) -> io::Result>> { + match &self.dicts.subject_reverse_tree { + Some(tree) => { + let keys: Vec> = parts + .iter() + .map(|&(ns_code, suffix)| { + crate::dict::reverse_leaf::subject_reverse_key(ns_code, suffix.as_bytes()) + }) + .collect(); + tree.reverse_lookup_many(keys.iter().map(Vec::as_slice)) + } + None => Ok(vec![None; parts.len()]), + } + } + /// Find all subject IDs whose suffix starts with `prefix` within a namespace. /// /// Uses a range scan on the reverse subject tree: scans the key range diff --git a/fluree-db-indexer/src/build/incremental.rs b/fluree-db-indexer/src/build/incremental.rs index aab8d530a3..d6e805055e 100644 --- a/fluree-db-indexer/src/build/incremental.rs +++ b/fluree-db-indexer/src/build/incremental.rs @@ -3352,31 +3352,35 @@ pub async fn incremental_index( } } - // Seed from base root per-graph class entries. - if let Some(ref base_stats) = base_root.stats { - if let Some(ref graphs) = base_stats.graphs { - for g in graphs { - if let Some(ref classes) = g.classes { - for entry in classes { - // Try to resolve class Sid → sid64 via store. - base_class_sid_lookups.fetch_add(1, Ordering::Relaxed); - let sid64 = store_opt - .as_ref() - .and_then(|s| { - s.find_subject_id_by_parts( - entry.class_sid.namespace_code, - &entry.class_sid.name, - ) - .ok() - .flatten() - }) - .unwrap_or(0); - if sid64 != 0 { - base_class_sid_hits.fetch_add(1, Ordering::Relaxed); - entries_by_key.insert((g.g_id, sid64), entry.clone()); - } - } - } + // Seed from base root per-graph class entries, resolving every + // class Sid → sid64 in one batched reverse-tree pass (a lookup + // per class rereads a whole leaf). + if let (Some(store), Some(graphs)) = ( + store_opt.as_ref(), + base_root.stats.as_ref().and_then(|s| s.graphs.as_ref()), + ) { + let entries: Vec<(u16, &is::ClassStatEntry)> = graphs + .iter() + .filter_map(|g| g.classes.as_ref().map(|c| (g.g_id, c))) + .flat_map(|(g_id, classes)| classes.iter().map(move |e| (g_id, e))) + .collect(); + let parts: Vec<(u16, &str)> = entries + .iter() + .map(|(_, e)| (e.class_sid.namespace_code, e.class_sid.name.as_ref())) + .collect(); + base_class_sid_lookups.fetch_add(parts.len(), Ordering::Relaxed); + // A failed read would drop the class from the published + // stats; recompute them in a full rebuild instead. + let sid64s = store.find_subject_ids_by_parts(&parts).map_err(|e| { + IndexerError::IncrementalAbort(format!( + "Phase 3b base class Sid lookup failed: {e} (deferring \ + class-stat recompute to full rebuild)" + )) + })?; + for ((g_id, entry), sid64) in entries.into_iter().zip(sid64s) { + if let Some(sid64) = sid64.filter(|&s| s != 0) { + base_class_sid_hits.fetch_add(1, Ordering::Relaxed); + entries_by_key.insert((g_id, sid64), entry.clone()); } } } From d7760a1104b10f03c59d1df7405cd4c7a50dbfae Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 20:09:20 -0400 Subject: [PATCH 16/92] perf(api): read the index root once when loading a ledger Loading an indexed ledger read the root twice: `LedgerState::load` for the snapshot metadata, then the binary store load for the full decode. Against a remote store that is a second whole-object GET of the largest single artifact on every cold load (83 MB twice on the annotation benchmark slice). `LedgerState::load_with_root` / `load_with_store_and_root` hand back the root they read, and `load_and_attach_binary_store` takes it when its CID is the one the record names. Every load-then-attach path passes it through. --- .../src/indexer_fulltext_provider.rs | 8 +- fluree-db-api/src/ledger/loading.rs | 19 ++- fluree-db-api/src/ledger_manager.rs | 37 +++-- fluree-db-api/tests/grp_index.rs | 2 + fluree-db-api/tests/it_index_root_reads.rs | 144 ++++++++++++++++++ fluree-db-ledger/src/lib.rs | 53 ++++++- 6 files changed, 234 insertions(+), 29 deletions(-) create mode 100644 fluree-db-api/tests/it_index_root_reads.rs diff --git a/fluree-db-api/src/indexer_fulltext_provider.rs b/fluree-db-api/src/indexer_fulltext_provider.rs index b942277279..fed28973b2 100644 --- a/fluree-db-api/src/indexer_fulltext_provider.rs +++ b/fluree-db-api/src/indexer_fulltext_provider.rs @@ -138,9 +138,10 @@ impl ApiFulltextConfigProvider { } // 1. Load ledger state (snapshot + novelty). - let mut state = LedgerState::load(self.nameservice.as_ref(), ledger_id, &self.backend) - .await - .map_err(|e| format!("LedgerState::load: {e}"))?; + let (mut state, root) = + LedgerState::load_with_root(self.nameservice.as_ref(), ledger_id, &self.backend) + .await + .map_err(|e| format!("LedgerState::load: {e}"))?; // 2. If an index exists, load the binary store so the config graph // can be read via the indexed side too. Without this, only @@ -154,6 +155,7 @@ impl ApiFulltextConfigProvider { &self.cache_dir, Some(Arc::clone(&self.leaflet_cache)), None, + root, ) .await .map_err(|e| format!("load binary index store: {e}"))?; diff --git a/fluree-db-api/src/ledger/loading.rs b/fluree-db-api/src/ledger/loading.rs index 0b532b4733..20e2daedbd 100644 --- a/fluree-db-api/src/ledger/loading.rs +++ b/fluree-db-api/src/ledger/loading.rs @@ -6,13 +6,19 @@ use crate::{ApiError, Fluree, HistoricalLedgerView, LedgerState, Result, TimeSpe use fluree_db_core::ContentStore; use fluree_db_core::LedgerId; use fluree_db_core::{collect_first_parent_cids, load_commit_envelope_by_id, CommitId, ContentId}; +use fluree_db_ledger::LoadedIndexRoot; use fluree_db_nameservice::{NameServiceError, NsRecord}; use fluree_db_query::QueryError; impl Fluree { /// Attach the binary index store and range provider to an already-loaded /// ledger state when its nameservice record points at a binary index root. - pub(crate) async fn attach_index(&self, state: &mut LedgerState) -> Result<()> { + /// `root` is the index root the state load already read, if any. + pub(crate) async fn attach_index( + &self, + state: &mut LedgerState, + root: Option, + ) -> Result<()> { crate::ledger_manager::load_and_attach_binary_store( self.backend(), self.nameservice(), @@ -20,6 +26,7 @@ impl Fluree { &self.binary_store_cache_dir(), Some(Arc::clone(self.leaflet_cache())), None, + root, ) .await?; Ok(()) @@ -54,8 +61,8 @@ impl Fluree { where C: ContentStore + Clone + 'static, { - let mut state = LedgerState::load_with_store(store, record).await?; - self.attach_index(&mut state).await?; + let (mut state, root) = LedgerState::load_with_store_and_root(store, record).await?; + self.attach_index(&mut state, root).await?; Ok(state) } @@ -79,9 +86,9 @@ impl Fluree { /// same lock) would deadlock the transact path — and the failure mode is /// a hang, not a red test. Anything added here must stay manager-free. pub(crate) async fn load_ledger_uncached(&self, ledger_id: &str) -> Result { - let mut state = - LedgerState::load(&self.nameservice_mode, ledger_id, self.backend()).await?; - self.attach_index(&mut state).await?; + let (mut state, root) = + LedgerState::load_with_root(&self.nameservice_mode, ledger_id, self.backend()).await?; + self.attach_index(&mut state, root).await?; // Default context is not loaded here. Opt-in callers route through // `Fluree::db_with_default_context` / `db_at_with_default_context`, // which fetch and attach the context onto the returned `GraphDb`. diff --git a/fluree-db-api/src/ledger_manager.rs b/fluree-db-api/src/ledger_manager.rs index 2f65dd2334..77ff820134 100644 --- a/fluree-db-api/src/ledger_manager.rs +++ b/fluree-db-api/src/ledger_manager.rs @@ -33,7 +33,7 @@ use fluree_db_core::dict_novelty::DictNovelty; use fluree_db_core::ledger_config::LedgerConfig; use fluree_db_core::trace_first_parent_commits_by_id; use fluree_db_core::{ContentId, ContentStore, LedgerId, Sid, StorageBackend}; -use fluree_db_ledger::{LedgerState, TypeErasedStore}; +use fluree_db_ledger::{LedgerState, LoadedIndexRoot, TypeErasedStore}; use fluree_db_nameservice::NsRecord; use rustc_hash::FxHashSet; use std::collections::VecDeque; @@ -1144,7 +1144,8 @@ async fn prefetch_novelty_translation( /// 404 on a fresh branch that hasn't yet had its own index built. /// /// `prev` is the store this load replaces, if any (a reload of a cached -/// ledger); artifacts the new root shares with it are carried over. +/// ledger); artifacts the new root shares with it are carried over. `root` is +/// the index root the state load already read, if any. pub(crate) async fn load_and_attach_binary_store( backend: &StorageBackend, nameservice: &dyn fluree_db_nameservice::NameServiceLookup, @@ -1152,6 +1153,7 @@ pub(crate) async fn load_and_attach_binary_store( cache_dir: &std::path::Path, leaflet_cache: Option>, prev: Option<&BinaryIndexStore>, + root: Option, ) -> std::result::Result>, ApiError> { let record = match state.ns_record.as_ref() { Some(r) => r, @@ -1171,10 +1173,13 @@ pub(crate) async fn load_and_attach_binary_store( fluree_db_nameservice::branched_content_store_for_record(backend, nameservice, record) .await?; let root_started = Instant::now(); - let bytes = cs - .get(&index_cid) - .await - .map_err(|e| ApiError::internal(format!("failed to read index root: {e}")))?; + let bytes = match root { + Some(root) if root.id == index_cid => root.bytes, + _ => cs + .get(&index_cid) + .await + .map_err(|e| ApiError::internal(format!("failed to read index root: {e}")))?, + }; let root_read_us = root_started.elapsed().as_micros() as u64; let decode_started = Instant::now(); @@ -1645,9 +1650,10 @@ impl LedgerManager { // ledger cache into a global mutex for the duration of any cold load. // Note: we pass the original address to nameservice (it handles // resolution), but cache under the canonical address. - let load_result = LedgerState::load(&self.nameservice_mode, ledger_id, &self.backend) - .await - .map_err(ApiError::from); // Convert LedgerError to ApiError + let load_result = + LedgerState::load_with_root(&self.nameservice_mode, ledger_id, &self.backend) + .await + .map_err(ApiError::from); // Convert LedgerError to ApiError tracing::debug!( alias = %canonical_alias, ok = load_result.is_ok(), @@ -1655,7 +1661,7 @@ impl LedgerManager { ); let publish = match load_result { - Ok(mut state) => { + Ok((mut state, root)) => { // Attempt to load binary index store (v2 only). // Non-fatal: if loading fails, log and continue without binary index. let binary_store = match load_and_attach_binary_store( @@ -1665,6 +1671,7 @@ impl LedgerManager { &self.config.cache_dir, self.config.leaflet_cache.clone(), None, + root, ) .await { @@ -1928,11 +1935,12 @@ impl LedgerManager { // not held over any of this and is acquired after the swap, so // the two locks never overlap (avoids the entries↔state ordering // hazard with `current_t`). - let loaded = LedgerState::load(&self.nameservice_mode, ledger_id, &self.backend) - .await - .map_err(ApiError::from); + let loaded = + LedgerState::load_with_root(&self.nameservice_mode, ledger_id, &self.backend) + .await + .map_err(ApiError::from); let result = match loaded { - Ok(mut new_state) => { + Ok((mut new_state, root)) => { // Attempt to load binary index store (v2 only) — still off-lock. // Artifacts shared with the store being replaced are // carried over; the slot is read on its own (no @@ -1945,6 +1953,7 @@ impl LedgerManager { &self.config.cache_dir, self.config.leaflet_cache.clone(), prev_store.as_deref(), + root, ) .await { diff --git a/fluree-db-api/tests/grp_index.rs b/fluree-db-api/tests/grp_index.rs index 16714b5ade..a3fba085fa 100644 --- a/fluree-db-api/tests/grp_index.rs +++ b/fluree-db-api/tests/grp_index.rs @@ -7,6 +7,8 @@ mod it_custom_datatype_limit; mod it_duration_index_roundtrip; #[path = "it_fwd_pack_compaction.rs"] mod it_fwd_pack_compaction; +#[path = "it_index_root_reads.rs"] +mod it_index_root_reads; #[path = "it_index_sweep.rs"] mod it_index_sweep; #[path = "it_indexing_fuel.rs"] diff --git a/fluree-db-api/tests/it_index_root_reads.rs b/fluree-db-api/tests/it_index_root_reads.rs new file mode 100644 index 0000000000..d7040597e5 --- /dev/null +++ b/fluree-db-api/tests/it_index_root_reads.rs @@ -0,0 +1,144 @@ +//! Loading an indexed ledger reads its index root once. The state load and the +//! binary store both need the root; the second one takes the bytes the first +//! read instead of fetching them again (one more whole-object GET per cold +//! load against a remote store). + +use crate::support; +use async_trait::async_trait; +use fluree_db_api::{FlureeBuilder, NameServiceMode}; +use fluree_db_core::storage::{ + ContentAddressedWrite, ContentWriteResult, MemoryStorage, StorageMethod, StorageRead, + StorageWrite, +}; +use fluree_db_core::ContentId; +use fluree_db_nameservice::memory::MemoryNameService; +use serde_json::json; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::Arc; + +const LEDGER: &str = "it/index-root-reads:main"; + +/// Memory storage that counts reads of index roots. +#[derive(Debug, Clone)] +struct RootReadCounting { + inner: MemoryStorage, + root_reads: Arc, +} + +impl RootReadCounting { + fn count(&self, address: &str) { + if address.contains("/index/roots/") { + self.root_reads.fetch_add(1, Ordering::Relaxed); + } + } + + fn take(&self) -> usize { + self.root_reads.swap(0, Ordering::Relaxed) + } +} + +#[async_trait] +impl StorageRead for RootReadCounting { + fn permits_plaintext_cache(&self) -> bool { + self.inner.permits_plaintext_cache() + } + + fn encryption_admin(&self) -> Option> { + self.inner.encryption_admin() + } + + async fn read_bytes(&self, address: &str) -> fluree_db_core::Result> { + self.count(address); + self.inner.read_bytes(address).await + } + + async fn read_byte_range( + &self, + address: &str, + range: std::ops::Range, + ) -> fluree_db_core::Result> { + self.count(address); + self.inner.read_byte_range(address, range).await + } + + async fn exists(&self, address: &str) -> fluree_db_core::Result { + self.inner.exists(address).await + } + + async fn list_prefix(&self, prefix: &str) -> fluree_db_core::Result> { + self.inner.list_prefix(prefix).await + } + + fn resolve_cached_bytes(&self, id: &ContentId) -> Option> { + self.inner.resolve_cached_bytes(id) + } +} + +#[async_trait] +impl StorageWrite for RootReadCounting { + async fn write_bytes(&self, address: &str, bytes: &[u8]) -> fluree_db_core::Result<()> { + self.inner.write_bytes(address, bytes).await + } + + async fn delete(&self, address: &str) -> fluree_db_core::Result<()> { + self.inner.delete(address).await + } +} + +#[async_trait] +impl ContentAddressedWrite for RootReadCounting { + async fn content_write_bytes_with_hash( + &self, + kind: fluree_db_core::content_kind::ContentKind, + ledger_id: &str, + content_hash_hex: &str, + bytes: &[u8], + ) -> fluree_db_core::Result { + self.inner + .content_write_bytes_with_hash(kind, ledger_id, content_hash_hex, bytes) + .await + } +} + +impl StorageMethod for RootReadCounting { + fn storage_method(&self) -> &str { + self.inner.storage_method() + } +} + +#[tokio::test] +async fn loading_an_indexed_ledger_reads_its_root_once() { + let storage = RootReadCounting { + inner: MemoryStorage::new(), + root_reads: Arc::new(AtomicUsize::new(0)), + }; + let ns = Arc::new(MemoryNameService::new()); + let mode = || { + NameServiceMode::ReadWrite( + ns.clone() as Arc + ) + }; + let seed = FlureeBuilder::memory().build_with(storage.clone(), mode()); + let ledger = support::genesis_ledger_for_fluree(&seed, LEDGER); + seed.insert( + ledger, + &json!({ + "@context": { "ex": "http://example.org/" }, + "@graph": [{ "@id": "ex:a", "@type": "ex:T", "ex:name": "a" }] + }), + ) + .await + .expect("seed"); + support::rebuild_and_publish_index(&seed, LEDGER).await; + + // Fresh instances, so nothing is cached in-process. + let fluree = FlureeBuilder::memory().build_with(storage.clone(), mode()); + storage.take(); + let state = fluree.ledger(LEDGER).await.expect("uncached load"); + assert!(state.snapshot.t > 0, "the load must come from the index"); + assert_eq!(storage.take(), 1, "uncached load"); + + let fluree = FlureeBuilder::memory().build_with(storage.clone(), mode()); + fluree.ledger_cached(LEDGER).await.expect("cached load"); + assert_eq!(storage.take(), 1, "ledger-manager load"); +} diff --git a/fluree-db-ledger/src/lib.rs b/fluree-db-ledger/src/lib.rs index 7ec50ea530..5220c3b957 100644 --- a/fluree-db-ledger/src/lib.rs +++ b/fluree-db-ledger/src/lib.rs @@ -118,6 +118,13 @@ impl HeadTemporal { } } +/// An index root read while loading a [`LedgerState`], for a caller that +/// decodes the whole root next (the binary index store). +pub struct LoadedIndexRoot { + pub id: ContentId, + pub bytes: Vec, +} + /// Ledger state combining indexed LedgerSnapshot with novelty overlay /// /// Provides a consistent view of the ledger by combining: @@ -198,6 +205,17 @@ impl LedgerState { ledger_id: &str, backend: &StorageBackend, ) -> Result { + Ok(Self::load_with_root(ns, ledger_id, backend).await?.0) + } + + /// [`Self::load`], also handing back the index root it read, so a caller + /// that decodes the whole root next (the binary index store) does not + /// fetch it a second time. + pub async fn load_with_root( + ns: &dyn NameServiceLookup, + ledger_id: &str, + backend: &StorageBackend, + ) -> Result<(Self, Option)> { let record = ns .lookup(ledger_id) .await? @@ -217,11 +235,11 @@ impl LedgerState { // ancestor namespaces for pre-branch-point content. if record.source_branch.is_some() { let store = Self::build_branched_store(ns, &record, backend).await?; - return Self::load_with_store(store, record).await; + return Self::load_with_store_and_root(store, record).await; } let store = backend.content_store(&record.ledger_id); - Self::load_with_store(store, record).await + Self::load_with_store_and_root(store, record).await } /// Build a recursive `BranchedContentStore` by walking the branch ancestry. @@ -245,12 +263,35 @@ impl LedgerState { pub async fn load_with_store( store: C, record: NsRecord, + ) -> Result { + Ok(Self::load_with_store_and_root(store, record).await?.0) + } + + /// [`Self::load_with_store`], also handing back the index root it read. + pub async fn load_with_store_and_root( + store: C, + record: NsRecord, + ) -> Result<(Self, Option)> { + let root = match &record.index_head_id { + Some(id) => Some(LoadedIndexRoot { + id: id.clone(), + bytes: store.get(id).await?, + }), + None => None, + }; + let state = Self::load_from_root(store, record, root.as_ref()).await?; + Ok((state, root)) + } + + async fn load_from_root( + store: C, + record: NsRecord, + root: Option<&LoadedIndexRoot>, ) -> Result { // Handle missing index (genesis fallback) - let (mut snapshot, mut dict_novelty) = match &record.index_head_id { - Some(index_cid) => { - let root_bytes = store.get(index_cid).await?; - let loaded = LedgerSnapshot::from_root_bytes(&root_bytes)?; + let (mut snapshot, mut dict_novelty) = match root { + Some(root) => { + let loaded = LedgerSnapshot::from_root_bytes(&root.bytes)?; let dn = DictNovelty::with_watermarks( loaded.subject_watermarks.clone(), loaded.string_watermark, From b899450d8fd8737dbeb2562f4197e11e62c76df2 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 20:09:20 -0400 Subject: [PATCH 17/92] perf(index): stop storing the ledger-wide class table twice in the root The root's stats carry per-graph class tables and a ledger-wide one, and both build paths set the ledger-wide table to `union_per_graph_classes` of the per-graph ones. On a ledger with many classes the two copies are most of the root: 41.5 MB each for 203,269 classes on the annotation benchmark slice. The encoder now writes the ledger-wide table only when no graph carries classes, and both decoders derive it from the graphs when it is absent. Roots that still carry it decode as before. Import roots, which never set it, now present the same class table a rebuild would, instead of none until the next reindex. A reader without this change sees no ledger-wide class table on a new root and plans without class statistics. --- .../src/format/stats_wire.rs | 71 ++++++++++++++++++- fluree-db-core/src/stats_wire.rs | 3 +- 2 files changed, 70 insertions(+), 4 deletions(-) diff --git a/fluree-db-binary-index/src/format/stats_wire.rs b/fluree-db-binary-index/src/format/stats_wire.rs index 6305947467..c7539ff5f6 100644 --- a/fluree-db-binary-index/src/format/stats_wire.rs +++ b/fluree-db-binary-index/src/format/stats_wire.rs @@ -9,6 +9,8 @@ //! - Graphs sorted by `g_id`, properties by `p_id` //! - Aggregate properties sorted by `(ns_code, suffix)` //! - Classes sorted by `(ns_code, suffix)`, properties within classes likewise +//! - The ledger-wide class list is written only when no graph carries classes; +//! otherwise decoders derive it as the union of the graphs' lists //! - Historical tail entries sorted by sid / `(g_id, p_id)`, tags sorted //! //! An optional historical-datatypes tail follows the classes section — see @@ -199,8 +201,14 @@ pub fn encode_stats(stats: &IndexStats) -> Vec { encode_datatypes(&mut buf, &p.datatypes); } - // Classes - let classes = stats.classes.as_deref().unwrap_or(&[]); + // Classes. The ledger-wide table is the union of the graphs' tables + // (`union_per_graph_classes`), so once graphs carry classes it is derived + // on decode rather than stored a second time. + let graphs_carry_classes = graphs.iter().any(|g| g.classes.is_some()); + let classes = match stats.classes.as_deref() { + Some(classes) if !graphs_carry_classes => classes, + _ => &[], + }; let mut sorted_classes: Vec<&ClassStatEntry> = classes.iter().collect(); sorted_classes.sort_by(|a, b| a.class_sid.cmp(&b.class_sid)); @@ -868,7 +876,7 @@ pub fn decode_stats_with_len(data: &[u8]) -> io::Result<(IndexStats, usize)> { Some(agg_props) }, classes: if classes.is_empty() { - None + fluree_db_core::index_stats::union_per_graph_classes(&graphs) } else { Some(classes) }, @@ -1149,6 +1157,63 @@ mod tests { assert_eq!(classes[0].properties[1].ref_classes.len(), 0); } + /// The ledger-wide class list is the union of the graphs' lists, so a root + /// does not store it twice: the bytes are those of a root without it, and + /// both decoders hand back the union, also for an import root that never + /// set it. + #[test] + fn the_ledger_wide_class_list_is_derived_from_the_graphs() { + let class = |name: &str, count: u64| ClassStatEntry { + class_sid: sid(5, name), + count, + properties: vec![ClassPropertyUsage { + property_sid: sid(5, "knows"), + datatypes: vec![(1, count)], + langs: vec![], + ref_classes: vec![ClassRefCount { + class_sid: sid(5, "Person"), + count, + }], + }], + }; + let graph = |g_id: u16, classes: Vec| GraphStatsEntry { + g_id, + flakes: 10, + size: 0, + properties: vec![], + classes: Some(classes), + }; + let graphs = vec![ + graph(0, vec![class("Org", 2), class("Person", 3)]), + graph(1, vec![class("Person", 4)]), + ]; + let union = fluree_db_core::index_stats::union_per_graph_classes(&graphs); + let built = IndexStats { + flakes: 20, + size: 0, + properties: None, + classes: union.clone(), + graphs: Some(graphs), + historical_since_t: None, + }; + let imported = IndexStats { + classes: None, + ..built.clone() + }; + + let bytes = encode_stats(&built); + assert_eq!(bytes, encode_stats(&imported), "the union is not stored"); + let want = format!("{union:?}"); + assert!( + want.contains("count: 7"), + "Person summed across graphs: {want}" + ); + assert_eq!(format!("{:?}", decode_stats(&bytes).unwrap().classes), want); + let (via_core, consumed) = fluree_db_core::stats_wire::decode_stats(&bytes).unwrap(); + assert_eq!(consumed, bytes.len()); + assert_eq!(format!("{:?}", via_core.classes), want); + } + #[test] fn test_stats_determinism() { let stats = IndexStats { diff --git a/fluree-db-core/src/stats_wire.rs b/fluree-db-core/src/stats_wire.rs index dfa76ff897..cf94972833 100644 --- a/fluree-db-core/src/stats_wire.rs +++ b/fluree-db-core/src/stats_wire.rs @@ -417,8 +417,9 @@ pub fn decode_stats(data: &[u8]) -> io::Result<(IndexStats, usize)> { } else { Some(agg_props) }, + // Not stored when graphs carry classes; see the binary-index encoder. classes: if classes.is_empty() { - None + crate::index_stats::union_per_graph_classes(&graphs) } else { Some(classes) }, From af58b71bbcb2661db04f130291c7cb26b8b1b91f Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 20:18:39 -0400 Subject: [PATCH 18/92] perf(index): store per-graph class tables in a compact stats tail Each class-property usage spelled out its property Sid, and each ref-class count its class Sid, with fixed-width counts. On a ledger with many classes that is nearly the whole root: 203,269 classes, 758,489 usages and 347,070 ref counts took 41.5 MB on the annotation benchmark slice, though only 607 predicates appear among the usages. The class tables now travel in an appended stats section (tag 2): every Sid written once in a sorted, front-coded table, entries referring to it by index, counts as varints. Re-encoding the slice's 83 MB root gives 8.4 MB with identical decoded stats, and decoded entries share their Sid names. Readers skip appended sections they do not know, so the legacy per-graph class slots are left empty and a reader without this change sees no class tables rather than misparsing. Roots that carry classes in the legacy slots still decode. The binary-index stats decoder now delegates to the core one, so the format has a single parser. --- .../src/format/stats_wire.rs | 594 ++++++------------ fluree-db-core/src/stats_wire.rs | 246 +++++++- 2 files changed, 436 insertions(+), 404 deletions(-) diff --git a/fluree-db-binary-index/src/format/stats_wire.rs b/fluree-db-binary-index/src/format/stats_wire.rs index c7539ff5f6..f1f219fed0 100644 --- a/fluree-db-binary-index/src/format/stats_wire.rs +++ b/fluree-db-binary-index/src/format/stats_wire.rs @@ -69,25 +69,6 @@ fn read_sid(data: &[u8], pos: usize) -> io::Result<(Sid, usize)> { Ok((Sid::new(ns_code, suffix), p)) } -/// Decode a `(ns_code, suffix_string)` tuple. Returns `((ns_code, suffix), bytes_consumed)`. -fn read_sid_tuple(data: &[u8], pos: usize) -> io::Result<((u16, String), usize)> { - let mut p = pos; - ensure_len(data, p, 4, "sid tuple header")?; - let ns_code = u16::from_le_bytes(data[p..p + 2].try_into().unwrap()); - p += 2; - let suffix_len = u16::from_le_bytes(data[p..p + 2].try_into().unwrap()) as usize; - p += 2; - ensure_len(data, p, suffix_len, "sid tuple suffix")?; - let suffix = std::str::from_utf8(&data[p..p + suffix_len]).map_err(|e| { - io::Error::new( - io::ErrorKind::InvalidData, - format!("invalid UTF-8 in sid tuple: {e}"), - ) - })?; - p += suffix_len; - Ok(((ns_code, suffix.to_string()), p)) -} - /// Check that `data[pos..pos+need]` is within bounds. #[inline] fn ensure_len(data: &[u8], pos: usize, need: usize, ctx: &str) -> io::Result<()> { @@ -104,15 +85,6 @@ fn ensure_len(data: &[u8], pos: usize, need: usize, ctx: &str) -> io::Result<()> } } -/// Read a u8 at `pos`, advancing. -#[inline] -fn read_u8(data: &[u8], pos: &mut usize) -> io::Result { - ensure_len(data, *pos, 1, "u8")?; - let v = data[*pos]; - *pos += 1; - Ok(v) -} - /// Read a u16 LE at `pos`, advancing. #[inline] fn read_u16(data: &[u8], pos: &mut usize) -> io::Result { @@ -131,15 +103,6 @@ fn read_u32(data: &[u8], pos: &mut usize) -> io::Result { Ok(v) } -/// Read a u64 LE at `pos`, advancing. -#[inline] -fn read_u64(data: &[u8], pos: &mut usize) -> io::Result { - ensure_len(data, *pos, 8, "u64")?; - let v = u64::from_le_bytes(data[*pos..*pos + 8].try_into().unwrap()); - *pos += 8; - Ok(v) -} - /// Read an i64 LE at `pos`, advancing. #[inline] fn read_i64(data: &[u8], pos: &mut usize) -> io::Result { @@ -182,8 +145,10 @@ pub fn encode_stats(stats: &IndexStats) -> Vec { encode_graph_property(&mut buf, p); } - // Per-graph classes (optional) - encode_optional_classes(&mut buf, g.classes.as_deref()); + // Per-graph classes travel in the compact class tail; the legacy slot + // stays empty so readers that predate the tail see no classes rather + // than misparse. + buf.push(0); } // Aggregate properties (SID-keyed) @@ -229,6 +194,9 @@ pub fn encode_stats(stats: &IndexStats) -> Vec { // Historical tail (see `encode_historical_tail` for the evolution rules). encode_historical_tail(&mut buf, stats); + if graphs_carry_classes { + fluree_db_core::stats_wire::encode_class_tail(&mut buf, &sorted_graphs); + } buf } @@ -393,146 +361,6 @@ fn encode_datatypes(buf: &mut Vec, datatypes: &[(u8, u64)]) { } } -/// Encode optional per-graph classes. -/// -/// Wire format: -/// ```text -/// [has_classes: u8] (0 = absent, 1 = present) -/// if has_classes == 1: -/// [class_count: u32 LE] -/// for each class: -/// [class_sid encoded] -/// [instance_count: u64 LE] -/// [property_count: u16 LE] -/// for each property: -/// [property_sid encoded] -/// [ref_class_count: u16 LE] -/// for each ref_class: -/// [ref_class_sid encoded] -/// [count: u64 LE] -/// ``` -fn encode_optional_classes(buf: &mut Vec, classes: Option<&[ClassStatEntry]>) { - match classes { - None => buf.push(0), - Some(entries) => { - buf.push(1); - - let mut sorted: Vec<&ClassStatEntry> = entries.iter().collect(); - sorted.sort_by(|a, b| a.class_sid.cmp(&b.class_sid)); - - buf.extend_from_slice(&(sorted.len() as u32).to_le_bytes()); - for c in &sorted { - write_sid(buf, &c.class_sid); - buf.extend_from_slice(&c.count.to_le_bytes()); - - let mut sorted_props: Vec<&ClassPropertyUsage> = c.properties.iter().collect(); - sorted_props.sort_by(|a, b| a.property_sid.cmp(&b.property_sid)); - - buf.extend_from_slice(&(sorted_props.len() as u16).to_le_bytes()); - for pu in &sorted_props { - write_sid(buf, &pu.property_sid); - encode_class_property_payload(buf, pu); - } - } - } - } -} - -/// Decode the per-property payload within a class section: datatypes, langs, ref_classes. -fn decode_class_property_payload( - data: &[u8], - pos: &mut usize, - property_sid: Sid, -) -> io::Result { - // Datatypes - let dt_count = read_u16(data, pos)? as usize; - let mut datatypes = Vec::with_capacity(dt_count); - for _ in 0..dt_count { - let tag = read_u8(data, pos)?; - let count = read_u64(data, pos)?; - datatypes.push((tag, count)); - } - - // Langs - let lang_count = read_u16(data, pos)? as usize; - let mut langs = Vec::with_capacity(lang_count); - for _ in 0..lang_count { - let lang_len = read_u16(data, pos)? as usize; - ensure_len(data, *pos, lang_len, "lang string")?; - let lang = std::str::from_utf8(&data[*pos..*pos + lang_len]).map_err(|e| { - io::Error::new( - io::ErrorKind::InvalidData, - format!("invalid UTF-8 in lang tag: {e}"), - ) - })?; - *pos += lang_len; - let count = read_u64(data, pos)?; - langs.push((lang.to_string(), count)); - } - - // Ref classes - let rc_count = read_u16(data, pos)? as usize; - let mut ref_classes = Vec::with_capacity(rc_count); - for _ in 0..rc_count { - let (ref_sid, new_pos) = read_sid(data, *pos)?; - *pos = new_pos; - let ref_count = read_u64(data, pos)?; - ref_classes.push(ClassRefCount { - class_sid: ref_sid, - count: ref_count, - }); - } - - Ok(ClassPropertyUsage { - property_sid, - datatypes, - langs, - ref_classes, - }) -} - -/// Decode optional per-graph classes. -/// -/// Returns `None` if `has_classes == 0`, or `Some(vec)` if present. -/// Empty class lists are returned as `None` for consistency. -fn decode_optional_classes( - data: &[u8], - pos: &mut usize, -) -> io::Result>> { - let has_classes = read_u8(data, pos)?; - if has_classes == 0 { - return Ok(None); - } - - let class_count = read_u32(data, pos)? as usize; - let mut classes = Vec::with_capacity(class_count); - for _ in 0..class_count { - let (class_sid, new_pos) = read_sid(data, *pos)?; - *pos = new_pos; - let instance_count = read_u64(data, pos)?; - - let pu_count = read_u16(data, pos)? as usize; - let mut properties = Vec::with_capacity(pu_count); - for _ in 0..pu_count { - let (property_sid, new_pos2) = read_sid(data, *pos)?; - *pos = new_pos2; - properties.push(decode_class_property_payload(data, pos, property_sid)?); - } - - classes.push(ClassStatEntry { - class_sid, - count: instance_count, - properties, - }); - } - - Ok(if classes.is_empty() { - None - } else { - Some(classes) - }) -} - // ============================================================================ // Stats decode // ============================================================================ @@ -546,134 +374,9 @@ pub fn decode_stats(data: &[u8]) -> io::Result { decode_stats_with_len(data).map(|(stats, _)| stats) } -fn decode_graph_property(data: &[u8], pos: &mut usize) -> io::Result { - let p_id = read_u32(data, pos)?; - let count = read_u64(data, pos)?; - let ndv_values = read_u64(data, pos)?; - let ndv_subjects = read_u64(data, pos)?; - let last_modified_t = read_i64(data, pos)?; - let datatypes = decode_datatypes(data, pos)?; - let observed_datatypes = PropertyStatEntry::tags_of(&datatypes); - - Ok(GraphPropertyStatEntry { - p_id, - count, - ndv_values, - ndv_subjects, - last_modified_t, - datatypes, - observed_datatypes, - historical_datatypes: Vec::new(), - }) -} - /// One graph's rows in the tail: `(g_id, [(p_id, tags)])`. type GraphTagSets = Vec<(u16, Vec<(u32, Vec)>)>; -/// Decoded historical tail section — see [`encode_historical_tail`] for the -/// layout and the evolution rules. -struct HistoricalTail { - since_t: i64, - agg: Vec<((u16, String), Vec)>, - graphs: GraphTagSets, -} - -fn read_tag_set(data: &[u8], pos: &mut usize) -> io::Result> { - let n = read_u8(data, pos)? as usize; - ensure_len(data, *pos, n, "historical tag set")?; - let tags = data[*pos..*pos + n].to_vec(); - *pos += n; - Ok(tags) -} - -/// Decode the optional historical tail. `None` when the section is absent -/// (an old blob, exactly `pos == data.len()`) or carries an unknown future -/// tag — in which case the remainder is consumed, which is safe because the -/// root length-prefixes the whole stats section. -fn decode_historical_tail(data: &[u8], pos: &mut usize) -> io::Result> { - if *pos >= data.len() { - return Ok(None); - } - let tag = read_u8(data, pos)?; - if tag != HISTORICAL_TAIL_TAG { - *pos = data.len(); - return Ok(None); - } - let since_t = read_i64(data, pos)?; - let agg_count = read_u32(data, pos)? as usize; - let mut agg = Vec::with_capacity(agg_count); - for _ in 0..agg_count { - let (sid, new_pos) = read_sid_tuple(data, *pos)?; - *pos = new_pos; - let tags = read_tag_set(data, pos)?; - agg.push((sid, tags)); - } - let graph_count = read_u16(data, pos)? as usize; - let mut graphs = Vec::with_capacity(graph_count); - for _ in 0..graph_count { - let g_id = read_u16(data, pos)?; - let prop_count = read_u32(data, pos)? as usize; - let mut props = Vec::with_capacity(prop_count); - for _ in 0..prop_count { - let p_id = read_u32(data, pos)?; - let tags = read_tag_set(data, pos)?; - props.push((p_id, tags)); - } - graphs.push((g_id, props)); - } - Ok(Some(HistoricalTail { - since_t, - agg, - graphs, - })) -} - -/// Attach a decoded historical tail to the stats: set the boundary and fill -/// the per-entry `historical_datatypes` sets. Entries the tail does not name -/// keep their empty (unknown) set, which fails closed. -fn apply_historical_tail(stats: &mut IndexStats, tail: Option) { - let Some(tail) = tail else { return }; - stats.historical_since_t = Some(tail.since_t); - if let Some(props) = stats.properties.as_mut() { - let mut by_sid: std::collections::HashMap<(u16, String), Vec> = - tail.agg.into_iter().collect(); - for entry in &mut *props { - if let Some(tags) = by_sid.remove(&entry.sid) { - entry.historical_datatypes = tags; - } - } - } - if let Some(graphs) = stats.graphs.as_mut() { - let mut by_key: std::collections::HashMap<(u16, u32), Vec> = tail - .graphs - .into_iter() - .flat_map(|(g_id, props)| { - props - .into_iter() - .map(move |(p_id, tags)| ((g_id, p_id), tags)) - }) - .collect(); - for graph in &mut *graphs { - for prop in &mut graph.properties { - if let Some(tags) = by_key.remove(&(graph.g_id, prop.p_id)) { - prop.historical_datatypes = tags; - } - } - } - } -} - -fn decode_datatypes(data: &[u8], pos: &mut usize) -> io::Result> { - let count = read_u8(data, pos)? as usize; - let mut result = Vec::with_capacity(count); - for _ in 0..count { - let dt_tag = read_u8(data, pos)?; - let dt_count = read_u64(data, pos)?; - result.push((dt_tag, dt_count)); - } - Ok(result) -} - // ============================================================================ // Schema encode // ============================================================================ @@ -791,105 +494,11 @@ pub fn decode_schema(data: &[u8]) -> io::Result { // Public helpers for root encoder // ============================================================================ -/// Returns the number of bytes consumed when reading stats from a slice. -/// Used by the root decoder to know where the stats section ends. +/// Decode stats and the number of bytes consumed. The format has one parser, +/// `fluree_db_core::stats_wire::decode_stats`, shared with the snapshot's +/// metadata-only root decoder. pub fn decode_stats_with_len(data: &[u8]) -> io::Result<(IndexStats, usize)> { - let mut pos = 0usize; - - let flakes = read_u64(data, &mut pos)?; - let size = read_u64(data, &mut pos)?; - - let graph_count = read_u16(data, &mut pos)? as usize; - let mut graphs = Vec::with_capacity(graph_count); - for _ in 0..graph_count { - let g_id = read_u16(data, &mut pos)?; - let g_flakes = read_u64(data, &mut pos)?; - let g_size = read_u64(data, &mut pos)?; - let prop_count = read_u32(data, &mut pos)? as usize; - let mut properties = Vec::with_capacity(prop_count); - for _ in 0..prop_count { - properties.push(decode_graph_property(data, &mut pos)?); - } - // Per-graph classes (optional section after properties) - let graph_classes = decode_optional_classes(data, &mut pos)?; - - graphs.push(GraphStatsEntry { - g_id, - flakes: g_flakes, - size: g_size, - properties, - classes: graph_classes, - }); - } - - let agg_count = read_u32(data, &mut pos)? as usize; - let mut agg_props = Vec::with_capacity(agg_count); - for _ in 0..agg_count { - let (sid, new_pos) = read_sid_tuple(data, pos)?; - pos = new_pos; - let count = read_u64(data, &mut pos)?; - let ndv_values = read_u64(data, &mut pos)?; - let ndv_subjects = read_u64(data, &mut pos)?; - let last_modified_t = read_i64(data, &mut pos)?; - let datatypes = decode_datatypes(data, &mut pos)?; - let observed_datatypes = PropertyStatEntry::tags_of(&datatypes); - agg_props.push(PropertyStatEntry { - sid, - count, - ndv_values, - ndv_subjects, - last_modified_t, - datatypes, - observed_datatypes, - historical_datatypes: Vec::new(), - }); - } - - let class_count = read_u32(data, &mut pos)? as usize; - let mut classes = Vec::with_capacity(class_count); - for _ in 0..class_count { - let (class_sid, new_pos) = read_sid(data, pos)?; - pos = new_pos; - let instance_count = read_u64(data, &mut pos)?; - let pu_count = read_u16(data, &mut pos)? as usize; - let mut properties = Vec::with_capacity(pu_count); - for _ in 0..pu_count { - let (property_sid, new_pos2) = read_sid(data, pos)?; - pos = new_pos2; - properties.push(decode_class_property_payload(data, &mut pos, property_sid)?); - } - classes.push(ClassStatEntry { - class_sid, - count: instance_count, - properties, - }); - } - - let tail = decode_historical_tail(data, &mut pos)?; - - let mut stats = IndexStats { - flakes, - size, - properties: if agg_props.is_empty() { - None - } else { - Some(agg_props) - }, - classes: if classes.is_empty() { - fluree_db_core::index_stats::union_per_graph_classes(&graphs) - } else { - Some(classes) - }, - graphs: if graphs.is_empty() { - None - } else { - Some(graphs) - }, - historical_since_t: None, - }; - apply_historical_tail(&mut stats, tail); - - Ok((stats, pos)) + fluree_db_core::stats_wire::decode_stats(data) } /// Decode schema and return bytes consumed. @@ -1161,6 +770,187 @@ mod tests { /// does not store it twice: the bytes are those of a root without it, and /// both decoders hand back the union, also for an import root that never /// set it. + fn class_tables() -> Vec { + let usage = |property: Sid, refs: &[(Sid, u64)]| ClassPropertyUsage { + property_sid: property, + datatypes: vec![(7, 3), (1, 300)], + langs: vec![("fr".to_string(), 2), ("en".to_string(), 1)], + ref_classes: refs + .iter() + .map(|(class_sid, count)| ClassRefCount { + class_sid: class_sid.clone(), + count: *count, + }) + .collect(), + }; + let class = |class_sid: Sid, count: u64, properties| ClassStatEntry { + class_sid, + count, + properties, + }; + let graph = |g_id: u16, classes: Option>| GraphStatsEntry { + g_id, + flakes: 10, + size: 0, + properties: vec![], + classes, + }; + vec![ + graph( + 0, + Some(vec![ + class( + sid(5, "PersonB"), + 1 << 62, + vec![ + usage(sid(5, "name"), &[]), + usage( + sid(5, "knows"), + &[(sid(6, "Person"), 9), (sid(5, "PersonA"), 1 << 40)], + ), + ], + ), + class(sid(5, "PersonA"), u64::MAX, vec![]), + ]), + ), + graph(1, None), + // Its first class is not the table's first Sid, so a class index + // read as absolute instead of as a delta lands elsewhere. + graph( + 2, + Some(vec![ + class( + sid(6, "Person"), + 1, + vec![usage(sid(5, "knows"), &[(sid(5, "PersonB"), 2)])], + ), + class(sid(5, "PersonB"), 2, vec![]), + ]), + ), + ] + } + + /// The canonical (sorted) form both encoders have always emitted. + fn sorted_classes(classes: &[ClassStatEntry]) -> Vec { + let mut classes = classes.to_vec(); + classes.sort_by(|a, b| a.class_sid.cmp(&b.class_sid)); + for class in &mut classes { + class + .properties + .sort_by(|a, b| a.property_sid.cmp(&b.property_sid)); + for usage in &mut class.properties { + usage.datatypes.sort_by_key(|d| d.0); + usage.langs.sort(); + usage + .ref_classes + .sort_by(|a, b| a.class_sid.cmp(&b.class_sid)); + } + } + classes + } + + #[test] + fn per_graph_class_tables_round_trip_through_the_class_tail() { + let graphs = class_tables(); + let stats = IndexStats { + flakes: 30, + size: 0, + properties: None, + classes: None, + graphs: Some(graphs.clone()), + historical_since_t: Some(0), + }; + let bytes = encode_stats(&stats); + let (via_core, consumed) = fluree_db_core::stats_wire::decode_stats(&bytes).unwrap(); + assert_eq!(consumed, bytes.len()); + for decoded in [decode_stats(&bytes).unwrap(), via_core] { + assert_eq!( + decoded.historical_since_t, + Some(0), + "the historical tail still reads" + ); + let got = decoded.graphs.unwrap(); + assert_eq!(got.len(), 3); + for (got, want) in got.iter().zip(&graphs) { + assert_eq!(got.g_id, want.g_id); + assert_eq!( + format!("{:?}", got.classes), + format!("{:?}", want.classes.as_deref().map(sorted_classes)), + "graph {}", + want.g_id + ); + } + } + } + + /// A reader that predates the class tail parses up to it and stops; what + /// it sees must be exactly the stats without class tables. + #[test] + fn a_reader_without_the_class_tail_sees_no_classes() { + let with = IndexStats { + flakes: 30, + size: 0, + properties: None, + classes: None, + graphs: Some(class_tables()), + historical_since_t: Some(0), + }; + let mut without = with.clone(); + for g in without.graphs.iter_mut().flatten() { + g.classes = None; + } + let with = encode_stats(&with); + let without = encode_stats(&without); + assert!(with.len() > without.len()); + assert_eq!(&with[..without.len()], &without[..]); + assert_eq!(with[without.len()], 2, "the class tail follows"); + } + + /// Roots written before the class tail carry each graph's classes in the + /// legacy per-graph slot, Sids spelled out, fixed-width counts. + #[test] + fn legacy_per_graph_class_slots_still_decode() { + let mut b = Vec::new(); + let put_sid = |b: &mut Vec, ns: u16, name: &str| { + b.extend_from_slice(&ns.to_le_bytes()); + b.extend_from_slice(&(name.len() as u16).to_le_bytes()); + b.extend_from_slice(name.as_bytes()); + }; + b.extend_from_slice(&5u64.to_le_bytes()); // flakes + b.extend_from_slice(&0u64.to_le_bytes()); // size + b.extend_from_slice(&1u16.to_le_bytes()); // graphs + b.extend_from_slice(&0u16.to_le_bytes()); // g_id + b.extend_from_slice(&5u64.to_le_bytes()); + b.extend_from_slice(&0u64.to_le_bytes()); + b.extend_from_slice(&0u32.to_le_bytes()); // properties + b.push(1); // has_classes + b.extend_from_slice(&1u32.to_le_bytes()); + put_sid(&mut b, 5, "Person"); + b.extend_from_slice(&3u64.to_le_bytes()); + b.extend_from_slice(&1u16.to_le_bytes()); // property usages + put_sid(&mut b, 5, "knows"); + b.extend_from_slice(&1u16.to_le_bytes()); // datatypes + b.push(1); + b.extend_from_slice(&3u64.to_le_bytes()); + b.extend_from_slice(&0u16.to_le_bytes()); // langs + b.extend_from_slice(&1u16.to_le_bytes()); // ref classes + put_sid(&mut b, 5, "Person"); + b.extend_from_slice(&3u64.to_le_bytes()); + b.extend_from_slice(&0u32.to_le_bytes()); // aggregate properties + b.extend_from_slice(&0u32.to_le_bytes()); // aggregate classes + + let decoded = decode_stats(&b).unwrap(); + let graph = &decoded.graphs.as_ref().unwrap()[0]; + let classes = graph.classes.as_ref().unwrap(); + assert_eq!(classes[0].class_sid, sid(5, "Person")); + assert_eq!(classes[0].properties[0].ref_classes[0].count, 3); + assert_eq!( + format!("{:?}", decoded.classes), + format!("{:?}", graph.classes), + "the ledger-wide table is derived" + ); + } + #[test] fn the_ledger_wide_class_list_is_derived_from_the_graphs() { let class = |name: &str, count: u64| ClassStatEntry { diff --git a/fluree-db-core/src/stats_wire.rs b/fluree-db-core/src/stats_wire.rs index cf94972833..b5c8ac23df 100644 --- a/fluree-db-core/src/stats_wire.rs +++ b/fluree-db-core/src/stats_wire.rs @@ -1,7 +1,9 @@ //! Binary wire format decoders for index stats and schema sections. //! //! These decode the binary stats/schema sections embedded in `IndexRoot` -//! (FIR6). The encode functions live in `fluree-db-indexer`. +//! (FIR6). The encode functions live in `fluree-db-binary-index`, except the +//! compact class tail ([`encode_class_tail`]), which lives here beside its +//! decoder. use crate::index_schema::{IndexSchema, SchemaPredicateInfo, SchemaPredicates}; use crate::index_stats::{ @@ -239,6 +241,226 @@ fn apply_historical_tail(stats: &mut IndexStats, tail: Option) { } } +/// Wire tag identifying the compact per-graph class tail. +const CLASS_TAIL_TAG: u8 = 2; + +fn read_varint(data: &[u8], pos: &mut usize) -> io::Result { + crate::commit::codec::varint::decode_varint(data, pos) + .map_err(|e| io::Error::new(io::ErrorKind::InvalidData, format!("stats varint: {e}"))) +} + +/// A count read from the wire, with capacity capped by the bytes left so a +/// corrupt count cannot allocate past the blob. +fn read_count(data: &[u8], pos: &mut usize) -> io::Result<(usize, usize)> { + let n = read_varint(data, pos)? as usize; + Ok((n, n.min(data.len().saturating_sub(*pos)))) +} + +/// Append the per-graph class tables as one compact tail section. +/// +/// Every Sid the tables name (classes, their properties, ref targets) is +/// written once, in a sorted table, front-coded against the previous name in +/// the same namespace; entries refer to it by index and counts are varints. +/// Readers that predate the section skip it (see the historical tail for why +/// an appended section is safe), so the encoder leaves the legacy per-graph +/// class slots empty: those readers see no class tables rather than +/// misparsing. +/// +/// ```text +/// [tag: u8 = 2] +/// [sid_count: varint] +/// per Sid, sorted: [ns_code: u16 LE][shared: varint][rest_len: varint][rest] +/// [graph_count: varint] +/// per graph with classes, by g_id: [g_id: u16 LE][class_count: varint] +/// per class, by Sid: [index delta from the previous class: varint] +/// [count: varint][property_count: varint] +/// per property, by Sid: [index: varint] +/// [n: varint] n × [tag: u8][count: varint] +/// [n: varint] n × [len: varint][lang bytes][count: varint] +/// [n: varint] n × [ref class index: varint][count: varint] +/// ``` +pub fn encode_class_tail(buf: &mut Vec, graphs: &[&GraphStatsEntry]) { + use crate::commit::codec::varint::encode_varint; + use std::collections::{BTreeSet, HashMap}; + + let with_classes: Vec<(u16, &[ClassStatEntry])> = graphs + .iter() + .filter_map(|g| g.classes.as_deref().map(|c| (g.g_id, c))) + .collect(); + let mut sids: BTreeSet<&Sid> = BTreeSet::new(); + for (_, classes) in &with_classes { + for class in *classes { + sids.insert(&class.class_sid); + for usage in &class.properties { + sids.insert(&usage.property_sid); + sids.extend(usage.ref_classes.iter().map(|r| &r.class_sid)); + } + } + } + let index: HashMap<&Sid, u64> = sids + .iter() + .enumerate() + .map(|(i, s)| (*s, i as u64)) + .collect(); + + buf.push(CLASS_TAIL_TAG); + encode_varint(sids.len() as u64, buf); + let mut prev: Option<&Sid> = None; + for sid in &sids { + let name = sid.name.as_bytes(); + let shared = match prev { + Some(p) if p.namespace_code == sid.namespace_code => name + .iter() + .zip(p.name.as_bytes()) + .take_while(|(a, b)| a == b) + .count(), + _ => 0, + }; + buf.extend_from_slice(&sid.namespace_code.to_le_bytes()); + encode_varint(shared as u64, buf); + encode_varint((name.len() - shared) as u64, buf); + buf.extend_from_slice(&name[shared..]); + prev = Some(sid); + } + + let mut with_classes = with_classes; + with_classes.sort_by_key(|(g_id, _)| *g_id); + encode_varint(with_classes.len() as u64, buf); + for (g_id, classes) in with_classes { + buf.extend_from_slice(&g_id.to_le_bytes()); + let mut classes: Vec<&ClassStatEntry> = classes.iter().collect(); + classes.sort_by(|a, b| a.class_sid.cmp(&b.class_sid)); + encode_varint(classes.len() as u64, buf); + let mut prev_class = 0u64; + for class in classes { + let i = index[&class.class_sid]; + encode_varint(i - prev_class, buf); + prev_class = i; + encode_varint(class.count, buf); + let mut usages: Vec<&ClassPropertyUsage> = class.properties.iter().collect(); + usages.sort_by(|a, b| a.property_sid.cmp(&b.property_sid)); + encode_varint(usages.len() as u64, buf); + for usage in usages { + encode_varint(index[&usage.property_sid], buf); + let mut dts: Vec<&(u8, u64)> = usage.datatypes.iter().collect(); + dts.sort_by_key(|d| d.0); + encode_varint(dts.len() as u64, buf); + for &&(tag, count) in &dts { + buf.push(tag); + encode_varint(count, buf); + } + let mut langs: Vec<&(String, u64)> = usage.langs.iter().collect(); + langs.sort_by(|a, b| a.0.cmp(&b.0)); + encode_varint(langs.len() as u64, buf); + for (lang, count) in langs { + encode_varint(lang.len() as u64, buf); + buf.extend_from_slice(lang.as_bytes()); + encode_varint(*count, buf); + } + let mut refs: Vec<&ClassRefCount> = usage.ref_classes.iter().collect(); + refs.sort_by(|a, b| a.class_sid.cmp(&b.class_sid)); + encode_varint(refs.len() as u64, buf); + for r in refs { + encode_varint(index[&r.class_sid], buf); + encode_varint(r.count, buf); + } + } + } + } +} + +/// Decode the class tail written by [`encode_class_tail`] (tag already +/// consumed): each graph's class table. Decoded Sids share their names. +fn decode_class_tail(data: &[u8], pos: &mut usize) -> io::Result)>> { + let invalid = + |what: &str| io::Error::new(io::ErrorKind::InvalidData, format!("class tail: {what}")); + + let (sid_count, cap) = read_count(data, pos)?; + let mut sids: Vec = Vec::with_capacity(cap); + let mut prev: Vec = Vec::new(); + let mut prev_ns: Option = None; + for _ in 0..sid_count { + let ns_code = read_u16(data, pos)?; + let shared = read_varint(data, pos)? as usize; + let rest = read_varint(data, pos)? as usize; + if shared > 0 && (prev_ns != Some(ns_code) || shared > prev.len()) { + return Err(invalid("shared prefix out of range")); + } + ensure_len(data, *pos, rest, "class tail sid")?; + prev.truncate(shared); + prev.extend_from_slice(&data[*pos..*pos + rest]); + *pos += rest; + let name = std::str::from_utf8(&prev).map_err(|_| invalid("sid name is not UTF-8"))?; + sids.push(Sid::new(ns_code, name)); + prev_ns = Some(ns_code); + } + let sid_at = |i: u64| -> io::Result { + sids.get(i as usize) + .cloned() + .ok_or_else(|| invalid("sid index out of range")) + }; + + let (graph_count, cap) = read_count(data, pos)?; + let mut graphs = Vec::with_capacity(cap); + for _ in 0..graph_count { + let g_id = read_u16(data, pos)?; + let (class_count, cap) = read_count(data, pos)?; + let mut classes = Vec::with_capacity(cap); + let mut class_index = 0u64; + for _ in 0..class_count { + class_index = class_index + .checked_add(read_varint(data, pos)?) + .ok_or_else(|| invalid("class index overflow"))?; + let class_sid = sid_at(class_index)?; + let count = read_varint(data, pos)?; + let (usage_count, cap) = read_count(data, pos)?; + let mut properties = Vec::with_capacity(cap); + for _ in 0..usage_count { + let property_sid = sid_at(read_varint(data, pos)?)?; + let (n, cap) = read_count(data, pos)?; + let mut datatypes = Vec::with_capacity(cap); + for _ in 0..n { + let tag = read_u8(data, pos)?; + datatypes.push((tag, read_varint(data, pos)?)); + } + let (n, cap) = read_count(data, pos)?; + let mut langs = Vec::with_capacity(cap); + for _ in 0..n { + let len = read_varint(data, pos)? as usize; + ensure_len(data, *pos, len, "class tail lang")?; + let lang = std::str::from_utf8(&data[*pos..*pos + len]) + .map_err(|_| invalid("lang tag is not UTF-8"))? + .to_string(); + *pos += len; + langs.push((lang, read_varint(data, pos)?)); + } + let (n, cap) = read_count(data, pos)?; + let mut ref_classes = Vec::with_capacity(cap); + for _ in 0..n { + let class_sid = sid_at(read_varint(data, pos)?)?; + ref_classes.push(ClassRefCount { + class_sid, + count: read_varint(data, pos)?, + }); + } + properties.push(ClassPropertyUsage { + property_sid, + datatypes, + langs, + ref_classes, + }); + } + classes.push(ClassStatEntry { + class_sid, + count, + properties, + }); + } + graphs.push((g_id, classes)); + } + Ok(graphs) +} + /// Decode the per-property payload within a class section: datatypes, langs, ref_classes. fn decode_class_property_payload( data: &[u8], @@ -407,7 +629,27 @@ pub fn decode_stats(data: &[u8]) -> io::Result<(IndexStats, usize)> { }); } - let tail = decode_historical_tail(data, &mut pos)?; + // Appended sections, each led by its tag; an unknown tag ends the parse + // (the root length-prefixes the stats section, so the rest is skipped). + let mut tail = None; + while pos < data.len() { + match data[pos] { + HISTORICAL_TAIL_TAG => tail = decode_historical_tail(data, &mut pos)?, + CLASS_TAIL_TAG => { + pos += 1; + for (g_id, classes) in decode_class_tail(data, &mut pos)? { + let graph = graphs.iter_mut().find(|g| g.g_id == g_id).ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + format!("class tail names graph {g_id}, which the stats do not hold"), + ) + })?; + graph.classes = (!classes.is_empty()).then_some(classes); + } + } + _ => pos = data.len(), + } + } let mut stats = IndexStats { flakes, From 46e5b155a1e3c8ba0912a11b5a02eb2c3f6862c0 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 20:37:38 -0400 Subject: [PATCH 19/92] perf(api): decode the index root once when loading a ledger The state load decoded the root for the snapshot metadata (`LedgerSnapshot::from_root_bytes`) and the binary store load decoded the same bytes again as an `IndexRoot`; on the annotation benchmark slice each decode took about 110 ms. `LedgerState::load_decoding_root` / `load_with_store_decoding_root` take the root decoder from the caller and hand back what it kept. The API decodes the root once as an `IndexRoot`, builds the snapshot from it through `IndexRoot::take_snapshot_metadata` (stats and schema move into the snapshot rather than being copied), and passes the rest of the root to the binary store load. Index publish builds its snapshot through the same method. `load` and `load_with_store` keep the metadata-only decode. --- .../src/indexer_fulltext_provider.rs | 12 +- fluree-db-api/src/ledger/loading.rs | 23 +++- fluree-db-api/src/ledger_manager.rs | 114 ++++++++++-------- .../src/format/index_root.rs | 113 +++++++++++++++++ fluree-db-ledger/src/lib.rs | 71 ++++++----- 5 files changed, 242 insertions(+), 91 deletions(-) diff --git a/fluree-db-api/src/indexer_fulltext_provider.rs b/fluree-db-api/src/indexer_fulltext_provider.rs index fed28973b2..2b17cec2b5 100644 --- a/fluree-db-api/src/indexer_fulltext_provider.rs +++ b/fluree-db-api/src/indexer_fulltext_provider.rs @@ -138,10 +138,14 @@ impl ApiFulltextConfigProvider { } // 1. Load ledger state (snapshot + novelty). - let (mut state, root) = - LedgerState::load_with_root(self.nameservice.as_ref(), ledger_id, &self.backend) - .await - .map_err(|e| format!("LedgerState::load: {e}"))?; + let (mut state, root) = LedgerState::load_decoding_root( + self.nameservice.as_ref(), + ledger_id, + &self.backend, + crate::ledger_manager::decode_index_root, + ) + .await + .map_err(|e| format!("LedgerState::load: {e}"))?; // 2. If an index exists, load the binary store so the config graph // can be read via the indexed side too. Without this, only diff --git a/fluree-db-api/src/ledger/loading.rs b/fluree-db-api/src/ledger/loading.rs index 20e2daedbd..1a370139c7 100644 --- a/fluree-db-api/src/ledger/loading.rs +++ b/fluree-db-api/src/ledger/loading.rs @@ -3,21 +3,22 @@ use std::sync::Arc; use crate::ledger_view::{CommitRef, LedgerView}; use crate::time_resolve; use crate::{ApiError, Fluree, HistoricalLedgerView, LedgerState, Result, TimeSpec}; +use fluree_db_binary_index::IndexRoot; use fluree_db_core::ContentStore; use fluree_db_core::LedgerId; use fluree_db_core::{collect_first_parent_cids, load_commit_envelope_by_id, CommitId, ContentId}; -use fluree_db_ledger::LoadedIndexRoot; +use fluree_db_ledger::DecodedIndexRoot; use fluree_db_nameservice::{NameServiceError, NsRecord}; use fluree_db_query::QueryError; impl Fluree { /// Attach the binary index store and range provider to an already-loaded /// ledger state when its nameservice record points at a binary index root. - /// `root` is the index root the state load already read, if any. + /// `root` is the index root the state load already decoded, if any. pub(crate) async fn attach_index( &self, state: &mut LedgerState, - root: Option, + root: Option>, ) -> Result<()> { crate::ledger_manager::load_and_attach_binary_store( self.backend(), @@ -61,7 +62,12 @@ impl Fluree { where C: ContentStore + Clone + 'static, { - let (mut state, root) = LedgerState::load_with_store_and_root(store, record).await?; + let (mut state, root) = LedgerState::load_with_store_decoding_root( + store, + record, + crate::ledger_manager::decode_index_root, + ) + .await?; self.attach_index(&mut state, root).await?; Ok(state) } @@ -86,8 +92,13 @@ impl Fluree { /// same lock) would deadlock the transact path — and the failure mode is /// a hang, not a red test. Anything added here must stay manager-free. pub(crate) async fn load_ledger_uncached(&self, ledger_id: &str) -> Result { - let (mut state, root) = - LedgerState::load_with_root(&self.nameservice_mode, ledger_id, self.backend()).await?; + let (mut state, root) = LedgerState::load_decoding_root( + &self.nameservice_mode, + ledger_id, + self.backend(), + crate::ledger_manager::decode_index_root, + ) + .await?; self.attach_index(&mut state, root).await?; // Default context is not loaded here. Opt-in callers route through // `Fluree::db_with_default_context` / `db_at_with_default_context`, diff --git a/fluree-db-api/src/ledger_manager.rs b/fluree-db-api/src/ledger_manager.rs index 77ff820134..649a682efc 100644 --- a/fluree-db-api/src/ledger_manager.rs +++ b/fluree-db-api/src/ledger_manager.rs @@ -27,13 +27,14 @@ use std::time::Duration; use std::path::PathBuf; +use fluree_db_binary_index::IndexRoot; use fluree_db_binary_index::{BinaryIndexStore, LeafletCache}; -use fluree_db_core::db::{LedgerSnapshot, LedgerSnapshotMetadata}; +use fluree_db_core::db::LedgerSnapshot; use fluree_db_core::dict_novelty::DictNovelty; use fluree_db_core::ledger_config::LedgerConfig; use fluree_db_core::trace_first_parent_commits_by_id; use fluree_db_core::{ContentId, ContentStore, LedgerId, Sid, StorageBackend}; -use fluree_db_ledger::{LedgerState, LoadedIndexRoot, TypeErasedStore}; +use fluree_db_ledger::{DecodedIndexRoot, LedgerState, TypeErasedStore}; use fluree_db_nameservice::NsRecord; use rustc_hash::FxHashSet; use std::collections::VecDeque; @@ -718,23 +719,10 @@ impl LedgerHandle { drop(prev_store); // Build metadata-only LedgerSnapshot from FIR6 root. - let meta = LedgerSnapshotMetadata { - ledger_id: LedgerId::parse(&root.ledger_id) - .map_err(|e| ApiError::internal(format!("index root ledger id: {e}")))?, - t: root.index_t, - base_t: root.base_t, - namespace_codes: root.namespace_codes.into_iter().collect(), - ns_split_mode: root.ns_split_mode, - stats: root.stats.map(Arc::new), - schema: root.schema, - subject_watermarks: root.subject_watermarks, - string_watermark: root.string_watermark, - graph_iris: root.graph_iris, - has_annotations: root.has_annotations, - annotation_index: root.annotation_index.clone(), - had_annotation_arena: root.had_annotation_arena, - has_list_meta: root.has_list_meta, - }; + let mut root = root; + let meta = root + .take_snapshot_metadata() + .map_err(|e| ApiError::internal(e.to_string()))?; tracing::Span::current().record("index_t", meta.t); let db = LedgerSnapshot::new_meta(meta) .map_err(|e| ApiError::internal(format!("graph registry from root: {e}")))?; @@ -1145,7 +1133,22 @@ async fn prefetch_novelty_translation( /// /// `prev` is the store this load replaces, if any (a reload of a cached /// ledger); artifacts the new root shares with it are carried over. `root` is -/// the index root the state load already read, if any. +/// the index root the state load already decoded ([`decode_index_root`]), if +/// any. +/// Decode an index root once for both consumers: the snapshot's metadata +/// (stats and schema move into it) and, kept, the rest of the root for +/// [`load_and_attach_binary_store`]. +pub(crate) fn decode_index_root( + bytes: Vec, +) -> fluree_db_ledger::Result<(LedgerSnapshot, IndexRoot)> { + let invalid = |e: std::io::Error| { + fluree_db_core::Error::invalid_index(format!("index root: FIR6 decode: {e}")) + }; + let mut root = IndexRoot::decode(&bytes).map_err(invalid)?; + let meta = root.take_snapshot_metadata().map_err(invalid)?; + Ok((LedgerSnapshot::new_meta(meta)?, root)) +} + pub(crate) async fn load_and_attach_binary_store( backend: &StorageBackend, nameservice: &dyn fluree_db_nameservice::NameServiceLookup, @@ -1153,7 +1156,7 @@ pub(crate) async fn load_and_attach_binary_store( cache_dir: &std::path::Path, leaflet_cache: Option>, prev: Option<&BinaryIndexStore>, - root: Option, + root: Option>, ) -> std::result::Result>, ApiError> { let record = match state.ns_record.as_ref() { Some(r) => r, @@ -1172,30 +1175,29 @@ pub(crate) async fn load_and_attach_binary_store( let cs: Arc = fluree_db_nameservice::branched_content_store_for_record(backend, nameservice, record) .await?; - let root_started = Instant::now(); - let bytes = match root { - Some(root) if root.id == index_cid => root.bytes, - _ => cs - .get(&index_cid) - .await - .map_err(|e| ApiError::internal(format!("failed to read index root: {e}")))?, + let root = match root { + Some((id, root)) if id == index_cid => root, + _ => { + let root_started = Instant::now(); + let bytes = cs + .get(&index_cid) + .await + .map_err(|e| ApiError::internal(format!("failed to read index root: {e}")))?; + let root_read_us = root_started.elapsed().as_micros() as u64; + let decode_started = Instant::now(); + let root = IndexRoot::decode(&bytes) + .map_err(|e| ApiError::internal(format!("failed to decode FIR6 root: {e}")))?; + tracing::debug!( + target: "fluree::open", + ledger = %record.ledger_id, + bytes = bytes.len(), + root_read_us, + root_decode_us = decode_started.elapsed().as_micros() as u64, + "binary index root loaded" + ); + root + } }; - let root_read_us = root_started.elapsed().as_micros() as u64; - let decode_started = Instant::now(); - - // Decode FIR6 root metadata to populate snapshot watermarks. - // `LedgerSnapshot::from_root_bytes` only parses the header; watermarks are needed for - // DictNovelty/DictOverlay correctness (especially bound-object filters and overlay merges). - let root = fluree_db_binary_index::IndexRoot::decode(&bytes) - .map_err(|e| ApiError::internal(format!("failed to decode FIR6 root: {e}")))?; - tracing::debug!( - target: "fluree::open", - ledger = %record.ledger_id, - bytes = bytes.len(), - root_read_us, - root_decode_us = decode_started.elapsed().as_micros() as u64, - "binary index root loaded" - ); let mut store = BinaryIndexStore::load_from_root_v6_reusing( Arc::clone(&cs), @@ -1650,10 +1652,14 @@ impl LedgerManager { // ledger cache into a global mutex for the duration of any cold load. // Note: we pass the original address to nameservice (it handles // resolution), but cache under the canonical address. - let load_result = - LedgerState::load_with_root(&self.nameservice_mode, ledger_id, &self.backend) - .await - .map_err(ApiError::from); // Convert LedgerError to ApiError + let load_result = LedgerState::load_decoding_root( + &self.nameservice_mode, + ledger_id, + &self.backend, + decode_index_root, + ) + .await + .map_err(ApiError::from); // Convert LedgerError to ApiError tracing::debug!( alias = %canonical_alias, ok = load_result.is_ok(), @@ -1935,10 +1941,14 @@ impl LedgerManager { // not held over any of this and is acquired after the swap, so // the two locks never overlap (avoids the entries↔state ordering // hazard with `current_t`). - let loaded = - LedgerState::load_with_root(&self.nameservice_mode, ledger_id, &self.backend) - .await - .map_err(ApiError::from); + let loaded = LedgerState::load_decoding_root( + &self.nameservice_mode, + ledger_id, + &self.backend, + decode_index_root, + ) + .await + .map_err(ApiError::from); let result = match loaded { Ok((mut new_state, root)) => { // Attempt to load binary index store (v2 only) — still off-lock. diff --git a/fluree-db-binary-index/src/format/index_root.rs b/fluree-db-binary-index/src/format/index_root.rs index 339d2998e8..bdc838db97 100644 --- a/fluree-db-binary-index/src/format/index_root.rs +++ b/fluree-db-binary-index/src/format/index_root.rs @@ -890,6 +890,41 @@ impl IndexRoot { buf } + /// The snapshot metadata this root carries, so a caller holding the + /// decoded root does not decode the bytes a second time for the snapshot. + /// The large sections (stats and schema) move out; everything the binary + /// index store reads stays in place. + pub fn take_snapshot_metadata( + &mut self, + ) -> io::Result { + let ledger_id = fluree_db_core::LedgerId::parse(&self.ledger_id).map_err(|e| { + io::Error::new( + io::ErrorKind::InvalidData, + format!("index root ledger id: {e}"), + ) + })?; + Ok(fluree_db_core::db::LedgerSnapshotMetadata { + ledger_id, + t: self.index_t, + base_t: self.base_t, + namespace_codes: self + .namespace_codes + .iter() + .map(|(&code, prefix)| (code, prefix.clone())) + .collect(), + ns_split_mode: self.ns_split_mode, + stats: self.stats.take().map(std::sync::Arc::new), + schema: self.schema.take(), + subject_watermarks: self.subject_watermarks.clone(), + string_watermark: self.string_watermark, + graph_iris: self.graph_iris.clone(), + has_annotations: self.has_annotations, + annotation_index: self.annotation_index.clone(), + had_annotation_arena: self.had_annotation_arena, + has_list_meta: self.has_list_meta, + }) + } + /// Decode from FIR6 binary bytes. pub fn decode(data: &[u8]) -> io::Result { if data.len() < Self::HEADER_LEN { @@ -1600,6 +1635,84 @@ mod tests { assert_eq!(snap_ann, ann); } + /// A caller holding the decoded root builds the snapshot from it instead + /// of decoding the bytes again; that snapshot must be the one the + /// metadata-only decoder builds from the same bytes. + #[test] + fn snapshot_metadata_from_the_decoded_root_matches_the_metadata_decoder() { + use fluree_db_core::index_stats::{ClassStatEntry, GraphStatsEntry}; + use fluree_db_core::sid::Sid; + use fluree_db_core::LedgerSnapshot; + + let mut root = minimal_root_v6(); + root.base_t = 7; + root.graph_iris = vec!["urn:g:txn".into(), "urn:g:config".into(), "urn:g:a".into()]; + root.subject_watermarks = vec![5, 9]; + root.string_watermark = 11; + root.has_annotations = true; + root.had_annotation_arena = true; + root.has_list_meta = Some(true); + root.stats = Some(IndexStats { + flakes: 3, + size: 30, + properties: None, + classes: None, + graphs: Some(vec![GraphStatsEntry { + g_id: 0, + flakes: 3, + size: 30, + properties: vec![], + classes: Some(vec![ClassStatEntry { + class_sid: Sid::new(0, "Person"), + count: 2, + properties: vec![], + }]), + }]), + historical_since_t: Some(0), + }); + root.schema = Some(IndexSchema { + t: 4, + ..Default::default() + }); + let bytes = root.encode(); + + let view = |s: &LedgerSnapshot| { + let namespaces: BTreeMap<_, _> = s.namespaces().iter().collect(); + let graphs: Vec<_> = s.graph_registry.iter_entries().collect(); + format!( + "{:?}", + ( + &s.ledger_id, + s.t, + s.base_t, + namespaces, + s.ns_split_mode(), + &s.stats, + &s.schema, + &s.subject_watermarks, + s.string_watermark, + graphs, + ( + s.has_annotations, + &s.annotation_index, + s.had_annotation_arena + ), + s.has_list_meta, + ) + ) + }; + let from_bytes = LedgerSnapshot::from_root_bytes(&bytes).unwrap(); + let mut decoded = IndexRoot::decode(&bytes).unwrap(); + let from_root = + LedgerSnapshot::new_meta(decoded.take_snapshot_metadata().unwrap()).unwrap(); + assert_eq!(view(&from_root), view(&from_bytes)); + assert!(from_root.stats.is_some() && from_root.schema.is_some()); + assert!( + decoded.stats.is_none() && decoded.schema.is_none(), + "stats and schema move into the snapshot rather than being copied" + ); + } + #[test] fn fir6_round_trip_has_list_meta_three_state() { // Two bits in the extended-flags byte: TRACKED distinguishes a diff --git a/fluree-db-ledger/src/lib.rs b/fluree-db-ledger/src/lib.rs index 5220c3b957..0742e5b442 100644 --- a/fluree-db-ledger/src/lib.rs +++ b/fluree-db-ledger/src/lib.rs @@ -118,11 +118,13 @@ impl HeadTemporal { } } -/// An index root read while loading a [`LedgerState`], for a caller that -/// decodes the whole root next (the binary index store). -pub struct LoadedIndexRoot { - pub id: ContentId, - pub bytes: Vec, +/// The index root a [`LedgerState`] load decoded, as kept by its decoder, +/// with the root's content id. +pub type DecodedIndexRoot = (ContentId, R); + +/// The default root decode: the snapshot's metadata, nothing kept. +fn decode_snapshot(bytes: Vec) -> Result<(LedgerSnapshot, ())> { + Ok((LedgerSnapshot::from_root_bytes(&bytes)?, ())) } /// Ledger state combining indexed LedgerSnapshot with novelty overlay @@ -205,17 +207,23 @@ impl LedgerState { ledger_id: &str, backend: &StorageBackend, ) -> Result { - Ok(Self::load_with_root(ns, ledger_id, backend).await?.0) + Ok( + Self::load_decoding_root(ns, ledger_id, backend, decode_snapshot) + .await? + .0, + ) } - /// [`Self::load`], also handing back the index root it read, so a caller - /// that decodes the whole root next (the binary index store) does not - /// fetch it a second time. - pub async fn load_with_root( + /// [`Self::load`] with the caller decoding the index root: `decode` turns + /// the root's bytes into the snapshot and whatever else the caller keeps + /// of it (the binary index store needs the whole root), which comes back + /// with the state, so the root is fetched and decoded once. + pub async fn load_decoding_root( ns: &dyn NameServiceLookup, ledger_id: &str, backend: &StorageBackend, - ) -> Result<(Self, Option)> { + decode: impl FnOnce(Vec) -> Result<(LedgerSnapshot, R)> + Send, + ) -> Result<(Self, Option>)> { let record = ns .lookup(ledger_id) .await? @@ -235,11 +243,11 @@ impl LedgerState { // ancestor namespaces for pre-branch-point content. if record.source_branch.is_some() { let store = Self::build_branched_store(ns, &record, backend).await?; - return Self::load_with_store_and_root(store, record).await; + return Self::load_with_store_decoding_root(store, record, decode).await; } let store = backend.content_store(&record.ledger_id); - Self::load_with_store_and_root(store, record).await + Self::load_with_store_decoding_root(store, record, decode).await } /// Build a recursive `BranchedContentStore` by walking the branch ancestry. @@ -264,34 +272,39 @@ impl LedgerState { store: C, record: NsRecord, ) -> Result { - Ok(Self::load_with_store_and_root(store, record).await?.0) + Ok( + Self::load_with_store_decoding_root(store, record, decode_snapshot) + .await? + .0, + ) } - /// [`Self::load_with_store`], also handing back the index root it read. - pub async fn load_with_store_and_root( + /// [`Self::load_with_store`] with the caller decoding the index root; see + /// [`Self::load_decoding_root`]. + pub async fn load_with_store_decoding_root( store: C, record: NsRecord, - ) -> Result<(Self, Option)> { - let root = match &record.index_head_id { - Some(id) => Some(LoadedIndexRoot { - id: id.clone(), - bytes: store.get(id).await?, - }), - None => None, + decode: impl FnOnce(Vec) -> Result<(LedgerSnapshot, R)> + Send, + ) -> Result<(Self, Option>)> { + let (snapshot, root) = match &record.index_head_id { + Some(id) => { + let (snapshot, kept) = decode(store.get(id).await?)?; + (Some(snapshot), Some((id.clone(), kept))) + } + None => (None, None), }; - let state = Self::load_from_root(store, record, root.as_ref()).await?; + let state = Self::load_from_snapshot(store, record, snapshot).await?; Ok((state, root)) } - async fn load_from_root( + async fn load_from_snapshot( store: C, record: NsRecord, - root: Option<&LoadedIndexRoot>, + snapshot: Option, ) -> Result { // Handle missing index (genesis fallback) - let (mut snapshot, mut dict_novelty) = match root { - Some(root) => { - let loaded = LedgerSnapshot::from_root_bytes(&root.bytes)?; + let (mut snapshot, mut dict_novelty) = match snapshot { + Some(loaded) => { let dn = DictNovelty::with_watermarks( loaded.subject_watermarks.clone(), loaded.string_watermark, From ef50fb0996d45fee8a34285f8531af82d8b4fc90 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 23:03:52 -0400 Subject: [PATCH 20/92] perf(index): compact triple-term pack streams on incremental builds Each incremental build appended a pack to every inner predicate's term stream that gained terms, and nothing merged them, so a predicate's routing table grew by one entry per build forever. The streams now go through the same tail-first compaction as the subject and string streams, within the cycle's shared budget (hoisted so every stream draws on one allowance). Packs a merge consumes, base packs and this cycle's own uploads alike, are recorded as garbage with the term dictionary's replaced reverse-tree leaves. --- fluree-db-api/tests/it_triple_term_links.rs | 93 +++++++++++++++++++++ fluree-db-indexer/src/build/incremental.rs | 54 +++++++++--- 2 files changed, 137 insertions(+), 10 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 5598b552ef..812decdbe1 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -729,3 +729,96 @@ async fn rdf_reifies_with_an_ordinary_object_is_refused() { }; assert!(format!("{err}").contains("reifies"), "{err}"); } + +/// Every incremental build appends a pack to each inner predicate's term +/// stream that gained terms; compaction keeps that stream bounded, records +/// the packs it merged away as garbage, and leaves every handle resolvable. +#[tokio::test] +async fn incremental_term_packs_are_compacted() { + use fluree_db_binary_index::format::index_root::IndexRoot; + use fluree_db_core::{ContentId, ContentStore}; + use fluree_db_indexer::IndexerConfig; + + const CYCLES: u64 = 12; + let fluree = FlureeBuilder::memory() + .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) + .build_memory(); + let ledger_id = "it/triple-term-links:term-pack-compaction"; + let (local, handle) = + support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + + local + .run_until(async { + let mut ledger = support::genesis_ledger(&fluree, ledger_id); + let mut roots: Vec = Vec::new(); + for cycle in 0..CYCLES { + let turtle = format!( + "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:s{cycle} ex:val {cycle} ~ ex:r{cycle} {{| ex:src ex:x |}} .\n" + ); + let r = fluree.upsert_turtle(ledger, &turtle).await.expect("claim"); + support::trigger_index_and_wait(&handle, ledger_id, r.receipt.t).await; + support::wait_for_index_application(&fluree, ledger_id, r.receipt.t).await; + ledger = fluree.ledger(ledger_id).await.expect("reload"); + let record = fluree + .nameservice() + .lookup(ledger_id) + .await + .expect("ns lookup") + .expect("ns record"); + roots.push(record.index_head_id.expect("index root")); + } + + let cs = fluree.content_store(ledger_id); + let mut decoded = Vec::new(); + for cid in &roots { + decoded.push(IndexRoot::decode(&cs.get(cid).await.expect("root")).expect("decode")); + } + let term_packs = |root: &IndexRoot| -> Vec { + root.term_dict + .iter() + .flat_map(|td| td.forward_packs.iter()) + .flat_map(|(_, refs)| refs.iter().map(|r| r.pack_cid.clone())) + .collect() + }; + let final_packs = term_packs(decoded.last().unwrap()); + assert!( + final_packs.len() < CYCLES as usize, + "{} term packs after {CYCLES} appending builds: nothing was compacted", + final_packs.len() + ); + + let consumed: Vec = decoded + .iter() + .flat_map(term_packs) + .filter(|cid| !final_packs.contains(cid)) + .collect(); + assert!(!consumed.is_empty(), "no term pack was merged away"); + let mut garbage = std::collections::HashSet::new(); + for root in &decoded { + let Some(g) = root.garbage.as_ref() else { + continue; + }; + let record: fluree_db_indexer::GarbageRecord = + serde_json::from_slice(&cs.get(&g.id).await.expect("garbage")).expect("parse"); + garbage.extend(record.garbage); + } + for cid in &consumed { + assert!( + garbage.contains(&cid.to_string()), + "merged-away term pack {cid} reached no garbage record" + ); + } + + let got = links(&fluree, &ledger).await; + assert_eq!(got.len(), CYCLES as usize, "{got:#?}"); + for cycle in 0..CYCLES { + assert!( + got.iter().any(|row| row[0].ends_with(&format!("r{cycle}")) + && row[1].contains(&format!("s{cycle}"))), + "reifier r{cycle} lost its term: {got:#?}" + ); + } + }) + .await; +} diff --git a/fluree-db-indexer/src/build/incremental.rs b/fluree-db-indexer/src/build/incremental.rs index d6e805055e..385be7681f 100644 --- a/fluree-db-indexer/src/build/incremental.rs +++ b/fluree-db-indexer/src/build/incremental.rs @@ -988,6 +988,14 @@ pub async fn incremental_index( new_dict_refs.string_reverse = updated.tree_refs; } + // One compaction budget per cycle, shared by every forward-pack stream. + let mut compaction_budget = if compaction_enabled() { + CompactionBudget::new() + } else { + CompactionBudget::disabled() + }; + let mut pack_sizes = PackSizeCache::new(); + // Forward pack updates (FPK1): append new pack artifacts for new subjects/strings, // then compact the tail of each stream that was touched. Appending alone adds a // routing entry per build forever; compaction is what bounds the table by data size. @@ -997,12 +1005,6 @@ pub async fn incremental_index( }; use fluree_db_binary_index::PackBranchEntry; - let mut compaction_budget = if compaction_enabled() { - CompactionBudget::new() - } else { - CompactionBudget::disabled() - }; - let mut pack_sizes = PackSizeCache::new(); // Every pack CID this cycle uploads, so any that compaction consumes // before publication can be garbaged explicitly. let mut uploaded_pack_cids: Vec = Vec::new(); @@ -1272,9 +1274,9 @@ pub async fn incremental_index( root_builder.set_dict_refs(new_dict_refs); - // Triple-term dictionary: append this window's new terms. Packs are - // per inner predicate and only ever append; the reverse tree is updated - // copy-on-write like the subject tree. + // Triple-term dictionary: append this window's new terms to each inner + // predicate's pack stream, then compact the streams that grew, as for + // subjects; the reverse tree is updated copy-on-write like the subject tree. if !novelty.new_terms.is_empty() { use fluree_db_binary_index::dict::forward_pack::KIND_TERM_FWD; use fluree_db_binary_index::dict::incremental::build_incremental_packs_for_stream; @@ -1293,6 +1295,7 @@ pub async fn incremental_index( .or_default() .push((*seq as u64, key.as_slice())); } + let mut uploaded_term_packs: Vec = Vec::new(); for (p_id, entries) in &by_pred { let existing: Vec = forward_packs .iter() @@ -1317,12 +1320,26 @@ pub async fn incremental_index( .put(kind, &pack.bytes) .await .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; + pack_sizes.insert(pack_cid.clone(), pack.bytes.len() as u64); + uploaded_term_packs.push(pack_cid.clone()); updated.push(PackBranchEntry { first_id: pack.first_id, last_id: pack.last_id, pack_cid, }); } + uploaded_term_packs.extend( + compact_forward_packs( + content_store.as_ref(), + kind, + &mut updated, + &mut pack_sizes, + &mut compaction_budget, + CompactionSpans::default(), + &format!("term p_id={p_id}"), + ) + .await?, + ); if let Some(entry) = forward_packs.iter_mut().find(|(p, _)| p == p_id) { entry.1 = updated; } else { @@ -1330,6 +1347,21 @@ pub async fn incremental_index( } } forward_packs.sort_by_key(|(p, _)| *p); + // Packs a merge consumed: base packs no longer routed, and this + // cycle's uploads that a later merge absorbed. + let live: std::collections::HashSet<&ContentId> = forward_packs + .iter() + .flat_map(|(_, refs)| refs.iter().map(|r| &r.pack_cid)) + .collect(); + let mut consumed: Vec = base + .iter() + .flat_map(|b| b.forward_packs.iter()) + .flat_map(|(_, refs)| refs.iter().map(|r| r.pack_cid.clone())) + .chain(uploaded_term_packs) + .filter(|cid| !live.contains(cid)) + .collect(); + consumed.sort(); + consumed.dedup(); let empty_tree = fluree_db_binary_index::DictTreeRefs { branch: fluree_db_core::ContentId::from_hex_digest( @@ -1376,7 +1408,9 @@ pub async fn incremental_index( term_count = refs.term_count, "V6 Phase 3: triple-term dictionary updated" ); - root_builder.set_term_dict(Some(refs), updated_tree.replaced_cids); + let mut replaced = updated_tree.replaced_cids; + replaced.extend(consumed); + root_builder.set_term_dict(Some(refs), replaced); } // Update metadata from resolver state. From e8356ab66b85a57389ae076ecd423a9e2c869300 Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 23:03:52 -0400 Subject: [PATCH 21/92] perf(index): prewarm triple-term packs with the other forward dictionaries The server's background warmer touched string and subject packs only, so the first query to decode reified edges fetched every term pack it needed one at a time. Term packs are now warmed last, within the same budget. --- fluree-db-binary-index/src/dict/term_dict.rs | 58 +++++++++++++++++++ .../src/read/binary_index_store.rs | 15 +++-- 2 files changed, 68 insertions(+), 5 deletions(-) diff --git a/fluree-db-binary-index/src/dict/term_dict.rs b/fluree-db-binary-index/src/dict/term_dict.rs index d36ea8dcaf..4218e3c27b 100644 --- a/fluree-db-binary-index/src/dict/term_dict.rs +++ b/fluree-db-binary-index/src/dict/term_dict.rs @@ -83,6 +83,20 @@ impl TermDictReader { }) } + /// Pre-warm forward packs into the OS page cache, predicate by predicate, + /// up to `budget_bytes`; returns the bytes touched. Blocking, like + /// [`ForwardPackReader::prewarm`]. + pub fn prewarm(&self, budget_bytes: u64) -> u64 { + let mut warmed = 0; + for reader in self.forward.values() { + if warmed >= budget_bytes { + break; + } + warmed += reader.prewarm(budget_bytes - warmed); + } + warmed + } + /// Distinct terms in the dictionary. pub fn term_count(&self) -> u64 { self.term_count @@ -362,6 +376,50 @@ mod tests { assert_eq!(b.watermarks(), vec![(5, 1), (9, 0)]); } + /// The server's background warmer reaches term packs through this: + /// every predicate's stream, in order, within the budget. + #[test] + fn prewarm_covers_every_predicate_stream_within_the_budget() { + let stream = |p_id: u32, n: u64| { + let keys: Vec<[u8; TermKey::LEN]> = + (0..n).map(|i| key(i + 1, p_id, i).to_be_bytes()).collect(); + let entries: Vec<(u64, &[u8])> = keys + .iter() + .enumerate() + .map(|(i, k)| (i as u64, &k[..])) + .collect(); + let packs = crate::dict::pack_builder::build_forward_packs_for_stream( + KIND_TERM_FWD, + pack_ns_code(p_id), + &entries, + DEFAULT_TARGET_PAGE_BYTES, + DEFAULT_TARGET_PACK_BYTES, + ) + .unwrap() + .packs; + let len: u64 = packs.iter().map(|p| p.bytes.len() as u64).sum(); + let reader = ForwardPackReader::from_memory( + packs + .into_iter() + .map(|p| Arc::from(p.bytes.into_boxed_slice())) + .collect(), + ) + .unwrap(); + (reader, len) + }; + let (a, len_a) = stream(5, 40); + let (b, len_b) = stream(9, 70); + let reader = TermDictReader { + forward: BTreeMap::from([(5, a), (9, b)]), + reverse: None, + watermarks: HashMap::new(), + term_count: 110, + }; + assert_eq!(reader.prewarm(u64::MAX), len_a + len_b); + assert_eq!(reader.prewarm(len_a + len_b / 2), len_a + len_b / 2); + assert_eq!(reader.prewarm(0), 0); + } + #[test] fn builder_continues_above_watermarks() { let mut b = TermDictBuilder::above_watermarks(&[(5, 41)]); diff --git a/fluree-db-binary-index/src/read/binary_index_store.rs b/fluree-db-binary-index/src/read/binary_index_store.rs index e033e53b91..dd640b1dd3 100644 --- a/fluree-db-binary-index/src/read/binary_index_store.rs +++ b/fluree-db-binary-index/src/read/binary_index_store.rs @@ -2632,16 +2632,16 @@ impl BinaryIndexStore { Ok(0) } - /// Pre-warm forward-dictionary pages (string + subject packs) into the OS - /// page cache, up to `budget_bytes` total across all packs. Returns the - /// number of bytes touched. + /// Pre-warm forward-dictionary pages (string, subject, then triple-term + /// packs) into the OS page cache, up to `budget_bytes` total across all + /// packs. Returns the number of bytes touched. /// /// The index root and reverse-dict tree readers are already resident after /// [`load_from_root_v6`](Self::load_from_root_v6); this targets the forward /// packs, which are opened lazily and otherwise load on the first query /// that resolves an IRI/string ID. String packs are warmed first (broadest - /// query impact), then per-namespace subject packs. Warming stops once the - /// budget is exhausted. + /// query impact), then per-namespace subject packs, then triple-term packs. + /// Warming stops once the budget is exhausted. /// /// Blocking (page faults / sequential reads) — call from a blocking context /// such as `tokio::task::spawn_blocking`, never on the hot async path. @@ -2653,6 +2653,11 @@ impl BinaryIndexStore { } warmed += reader.prewarm(budget_bytes - warmed); } + if let Some(terms) = &self.dicts.term_dict { + if warmed < budget_bytes { + warmed += terms.prewarm(budget_bytes - warmed); + } + } warmed } From 5f7b2842527badf02be96a9981429b420917b3cf Mon Sep 17 00:00:00 2001 From: bplatz Date: Wed, 30 Sep 2026 23:17:32 -0400 Subject: [PATCH 22/92] fix(query): keep a pending bind's target column alive through join steps A BIND into a variable that an earlier step already bound is an equality check against that column. When the bind's expression waits on a later pattern, the planner holds it back, and each intermediate join step trims its output to the variables still live: later triples, pending filters, and the variables pending binds' expressions read. The bind's own target was not among them, so a target nothing else read was trimmed, the bind found no value and bound afresh, and the check disappeared: a count became the cross product. JSON-LD reaches it with a rebind (`["bind", "?a", "?v"]` after `?a` is bound and one more join): 6 rows instead of 1. The reified-edge lowering reaches it from SPARQL, since every quoted position is a BIND and a shared component variable is joined that way: `<< ex:a ex:p ?o1 >> ... << ?o1 ex:q ?o2 >> ...` under COUNT(*) counted 167,551,488 instead of 189 on the annotation benchmark slice. A pending bind now keeps its target live. --- .../tests/it_query_expression_semantics.rs | 38 +++++++++++++++++++ fluree-db-api/tests/it_triple_term_links.rs | 33 ++++++++++++++++ fluree-db-query/src/execute/where_plan.rs | 10 ++++- 3 files changed, 80 insertions(+), 1 deletion(-) diff --git a/fluree-db-api/tests/it_query_expression_semantics.rs b/fluree-db-api/tests/it_query_expression_semantics.rs index b7c9f77cd6..d99d1ebbed 100644 --- a/fluree-db-api/tests/it_query_expression_semantics.rs +++ b/fluree-db-api/tests/it_query_expression_semantics.rs @@ -2245,3 +2245,41 @@ async fn sparql_stored_lang_equality_with_post_index_novelty() { "LANG() must see the novelty literal's tag" ); } + +// ============================================================================= +// A BIND into a variable an earlier pattern bound is an equality check +// ============================================================================= + +/// The bind's expression waits on the last pattern (`ex:val`, the largest +/// predicate), so the planner holds it back past the `ex:name` join; that +/// join must keep `?a` although nothing else reads it, or the check binds +/// afresh and the count becomes a cross product (6 here). +#[tokio::test] +async fn jsonld_bind_into_a_bound_variable_checks_it_when_only_counted() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger0 = genesis_ledger(&fluree, "exprsem/bind-check:jsonld"); + let tx = json!({ + "@context": ctx(), + "@graph": [ + { "@id": "ex:alice", "ex:age": 30, "ex:name": "Alice" }, + { "@id": "ex:bob", "ex:age": 25, "ex:name": "Bob" }, + { "@id": "ex:t1", "ex:val": 30 }, + { "@id": "ex:t2", "ex:val": 40 }, + { "@id": "ex:t3", "ex:val": 50 } + ] + }); + let ledger = fluree.insert(ledger0, &tx).await.expect("insert").ledger; + let where_ = json!([ + { "@id": "?s", "ex:age": "?a" }, + { "@id": "?s", "ex:name": "?n" }, + { "@id": "?t", "ex:val": "?v" }, + ["bind", "?a", "?v"] + ]); + let counted = json!({ "@context": ctx(), "select": ["(count ?s)"], "where": where_ }); + assert_eq!(jsonld_rows(&fluree, &ledger, &counted).await, json!([[1]])); + let listed = json!({ "@context": ctx(), "select": ["?s"], "where": where_ }); + assert_eq!( + jsonld_rows(&fluree, &ledger, &listed).await, + json!([["ex:alice"]]) + ); +} diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 812decdbe1..2c201ddc3f 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -822,3 +822,36 @@ async fn incremental_term_packs_are_compacted() { }) .await; } + +const CHAINED_CLAIMS: &str = r#"VERSION "1.2" +@prefix ex: . +ex:a ex:p ex:b ~ ex:r1 {| ex:src ex:x |} . +ex:b ex:q ex:c ~ ex:r2 {| ex:src ex:y |} . +ex:d ex:q ex:e ~ ex:r3 {| ex:src ex:z |} . +"#; + +/// Two quoted patterns joined through a component variable: the second +/// edge's subject must be the first edge's object. Both positions lower to a +/// `BIND` of the same variable, so the second is the join; the planner must +/// carry the first one's column to it although only a count reads it. +#[tokio::test] +async fn link_lowering_joins_chained_edges_when_only_counted() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let (fluree, ledger) = import( + &[("chained.ttl", CHAINED_CLAIMS)], + "it/triple-term-links:chained", + ) + .await; + let chain = "<< ex:a ex:p ?o1 >> ex:src ?s1 . << ?o1 ex:q ?o2 >> ex:src ?s2"; + let count = |select: &str| format!("SELECT {select} WHERE {{ {chain} }}"); + for select in ["(COUNT(*) AS ?n)", "(COUNT(DISTINCT ?o1) AS ?n)"] { + let got = run_link_query(&fluree, &ledger, count(select)).await; + assert_eq!(got, vec![vec!["1".to_string()]], "{select}: {got:#?}"); + } + let got = run_link_query(&fluree, &ledger, count("?o1 ?o2")).await; + assert_eq!( + got, + vec![vec!["ex:b".to_string(), "ex:c".to_string()]], + "{got:#?}" + ); +} diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index 9cdc10b1f5..ed4e9cbb47 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -1954,7 +1954,15 @@ fn build_sequential_join_block( .iter() .flat_map(|f| f.expr.referenced_vars()), ); - live.extend(pending_binds.iter().flat_map(|b| b.expr.referenced_vars())); + // A pending bind reads its target too: once an earlier step binds + // that variable, the bind is an equality check against it (the + // reified-edge lowering joins component positions this way). + live.extend(pending_binds.iter().flat_map(|b| { + b.expr + .referenced_vars() + .into_iter() + .chain(std::iter::once(b.var)) + })); live.into_iter().collect::>() }); From a4ce76657d5a9e94d45181538ff7cd924887aba9 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 06:22:53 -0400 Subject: [PATCH 23/92] fix(transact): refuse rdf:reifies with a non-term object after resolving templates The write firewall checked `rdf:reifies` where a surface spells the predicate out. A predicate variable is resolved only when a template is materialized per solution row, so `INSERT { ex:r ?p ex:x } WHERE { VALUES ?p { rdf:reifies } }`, or the JSON-LD update with a `?p` key, wrote a non-term link row. The count plan answers from leaf metadata, which would count it, while the scan drops it. Template materialization now refuses an asserted `rdf:reifies` flake whose resolved object is not a triple term, which covers SPARQL UPDATE, JSON-LD update and every other template path. Retracting a legacy row still works. --- fluree-db-api/tests/it_triple_term_links.rs | 32 +++++++++++++++++++++ fluree-db-transact/src/generate/flakes.rs | 11 +++++++ 2 files changed, 43 insertions(+) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 2c201ddc3f..cdce67b6a1 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -709,6 +709,38 @@ async fn rdf_reifies_with_an_ordinary_object_is_refused() { .expect_err("SPARQL UPDATE of rdf:reifies with an IRI object must be refused"); assert!(format!("{err}").contains("reifies"), "{err}"); + // A predicate variable reaches the predicate only per solution row. + let err = fluree + .stage(&handle) + .sparql_update( + "PREFIX ex: \n\ + PREFIX rdf: \n\ + INSERT { ex:r ?p ex:x } WHERE { VALUES ?p { rdf:reifies } }", + ) + .execute() + .await + .expect_err("a predicate variable bound to rdf:reifies must be refused too"); + assert!(format!("{err}").contains("reifies"), "{err}"); + let ledger = fluree + .ledger("it/triple-term-links:reifies-firewall-sparql") + .await + .expect("reload"); + let err = fluree + .update( + ledger, + &json!({ + "@context": { + "ex": "http://example.org/", + "rdf": "http://www.w3.org/1999/02/22-rdf-syntax-ns#" + }, + "where": [["values", ["?p", [{ "@type": "@id", "@value": "rdf:reifies" }]]]], + "insert": { "@id": "ex:r", "?p": { "@id": "ex:x" } } + }), + ) + .await + .expect_err("a JSON-LD predicate variable bound to rdf:reifies must be refused"); + assert!(format!("{err}").contains("reifies"), "{err}"); + let db_dir = tempfile::tempdir().expect("db tmpdir"); let data_dir = tempfile::tempdir().expect("data tmpdir"); std::fs::write(data_dir.path().join("bad.ttl"), bad).expect("write fixture"); diff --git a/fluree-db-transact/src/generate/flakes.rs b/fluree-db-transact/src/generate/flakes.rs index a95ed834b4..356712edcc 100644 --- a/fluree-db-transact/src/generate/flakes.rs +++ b/fluree-db-transact/src/generate/flakes.rs @@ -236,6 +236,17 @@ impl<'a> FlakeGenerator<'a> { // they hit the index. Bails out the whole transaction. validate_value_dt_pair(&o, &dt)?; + // `rdf:reifies` takes a triple term on every write path. The surface + // checks see literal predicates; a predicate variable is resolved only + // here. Retracting a legacy row stays possible. + if op && fluree_db_core::is_rdf_reifies(&p) && !matches!(o, FlakeValue::TripleTerm(_)) { + return Err(TransactError::InvalidTerm( + "'rdf:reifies' takes a triple term as its object; write the reified \ + triple (`<< s p o >>` or `~ `)" + .to_string(), + )); + } + // Create metadata if language tag or list_index is present let meta_lang = template_lang.or(bound_lang); let meta = FlakeMeta::from_parts(meta_lang.as_deref(), template.list_index); From 8711a81a967ddbe7a9c76a6336d967c65bbc90fb Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 06:26:32 -0400 Subject: [PATCH 24/92] fix(query): term accessors keep the object's datatype for any argument `term_component_binding` produced a binding, with the object's datatype or tag, only when the accessor's argument was a bare variable. Any other argument (`OBJECT(COALESCE(?t))`) fell through to the numeric comparable, which carries no datatype, so `DATATYPE` reported `xsd:integer` for an `xsd:int` object and a `BIND` over the accessor lost the subtype too. The general path now evaluates the argument to a term and builds the same binding; the encoded-handle shortcut for a bare variable is unchanged, and the comparable accessor converts that binding like any other. --- fluree-db-api/tests/it_triple_term_links.rs | 21 +++++++ fluree-db-query/src/eval/rdf.rs | 63 ++++++++++----------- 2 files changed, 51 insertions(+), 33 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index cdce67b6a1..d7e70b0b5c 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -644,6 +644,27 @@ async fn assert_object_types_survive(fluree: &fluree_db_api::Fluree, ledger: &Le got[0][0].ends_with("int") && got[1][0].ends_with("integer"), "{got:#?}" ); + // The accessor's argument need not be a bare variable. + for select in [ + "(DATATYPE(OBJECT(COALESCE(?t))) AS ?dt)", + "(DATATYPE(?o) AS ?dt)", + ] { + let got = run_link_query( + fluree, + ledger, + format!( + "SELECT {select} WHERE {{ ?r rdf:reifies ?t . \ + FILTER(PREDICATE(?t) = ex:size) BIND(OBJECT(COALESCE(?t)) AS ?o) }} \ + ORDER BY ?dt" + ), + ) + .await; + assert_eq!(got.len(), 2, "{select}: {got:#?}"); + assert!( + got[0][0].ends_with("int") && got[1][0].ends_with("integer"), + "{select}: {got:#?}" + ); + } } /// `DATATYPE` and `LANG` over an accessor read the component the way a diff --git a/fluree-db-query/src/eval/rdf.rs b/fluree-db-query/src/eval/rdf.rs index 435241e463..8b8a22586d 100644 --- a/fluree-db-query/src/eval/rdf.rs +++ b/fluree-db-query/src/eval/rdf.rs @@ -557,12 +557,12 @@ fn term_object_binding( )) } -/// The component `SUBJECT|PREDICATE|OBJECT(?term)` names, as a binding: from -/// the dictionary key when the term is a late-materialized handle (one -/// lookup, no materialization), or from the term itself when it is -/// materialized. Either way the object keeps its datatype or tag, so a -/// `BIND`, `DATATYPE` or `LANG` over it sees what a bound variable would. -/// `None` when the argument is not a variable bound to a term. +/// The component `SUBJECT|PREDICATE|OBJECT(arg)` names, as a binding: from +/// the dictionary key when the argument is a variable bound to a +/// late-materialized handle (one lookup, no materialization), else from the +/// term the argument evaluates to. Either way the object keeps its datatype +/// or tag, so a `BIND`, `DATATYPE` or `LANG` over it sees what a bound +/// variable would. `None` when the argument is not a term. pub(crate) fn term_component_binding( func: &Function, args: &[Expression], @@ -578,21 +578,29 @@ pub(crate) fn term_component_binding( }; return Ok(Some(binding)); } - let [Expression::Var(v)] = args else { - return Ok(None); + let name = match func { + Function::TripleSubject => "SUBJECT", + Function::TriplePredicate => "PREDICATE", + Function::TripleObject => "OBJECT", + _ => return Ok(None), }; - let Some(Binding::Lit { - val: fluree_db_core::FlakeValue::TripleTerm(term), - .. - }) = row.get(*v) - else { - return Ok(None); + let term = match args { + [Expression::Var(v)] => match row.get(*v) { + Some(Binding::Lit { + val: fluree_db_core::FlakeValue::TripleTerm(term), + .. + }) => term.as_ref().clone(), + _ => return Ok(None), + }, + _ => match triple_term_arg(args, row, ctx, name)? { + Some(term) => term, + None => return Ok(None), + }, }; Ok(Some(match func { - Function::TripleSubject => Binding::sid(term.s.clone()), - Function::TriplePredicate => Binding::sid(term.p.clone()), - Function::TripleObject => materialized_term_object(term), - _ => return Ok(None), + Function::TripleSubject => Binding::sid(term.s), + Function::TriplePredicate => Binding::sid(term.p), + _ => materialized_term_object(&term), })) } @@ -616,7 +624,7 @@ fn materialized_term_object(term: &fluree_db_core::TripleTermValue) -> Binding { } /// One accessor: the component's binding converted exactly as a bound -/// variable is, else the value path for a non-variable argument. +/// variable is. fn eval_term_accessor( func: Function, args: &[Expression], @@ -625,21 +633,10 @@ fn eval_term_accessor( name: &str, ) -> Result> { check_arity(args, 1, name)?; - if let Some(binding) = term_component_binding(&func, args, row, ctx)? { - return super::binding_to_comparable(Some(&binding), ctx); - } - if matches!(args[0], Expression::Var(_)) { - return Ok(None); + match term_component_binding(&func, args, row, ctx)? { + Some(binding) => super::binding_to_comparable(Some(&binding), ctx), + None => Ok(None), } - let Some(term) = triple_term_arg(args, row, ctx, name)? else { - return Ok(None); - }; - let binding = match func { - Function::TripleSubject => Binding::sid(term.s), - Function::TriplePredicate => Binding::sid(term.p), - _ => materialized_term_object(&term), - }; - super::binding_to_comparable(Some(&binding), ctx) } pub fn eval_triple_subject( From fc67ac3c5215da1fa3f39588a909c48ab979d342 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 06:30:01 -0400 Subject: [PATCH 25/92] fix(index): give term pack streams a budget share and a maintenance sweep Term streams were compacted only in a cycle that added terms to them, after the string and subject streams had drawn on the shared budget, so a term stream that missed its turn and then went quiet stayed fragmented for good. The term dictionary now reserves a third of each cycle's compaction budget before the other dictionaries run, and takes over what they leave. Its streams that grew are compacted first, then the rest in a rotation keyed by the base index t, as the subject maintenance sweep does; a cycle with no new terms still runs the sweep and publishes the merged routing table. --- fluree-db-indexer/src/build/incremental.rs | 432 ++++++++++++++------- 1 file changed, 301 insertions(+), 131 deletions(-) diff --git a/fluree-db-indexer/src/build/incremental.rs b/fluree-db-indexer/src/build/incremental.rs index 385be7681f..7c0a1245b3 100644 --- a/fluree-db-indexer/src/build/incremental.rs +++ b/fluree-db-indexer/src/build/incremental.rs @@ -989,11 +989,19 @@ pub async fn incremental_index( } // One compaction budget per cycle, shared by every forward-pack stream. + // The term dictionary's streams are served last, so they get a reserved + // share (one of three dictionaries) plus whatever the others leave. let mut compaction_budget = if compaction_enabled() { CompactionBudget::new() } else { CompactionBudget::disabled() }; + let mut term_budget = if novelty.base_root.term_dict.is_some() || !novelty.new_terms.is_empty() + { + compaction_budget.split_off(3) + } else { + CompactionBudget::disabled() + }; let mut pack_sizes = PackSizeCache::new(); // Forward pack updates (FPK1): append new pack artifacts for new subjects/strings, @@ -1275,142 +1283,81 @@ pub async fn incremental_index( root_builder.set_dict_refs(new_dict_refs); // Triple-term dictionary: append this window's new terms to each inner - // predicate's pack stream, then compact the streams that grew, as for - // subjects; the reverse tree is updated copy-on-write like the subject tree. - if !novelty.new_terms.is_empty() { - use fluree_db_binary_index::dict::forward_pack::KIND_TERM_FWD; - use fluree_db_binary_index::dict::incremental::build_incremental_packs_for_stream; - use fluree_db_binary_index::dict::term_dict::pack_ns_code; - use fluree_db_binary_index::format::wire_helpers::PackBranchEntry; - let base = novelty.base_root.term_dict.clone(); - let mut forward_packs: Vec<(u32, Vec)> = base - .as_ref() - .map(|b| b.forward_packs.clone()) - .unwrap_or_default(); - let mut by_pred: std::collections::BTreeMap> = - std::collections::BTreeMap::new(); - for (p_id, seq, key) in &novelty.new_terms { - by_pred - .entry(*p_id) - .or_default() - .push((*seq as u64, key.as_slice())); - } - let mut uploaded_term_packs: Vec = Vec::new(); - for (p_id, entries) in &by_pred { - let existing: Vec = forward_packs - .iter() - .find(|(p, _)| p == p_id) - .map(|(_, refs)| refs.clone()) - .unwrap_or_default(); - let pack_result = build_incremental_packs_for_stream( - KIND_TERM_FWD, - pack_ns_code(*p_id), - &existing, - entries, - ) - .map_err(|e| { - IndexerError::StorageWrite(format!("term fwd pack build p_id={p_id}: {e}")) - })?; - let kind = ContentKind::DictBlob { - dict: fluree_db_core::DictKind::TermForward { p_id: *p_id }, - }; - let mut updated = existing; - for pack in &pack_result.new_packs { - let pack_cid = content_store - .put(kind, &pack.bytes) - .await - .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; - pack_sizes.insert(pack_cid.clone(), pack.bytes.len() as u64); - uploaded_term_packs.push(pack_cid.clone()); - updated.push(PackBranchEntry { - first_id: pack.first_id, - last_id: pack.last_id, - pack_cid, - }); + // predicate's pack stream and compact, as for subjects; the reverse tree is + // updated copy-on-write like the subject tree. A cycle without new terms + // still compacts streams an earlier cycle left fragmented. + let base_terms = novelty.base_root.term_dict.clone(); + if !novelty.new_terms.is_empty() || base_terms.is_some() { + term_budget.absorb(std::mem::replace( + &mut compaction_budget, + CompactionBudget::disabled(), + )); + let (forward_packs, consumed) = update_term_forward_packs( + content_store.as_ref(), + base_terms.as_ref(), + &novelty.new_terms, + base_root.index_t.unsigned_abs(), + &mut pack_sizes, + &mut term_budget, + ) + .await?; + + if novelty.new_terms.is_empty() { + // Maintenance only: the dictionary changed if a merge ran. + if let (Some(mut refs), false) = (base_terms, consumed.is_empty()) { + refs.forward_packs = forward_packs; + root_builder.set_term_dict(Some(refs), consumed); } - uploaded_term_packs.extend( - compact_forward_packs( + } else { + let empty_tree = fluree_db_binary_index::DictTreeRefs { + branch: fluree_db_core::ContentId::from_hex_digest( + fluree_db_core::content_kind::CODEC_FLUREE_DICT_BLOB, + &fluree_db_core::sha256_hex(b""), + ) + .expect("valid digest"), + leaves: Vec::new(), + }; + let (base_reverse, base_count) = match &base_terms { + Some(b) => (b.reverse.clone(), b.term_count), + None => (empty_tree, 0), + }; + let updated_tree = if base_terms.is_some() { + super::dicts::upload_incremental_reverse_tree_async_terms( content_store.as_ref(), - kind, - &mut updated, - &mut pack_sizes, - &mut compaction_budget, - CompactionSpans::default(), - &format!("term p_id={p_id}"), + &base_reverse, + &novelty.new_terms, + warm_cache.as_deref(), ) - .await?, - ); - if let Some(entry) = forward_packs.iter_mut().find(|(p, _)| p == p_id) { - entry.1 = updated; + .await? } else { - forward_packs.push((*p_id, updated)); - } + // No base dictionary: build the tree from scratch through the + // same core, against an empty existing tree. + super::dicts::upload_incremental_reverse_tree_async_terms( + content_store.as_ref(), + &fluree_db_binary_index::DictTreeRefs { + branch: base_reverse.branch.clone(), + leaves: Vec::new(), + }, + &novelty.new_terms, + warm_cache.as_deref(), + ) + .await? + }; + let refs = fluree_db_binary_index::TermDictRefs { + forward_packs, + reverse: updated_tree.tree_refs, + watermarks: novelty.term_watermarks.clone(), + term_count: base_count + novelty.new_terms.len() as u64, + }; + tracing::debug!( + new_terms = novelty.new_terms.len(), + term_count = refs.term_count, + "V6 Phase 3: triple-term dictionary updated" + ); + let mut replaced = updated_tree.replaced_cids; + replaced.extend(consumed); + root_builder.set_term_dict(Some(refs), replaced); } - forward_packs.sort_by_key(|(p, _)| *p); - // Packs a merge consumed: base packs no longer routed, and this - // cycle's uploads that a later merge absorbed. - let live: std::collections::HashSet<&ContentId> = forward_packs - .iter() - .flat_map(|(_, refs)| refs.iter().map(|r| &r.pack_cid)) - .collect(); - let mut consumed: Vec = base - .iter() - .flat_map(|b| b.forward_packs.iter()) - .flat_map(|(_, refs)| refs.iter().map(|r| r.pack_cid.clone())) - .chain(uploaded_term_packs) - .filter(|cid| !live.contains(cid)) - .collect(); - consumed.sort(); - consumed.dedup(); - - let empty_tree = fluree_db_binary_index::DictTreeRefs { - branch: fluree_db_core::ContentId::from_hex_digest( - fluree_db_core::content_kind::CODEC_FLUREE_DICT_BLOB, - &fluree_db_core::sha256_hex(b""), - ) - .expect("valid digest"), - leaves: Vec::new(), - }; - let (base_reverse, base_count) = match &base { - Some(b) => (b.reverse.clone(), b.term_count), - None => (empty_tree, 0), - }; - let updated_tree = if base.is_some() { - super::dicts::upload_incremental_reverse_tree_async_terms( - content_store.as_ref(), - &base_reverse, - &novelty.new_terms, - warm_cache.as_deref(), - ) - .await? - } else { - // No base dictionary: build the tree from scratch through the - // same core, against an empty existing tree. - super::dicts::upload_incremental_reverse_tree_async_terms( - content_store.as_ref(), - &fluree_db_binary_index::DictTreeRefs { - branch: base_reverse.branch.clone(), - leaves: Vec::new(), - }, - &novelty.new_terms, - warm_cache.as_deref(), - ) - .await? - }; - let refs = fluree_db_binary_index::TermDictRefs { - forward_packs, - reverse: updated_tree.tree_refs, - watermarks: novelty.term_watermarks.clone(), - term_count: base_count + novelty.new_terms.len() as u64, - }; - tracing::debug!( - new_terms = novelty.new_terms.len(), - term_count = refs.term_count, - "V6 Phase 3: triple-term dictionary updated" - ); - let mut replaced = updated_tree.replaced_cids; - replaced.extend(consumed); - root_builder.set_term_dict(Some(refs), replaced); } // Update metadata from resolver state. @@ -4335,6 +4282,22 @@ impl CompactionBudget { granted } + /// Split off `1 / divisor` of what remains, for streams that must not be + /// starved by the ones served before them. + fn split_off(&mut self, divisor: u64) -> Self { + let bytes = self.bytes / divisor; + let requests = self.requests / divisor as usize; + self.bytes -= bytes; + self.requests -= requests; + Self { bytes, requests } + } + + /// Take over what another budget left unspent. + fn absorb(&mut self, other: Self) { + self.bytes += other.bytes; + self.requests += other.requests; + } + /// Reserve exactly `n` operations, or nothing at all. /// /// A merge is indivisible — it either fetches every input pack or does not @@ -4423,6 +4386,137 @@ async fn pack_encoded_sizes( sizes.into_iter().map_while(|s| s).collect() } +/// Append new terms to their inner predicates' pack streams and compact: +/// the streams that grew first, then the others in rotation (by `rotation`, +/// so no cursor is persisted), until `budget` runs out. Returns the routing +/// table and the packs a merge consumed (base packs no longer routed, and this +/// cycle's uploads a later merge absorbed), which are garbage. +async fn update_term_forward_packs( + content_store: &dyn ContentStore, + base: Option<&fluree_db_binary_index::TermDictRefs>, + new_terms: &[(u32, u32, Vec)], + rotation: u64, + pack_sizes: &mut PackSizeCache, + budget: &mut CompactionBudget, +) -> Result<( + Vec<(u32, Vec)>, + Vec, +)> { + use fluree_db_binary_index::dict::forward_pack::KIND_TERM_FWD; + use fluree_db_binary_index::dict::incremental::build_incremental_packs_for_stream; + use fluree_db_binary_index::dict::term_dict::pack_ns_code; + use fluree_db_binary_index::PackBranchEntry; + + let kind = |p_id: u32| ContentKind::DictBlob { + dict: fluree_db_core::DictKind::TermForward { p_id }, + }; + let mut forward_packs: Vec<(u32, Vec)> = + base.map(|b| b.forward_packs.clone()).unwrap_or_default(); + let mut by_pred: std::collections::BTreeMap> = + std::collections::BTreeMap::new(); + for (p_id, seq, key) in new_terms { + by_pred + .entry(*p_id) + .or_default() + .push((u64::from(*seq), key.as_slice())); + } + + let mut uploaded: Vec = Vec::new(); + for (p_id, entries) in &by_pred { + let existing: Vec = forward_packs + .iter() + .find(|(p, _)| p == p_id) + .map(|(_, refs)| refs.clone()) + .unwrap_or_default(); + let pack_result = build_incremental_packs_for_stream( + KIND_TERM_FWD, + pack_ns_code(*p_id), + &existing, + entries, + ) + .map_err(|e| IndexerError::StorageWrite(format!("term fwd pack build p_id={p_id}: {e}")))?; + let mut updated = existing; + for pack in &pack_result.new_packs { + let pack_cid = content_store + .put(kind(*p_id), &pack.bytes) + .await + .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; + pack_sizes.insert(pack_cid.clone(), pack.bytes.len() as u64); + uploaded.push(pack_cid.clone()); + updated.push(PackBranchEntry { + first_id: pack.first_id, + last_id: pack.last_id, + pack_cid, + }); + } + uploaded.extend( + compact_forward_packs( + content_store, + kind(*p_id), + &mut updated, + pack_sizes, + budget, + CompactionSpans::default(), + &format!("term p_id={p_id}"), + ) + .await?, + ); + if let Some(entry) = forward_packs.iter_mut().find(|(p, _)| p == p_id) { + entry.1 = updated; + } else { + forward_packs.push((*p_id, updated)); + } + } + + // Maintenance: a stream that went quiet keeps whatever fragmentation an + // earlier cycle could not afford to merge. + let mut quiet: Vec = forward_packs + .iter() + .filter(|(p, refs)| refs.len() > 1 && !by_pred.contains_key(p)) + .map(|(p, _)| *p) + .collect(); + if !quiet.is_empty() { + let offset = (rotation as usize) % quiet.len(); + quiet.rotate_left(offset); + } + for p_id in quiet { + if budget.exhausted() { + break; + } + let Some(entry) = forward_packs.iter_mut().find(|(p, _)| *p == p_id) else { + continue; + }; + uploaded.extend( + compact_forward_packs( + content_store, + kind(p_id), + &mut entry.1, + pack_sizes, + budget, + CompactionSpans::default(), + &format!("term p_id={p_id} (maintenance)"), + ) + .await?, + ); + } + forward_packs.sort_by_key(|(p, _)| *p); + + let live: std::collections::HashSet<&ContentId> = forward_packs + .iter() + .flat_map(|(_, refs)| refs.iter().map(|r| &r.pack_cid)) + .collect(); + let mut consumed: Vec = base + .iter() + .flat_map(|b| b.forward_packs.iter()) + .flat_map(|(_, refs)| refs.iter().map(|r| r.pack_cid.clone())) + .chain(uploaded) + .filter(|cid| !live.contains(cid)) + .collect(); + consumed.sort(); + consumed.dedup(); + Ok((forward_packs, consumed)) +} + /// Merge qualifying runs within one forward-dictionary stream, in place. /// /// Examines the **tail** first and unconditionally, because new packs land @@ -4982,6 +5076,82 @@ mod compaction_tests { refs } + /// Upload one small term pack of predicate `p_id` covering + /// `[first_seq, first_seq + count)`. + async fn put_term_pack( + store: &dyn ContentStore, + p_id: u32, + first_seq: u64, + count: usize, + ) -> PackBranchEntry { + use fluree_db_binary_index::dict::forward_pack::KIND_TERM_FWD; + use fluree_db_binary_index::dict::term_dict::pack_ns_code; + let owned: Vec<(u64, Vec)> = (0..count) + .map(|i| (first_seq + i as u64, vec![i as u8; 22])) + .collect(); + let refs: Vec<(u64, &[u8])> = owned.iter().map(|(id, v)| (*id, v.as_slice())).collect(); + let bytes = encode_forward_pack(&refs, KIND_TERM_FWD, pack_ns_code(p_id), 512).unwrap(); + let kind = ContentKind::DictBlob { + dict: fluree_db_core::DictKind::TermForward { p_id }, + }; + PackBranchEntry { + first_id: first_seq, + last_id: first_seq + count as u64 - 1, + pack_cid: store.put(kind, &bytes).await.unwrap(), + } + } + + /// A term stream fragmented by earlier cycles is compacted in a cycle + /// that brings it no terms, from the term dictionary's reserved share + /// even when the other dictionaries spent the rest of the budget. + #[tokio::test] + async fn quiet_term_streams_are_compacted_from_their_share() { + let storage = MemoryStorage::new(); + let store = content_store_for(storage.clone(), LEDGER); + let mut fragmented = Vec::new(); + for i in 0..9u64 { + fragmented.push(put_term_pack(&store, 7, i * 3, 3).await); + } + let single = vec![put_term_pack(&store, 9, 0, 3).await]; + let base = fluree_db_binary_index::TermDictRefs { + forward_packs: vec![(7, fragmented.clone()), (9, single.clone())], + reverse: fluree_db_binary_index::DictTreeRefs { + branch: fragmented[0].pack_cid.clone(), + leaves: Vec::new(), + }, + watermarks: vec![(7, 26), (9, 2)], + term_count: 30, + }; + + let mut cycle = CompactionBudget::new(); + let mut terms = cycle.split_off(3); + cycle.requests = 0; + terms.absorb(cycle); + + let (packs, consumed) = update_term_forward_packs( + &store, + Some(&base), + &[], + 42, + &mut PackSizeCache::new(), + &mut terms, + ) + .await + .expect("maintenance"); + let stream = |p: u32| &packs.iter().find(|(q, _)| *q == p).unwrap().1; + assert!(stream(7).len() < fragmented.len(), "{:?}", stream(7)); + assert_eq!(stream(7).first().unwrap().first_id, 0); + assert_eq!(stream(7).last().unwrap().last_id, 26); + assert_eq!(stream(9), &single, "a single-pack stream is left alone"); + assert!(!consumed.is_empty()); + assert!( + consumed + .iter() + .all(|cid| fragmented.iter().any(|e| &e.pack_cid == cid)), + "only merged-away packs are garbage: {consumed:?}" + ); + } + #[tokio::test] async fn tail_is_compacted_even_when_the_scan_budget_cannot_reach_it() { // New packs always land at the tail, so a design that sweeps from the From 60a2892e90b9a41507dcf9452c3f56181ad33af4 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 06:36:40 -0400 Subject: [PATCH 26/92] fix(index): a dictionary range starting between two leaves reads the later one `DictTreeReader::reverse_range_scan` found its first leaf with `find_leaf`, which answers only for a key inside some leaf's [first, last] range. A start key in the gap between two leaves (past one leaf's last key, before the next one's first) found no leaf, and the scan returned nothing although the later leaves held matching keys. A subject prefix lookup (`find_subjects_by_prefix`) whose prefix sorts into such a gap missed every match. The scan now starts at the first leaf whose last key is not below the start key. --- fluree-db-binary-index/src/dict/reader.rs | 51 +++++++++++++++++------ 1 file changed, 38 insertions(+), 13 deletions(-) diff --git a/fluree-db-binary-index/src/dict/reader.rs b/fluree-db-binary-index/src/dict/reader.rs index 3ce151ffd5..d6a22ac807 100644 --- a/fluree-db-binary-index/src/dict/reader.rs +++ b/fluree-db-binary-index/src/dict/reader.rs @@ -439,19 +439,13 @@ impl DictTreeReader { return Ok(Vec::new()); } - // Find the first leaf that might contain start_key. - let start_leaf = match self.branch.find_leaf(start_key) { - Some(idx) => idx, - None => { - // start_key is before the first leaf or after the last. - // If the first leaf's first_key >= start_key, it might have matches. - if self.branch.leaves[0].first_key.as_slice() >= start_key { - 0 - } else { - return Ok(Vec::new()); - } - } - }; + // The first leaf that can hold a key >= start_key: leaves are sorted + // and disjoint, so it is the first whose last key is not below it. A + // start key in the gap between two leaves begins at the later one. + let start_leaf = self + .branch + .leaves + .partition_point(|leaf| leaf.last_key.as_slice() < start_key); let mut results = Vec::new(); @@ -849,6 +843,37 @@ mod tests { DictTreeReader::from_memory(result.branch, leaf_map) } + /// A range whose start key sorts between two leaves (after one leaf's + /// last key, before the next one's first) begins at the later leaf. + #[test] + fn range_scan_starting_between_leaves_reads_the_next_leaf() { + let entries: Vec = (0..64u64) + .map(|i| ReverseEntry { + key: format!("k{:04}", i * 2).into_bytes(), + id: i, + }) + .collect(); + let result = builder::build_reverse_tree(entries, 64).unwrap(); + assert!(result.branch.leaves.len() > 2); + let gap_after = result.branch.leaves[0].last_key.clone(); + let next_first = result.branch.leaves[1].first_key.clone(); + let mut leaf_map = HashMap::new(); + for (leaf_artifact, branch_leaf) in result.leaves.iter().zip(result.branch.leaves.iter()) { + leaf_map.insert(branch_leaf.address.clone(), leaf_artifact.bytes.clone()); + } + let reader = DictTreeReader::from_memory(result.branch, leaf_map); + + // Odd numbers are absent, so "last key + 1" lies in the gap. + let mut start = gap_after.clone(); + *start.last_mut().unwrap() += 1; + assert!( + start.as_slice() > gap_after.as_slice() && start.as_slice() < next_first.as_slice() + ); + let got = reader.reverse_range_scan(&start, b"k9999").unwrap(); + assert_eq!(got.first().map(|(k, _)| k.clone()), Some(next_first)); + assert_eq!(got.len() as u64, 64 - got[0].1); + } + /// Content store whose blobs live in files it hands out through /// `resolve_local_path`, counting how often a reload asks — the probe a /// reload is meant to skip for every leaf it already had a path for. From 15995e5656af36354af3fa637e79adf879c551c2 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 07:07:58 -0400 Subject: [PATCH 27/92] feat(query): relate a triple term to its components as a joinable pattern The link lowering decomposed `<< s p o >>` into `?r rdf:reifies ?t` plus a BIND per variable position. Component equality was then visible only after the link scan: `<< X :treats ?o1 >> ... << ?o1 :causes ?o2 >> ...` read every `:causes` link once per left row and joined on `?o1` in the BIND check, 12 s against 90 ms for the bundle chain on the annotation benchmark slice. Quoted patterns now lower to the link plus `TermComponents(?t, s, p, o)`, an internal pattern the planner places like any other source. Its operator finds candidate terms per input row: - the term variable bound: one dictionary key decode supplies every component; - the subject bound, as a constant or by an earlier pattern: the reverse tree's subject-first `(s_id[, p_id])` prefix range, which yields keys and handles together; distinct prefixes are read once per batch; - neither: the terms of the bound predicate, or of the dictionary. A variable component another pattern bound first is joined on with the same equality a BIND check applied. The planner prices a bound term as a decode and an anchored subject as a small source, so the chained edge is found through its subject and its link becomes a bound-object lookup; the link keeps the graph, time, policy and novelty checks. Constant components stay filters on the link as well, so a link-first plan keeps its handle interval. Unread positions are elided as before, and a pattern left with nothing to bind or anchor goes, which keeps the count plan. On the slice the populated chained query takes 60-75 ms (bundle chain 92 ms), with every other measured shape unchanged or faster. --- fluree-db-api/src/explain.rs | 13 + fluree-db-api/tests/it_triple_term_links.rs | 59 +++ fluree-db-binary-index/src/dict/term_dict.rs | 128 +++++ fluree-db-query/src/binary_scan.rs | 4 +- fluree-db-query/src/datalog_rules/validate.rs | 2 + fluree-db-query/src/eval/rdf.rs | 4 +- fluree-db-query/src/execute/operator_tree.rs | 5 +- fluree-db-query/src/execute/where_plan.rs | 70 ++- fluree-db-query/src/explain.rs | 4 + fluree-db-query/src/ir.rs | 2 + fluree-db-query/src/ir/pattern.rs | 18 + fluree-db-query/src/ir/term_components.rs | 58 ++ fluree-db-query/src/lib.rs | 1 + fluree-db-query/src/planner.rs | 19 + fluree-db-query/src/r2rml/rewrite.rs | 6 + fluree-db-query/src/rewrite.rs | 1 + fluree-db-query/src/rewrite_owl_ql.rs | 1 + fluree-db-query/src/term_components.rs | 498 ++++++++++++++++++ fluree-db-sparql/src/lower/annotation.rs | 114 ++-- fluree-db-sparql/src/lower/construct.rs | 1 + fluree-db-sparql/src/lower/mod.rs | 7 + 21 files changed, 945 insertions(+), 70 deletions(-) create mode 100644 fluree-db-query/src/ir/term_components.rs create mode 100644 fluree-db-query/src/term_components.rs diff --git a/fluree-db-api/src/explain.rs b/fluree-db-api/src/explain.rs index 6fe5ea66ae..b86d58c456 100644 --- a/fluree-db-api/src/explain.rs +++ b/fluree-db-api/src/explain.rs @@ -281,6 +281,19 @@ fn logical_node( ); node.insert("rows".into(), json!(rows.len())); } + Pattern::TermComponents(tc) => { + node.insert("kind".into(), json!("term-components")); + node.insert("term".into(), json!(vars.name(tc.term).to_string())); + node.insert( + "vars".into(), + json!(tc + .components() + .into_iter() + .filter_map(fluree_db_query::ir::Component::var) + .map(|v| vars.name(v).to_string()) + .collect::>()), + ); + } Pattern::PropertyPath(pp) => { node.insert("kind".into(), json!("property-path")); node.insert( diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index d7e70b0b5c..41df65d23c 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -908,3 +908,62 @@ async fn link_lowering_joins_chained_edges_when_only_counted() { "{got:#?}" ); } + +/// A quoted edge whose subject a previous edge bound is found through its +/// components (the reverse tree's subject prefix) and its link read as a +/// bound-object lookup, rather than every link of the predicate being read +/// and decoded per input row. (Which edge leads is a cost decision; on three +/// links the first edge's scan is cheapest.) +#[tokio::test] +async fn link_lowering_drives_chained_edges_through_their_subjects() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let alias = "it/triple-term-links:chained-plan"; + let (fluree, _ledger) = import(&[("chained.ttl", CHAINED_CLAIMS)], alias).await; + let view = fluree.db(alias).await.expect("view"); + let plan = fluree + .explain_sparql( + &view, + "PREFIX ex: \n\ + SELECT (COUNT(*) AS ?n) WHERE { \ + << ex:a ex:p ?o1 >> ex:src ?s1 . << ?o1 ex:q ?o2 >> ex:src ?s2 }", + ) + .await + .expect("explain"); + + /// The `TermComponentsOperator` for `term` in this subtree, if any. + fn components_for<'a>(node: &'a JsonValue, term: &str) -> Option<&'a JsonValue> { + if node["op"] == "TermComponentsOperator" && node["details"]["term"] == term { + return Some(node); + } + node["children"] + .as_array() + .into_iter() + .flatten() + .find_map(|c| components_for(&c["node"], term)) + } + fn link_joins(node: &JsonValue, out: &mut Vec) { + if node["op"] == "NestedLoopJoinOperator" + && node["details"]["right"] + .as_str() + .is_some_and(|r| r.contains("reifies")) + { + out.push(node.clone()); + } + for c in node["children"].as_array().into_iter().flatten() { + link_joins(&c["node"], out); + } + } + let mut joins = Vec::new(); + link_joins(&plan["plan"]["physical"], &mut joins); + assert!( + !joins.is_empty(), + "the chained link must be a lookup: {plan:#}" + ); + for join in &joins { + let right = join["details"]["right"].as_str().unwrap(); + let term = right.rsplit(' ').next().unwrap(); + let components = components_for(join, term) + .unwrap_or_else(|| panic!("link {right} was read without its components: {plan:#}")); + assert_eq!(components["details"]["access"], "subject", "{plan:#}"); + } +} diff --git a/fluree-db-binary-index/src/dict/term_dict.rs b/fluree-db-binary-index/src/dict/term_dict.rs index 4218e3c27b..f2f8794849 100644 --- a/fluree-db-binary-index/src/dict/term_dict.rs +++ b/fluree-db-binary-index/src/dict/term_dict.rs @@ -32,6 +32,18 @@ pub fn pack_ns_code(p_id: u32) -> u16 { (p_id & 0xFFFF) as u16 } +/// The smallest reverse key past every key of subject `s_id`. +fn next_subject(s_id: u64) -> [u8; TermKey::LEN] { + match s_id.checked_add(1) { + Some(next) => { + let mut end = [0u8; TermKey::LEN]; + end[..8].copy_from_slice(&next.to_be_bytes()); + end + } + None => [0xFF; TermKey::LEN], + } +} + // ── Reader ────────────────────────────────────────────────────────────────── /// Read side of the triple-term dictionary. @@ -132,6 +144,65 @@ impl TermDictReader { } } + /// Every term whose base edge has subject `s_id` (and predicate `p_id`, + /// when given), with its handle: one range over the subject-first reverse + /// tree, which yields the keys themselves, so nothing is decoded. + pub fn terms_with_subject( + &self, + s_id: u64, + p_id: Option, + ) -> io::Result> { + let Some(tree) = &self.reverse else { + return Ok(Vec::new()); + }; + let mut start = [0u8; TermKey::LEN]; + start[..8].copy_from_slice(&s_id.to_be_bytes()); + let mut end = [0u8; TermKey::LEN]; + // Exclusive end: the next subject, or the next predicate of this one. + match p_id { + Some(p) => { + start[8..12].copy_from_slice(&p.to_be_bytes()); + match p.checked_add(1) { + Some(next) => { + end[..8].copy_from_slice(&s_id.to_be_bytes()); + end[8..12].copy_from_slice(&next.to_be_bytes()); + } + None => end = next_subject(s_id), + } + } + None => end = next_subject(s_id), + } + tree.reverse_range_scan(&start, &end)? + .into_iter() + .map(|(bytes, handle)| { + TermKey::from_be_bytes(&bytes) + .map(|key| (key, handle)) + .ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + "term reverse key has the wrong width", + ) + }) + }) + .collect() + } + + /// Every term of inner predicate `p_id`, with its handle, read through the + /// forward packs in sequence order. + pub fn terms_of_predicate(&self, p_id: u32) -> io::Result> { + let Some(watermark) = self.watermark(p_id) else { + return Ok(Vec::new()); + }; + let mut out = Vec::with_capacity(watermark as usize + 1); + for seq in 0..=watermark { + let handle = term_handle(p_id, seq); + if let Some(key) = self.resolve(handle)? { + out.push((key, handle)); + } + } + Ok(out) + } + /// The encoded base edge behind `handle`, or `None` for an unknown handle. pub fn resolve(&self, handle: u64) -> io::Result> { let Some(reader) = self.forward.get(&term_handle_p_id(handle)) else { @@ -420,6 +491,63 @@ mod tests { assert_eq!(reader.prewarm(0), 0); } + /// A subject prefix (and subject + predicate prefix) selects exactly its + /// terms, across leaves, at the edges of the key space too. + #[test] + fn terms_with_subject_reads_the_prefix_range() { + use crate::dict::builder::build_reverse_tree; + use crate::dict::reader::DictTreeReader; + let keys = [ + key(4, 9, 1), + key(5, 2, 7), + key(5, 9, 1), + key(5, 9, 2), + key(5, u32::MAX, 3), + key(6, 0, 0), + key(u64::MAX, 1, 1), + ]; + let mut entries: Vec = keys + .iter() + .enumerate() + .map(|(i, k)| ReverseEntry { + key: k.to_be_bytes().to_vec(), + id: i as u64 + 100, + }) + .collect(); + entries.sort_by(|a, b| a.key.cmp(&b.key)); + let built = build_reverse_tree(entries, 64).unwrap(); + let leaves = built + .leaves + .iter() + .zip(&built.branch.leaves) + .map(|(leaf, entry)| (entry.address.clone(), leaf.bytes.clone())) + .collect(); + assert!(built.branch.leaves.len() > 1, "the range must cross leaves"); + let reader = TermDictReader { + forward: BTreeMap::new(), + reverse: Some(Arc::new(DictTreeReader::from_memory(built.branch, leaves))), + watermarks: HashMap::new(), + term_count: keys.len() as u64, + }; + let handles = |s: u64, p: Option| -> Vec { + reader + .terms_with_subject(s, p) + .unwrap() + .into_iter() + .map(|(k, h)| { + assert_eq!(k, keys[(h - 100) as usize]); + h + }) + .collect() + }; + assert_eq!(handles(5, None), vec![101, 102, 103, 104]); + assert_eq!(handles(5, Some(9)), vec![102, 103]); + assert_eq!(handles(5, Some(u32::MAX)), vec![104]); + assert_eq!(handles(5, Some(3)), Vec::::new()); + assert_eq!(handles(u64::MAX, None), vec![106]); + assert_eq!(handles(7, None), Vec::::new()); + } + #[test] fn builder_continues_above_watermarks() { let mut b = TermDictBuilder::above_watermarks(&[(5, 41)]); diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index eb24c28a7d..04bb576668 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -3765,7 +3765,7 @@ pub(crate) fn translate_one_flake_v3_pub( } /// Resolve a subject Sid to s_id using persisted dict then DictNovelty. -fn resolve_subject_v3( +pub(crate) fn resolve_subject_v3( sid: &Sid, store: &BinaryIndexStore, dict_novelty: Option<&Arc>, @@ -3848,7 +3848,7 @@ fn string_not_found_error(value: &str) -> std::io::Error { /// - langString: OType must embed the lang_id, not use XSD_STRING /// - numeric subtypes: xsd:int vs xsd:integer can share the same FlakeValue::Long /// - string subtypes: xsd:anyURI vs xsd:string share FlakeValue::String -fn value_to_otype_okey( +pub(crate) fn value_to_otype_okey( val: &FlakeValue, dt_sid: &Sid, lang: Option<&str>, diff --git a/fluree-db-query/src/datalog_rules/validate.rs b/fluree-db-query/src/datalog_rules/validate.rs index 64038f1540..26de19669e 100644 --- a/fluree-db-query/src/datalog_rules/validate.rs +++ b/fluree-db-query/src/datalog_rules/validate.rs @@ -114,6 +114,8 @@ fn walk_patterns( // rule against. The protection that matters is the exhaustive // match itself — a surface that starts producing one has to come // back here first. + // Reads the ledger's own term dictionary. + Pattern::TermComponents(_) => {} Pattern::R2rml(_) => return Err(non_local(rule, "an R2RML graph source")), Pattern::Service(_) => { return Err(reject( diff --git a/fluree-db-query/src/eval/rdf.rs b/fluree-db-query/src/eval/rdf.rs index 8b8a22586d..dc59867163 100644 --- a/fluree-db-query/src/eval/rdf.rs +++ b/fluree-db-query/src/eval/rdf.rs @@ -529,7 +529,7 @@ fn encoded_term<'c, R: RowAccess>( /// The base edge's object as a binding, in the encoded form a scan would /// bind it when the kind allows, else materialized with its datatype or /// language tag. -fn term_object_binding( +pub(crate) fn term_object_binding( key: &fluree_db_core::triple_term::TermKey, t: i64, ctx: &ExecutionContext<'_>, @@ -607,7 +607,7 @@ pub(crate) fn term_component_binding( /// A materialized term's object with its datatype or language tag, which /// `OBJECT()` must keep: `"chat"@fr` is not `"chat"`, `"5"^^xsd:int` is not /// `5`. -fn materialized_term_object(term: &fluree_db_core::TripleTermValue) -> Binding { +pub(crate) fn materialized_term_object(term: &fluree_db_core::TripleTermValue) -> Binding { match &term.o { fluree_db_core::FlakeValue::Ref(sid) => Binding::sid(sid.clone()), other => Binding::Lit { diff --git a/fluree-db-query/src/execute/operator_tree.rs b/fluree-db-query/src/execute/operator_tree.rs index bcf718e7c1..b9df4ad4dd 100644 --- a/fluree-db-query/src/execute/operator_tree.rs +++ b/fluree-db-query/src/execute/operator_tree.rs @@ -2415,9 +2415,8 @@ fn result_is_multiplicity_blind(query: &Query) -> bool { } } -/// The query with the `BIND(SUBJECT|PREDICATE|OBJECT(?term) AS ?v)` patterns -/// nothing reads removed (see `elide_unread_term_binds`), or `None` when it -/// has none to drop. +/// The query with the reified-edge components nothing reads removed (see +/// `elide_unread_term_binds`), or `None` when it has none to drop. fn elide_unread_term_binds_in_query(query: &Query) -> Option { let deps = compute_variable_deps(query); let required = deps.as_ref().map(|d| d.required_where_vars.as_slice()); diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index ed4e9cbb47..f6597fb35e 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -104,6 +104,7 @@ pub(crate) fn pattern_tree_has_edge_annotation(patterns: &[Pattern]) -> bool { Pattern::DefaultGraphSource { patterns } => pattern_tree_has_edge_annotation(patterns), Pattern::Triple(_) | Pattern::PropertyPath(_) + | Pattern::TermComponents(_) | Pattern::ShortestPath(_) | Pattern::Filter(_) | Pattern::Bind { .. } @@ -780,6 +781,12 @@ pub fn collect_var_stats( vars.insert(v); } } + Pattern::TermComponents(tc) => { + for v in tc.referenced_vars() { + bump_count(counts, v); + vars.insert(v); + } + } Pattern::ShortestPath(sp) => { for v in sp.referenced_vars() { bump_count(counts, v); @@ -2319,11 +2326,13 @@ fn values_cell_as_ref_term(binding: &crate::binding::Binding) -> Option { } } -/// Drop `BIND(SUBJECT|PREDICATE|OBJECT(?term) AS ?v)` when nothing reads -/// `?v`: no other pattern, the seed, the post-WHERE pipeline or the -/// projection. A variable the seed binds keeps its BIND, which is then the -/// equality check on that position. The `f:reifies*` twin of this rule is -/// `elide_redundant_chain`. +/// Drop a reified-edge component nobody reads: a `TermComponents` position +/// (or a `BIND(SUBJECT|PREDICATE|OBJECT(?term) AS ?v)`) whose variable no +/// other pattern, the seed, the post-WHERE pipeline or the projection reads. +/// A variable the seed binds keeps its position, which is then the equality +/// check. A `TermComponents` left with no variable and no constant subject +/// to anchor on goes too: its constants are the link scan's filters. The +/// `f:reifies*` twin of this rule is `elide_redundant_chain`. pub(crate) fn elide_unread_term_binds( patterns: &[Pattern], needed_vars: &HashSet, @@ -2339,10 +2348,10 @@ pub(crate) fn elide_unread_term_binds( } if matches!(args.as_slice(), [Expression::Var(_)]) ) } - if !patterns - .iter() - .any(|p| matches!(p, Pattern::Bind { expr, .. } if is_term_accessor(expr))) - { + if !patterns.iter().any(|p| { + matches!(p, Pattern::Bind { expr, .. } if is_term_accessor(expr)) + || matches!(p, Pattern::TermComponents(_)) + }) { return None; } let mut counts: HashMap = HashMap::new(); @@ -2354,14 +2363,35 @@ pub(crate) fn elide_unread_term_binds( && !required_where_vars.is_some_and(|r| r.contains(&var)) && !seed_schema.contains(&var) }; + let mut changed = false; let kept: Vec = patterns .iter() - .filter( - |p| !matches!(p, Pattern::Bind { var, expr } if is_term_accessor(expr) && unread(*var)), - ) - .cloned() + .filter_map(|p| match p { + Pattern::Bind { var, expr } if is_term_accessor(expr) && unread(*var) => { + changed = true; + None + } + Pattern::TermComponents(tc) => { + let mut tc = tc.clone(); + for c in [&mut tc.subject, &mut tc.predicate, &mut tc.object] { + if c.var().is_some_and(unread) { + *c = crate::ir::Component::Any; + changed = true; + } + } + // Constant components are also the link scan's filters; the + // pattern stays for a variable to bind or a subject to anchor. + let keep = tc.components().into_iter().any(|c| c.var().is_some()) + || matches!(tc.subject, crate::ir::Component::Node(_)); + if !keep { + changed = true; + } + keep.then_some(Pattern::TermComponents(tc)) + } + other => Some(other.clone()), + }) .collect(); - (kept.len() != patterns.len()).then_some(kept) + changed.then_some(kept) } /// Drop VALUES columns that are UNDEF in every row, and a VALUES left with no @@ -2647,7 +2677,7 @@ pub fn build_where_operators_seeded_with_needed( let seed_vars = seed.as_deref().map(|op| seed_vars(op)).unwrap_or_default(); // A reified-edge position nobody reads costs a dictionary lookup per row - // and removes none; the link lowering binds every variable position. + // and removes none; the link lowering relates every variable position. let term_bind_storage = elide_unread_term_binds( patterns, needed_vars, @@ -3223,6 +3253,16 @@ pub fn build_where_operators_seeded_with_needed( i += 1; } + Pattern::TermComponents(tc) => { + operator = Some(Box::new( + crate::term_components::TermComponentsOperator::new( + get_or_empty_seed(operator.take()), + tc.clone(), + ), + )); + i += 1; + } + Pattern::PropertyPath(pp) => { // Property path - transitive graph traversal // Pass existing operator as child for correlation diff --git a/fluree-db-query/src/explain.rs b/fluree-db-query/src/explain.rs index 5e972080a3..6e6b5f16ce 100644 --- a/fluree-db-query/src/explain.rs +++ b/fluree-db-query/src/explain.rs @@ -578,6 +578,10 @@ pub fn format_general_pattern(pattern: &Pattern) -> String { let var_names: Vec = sq.select.iter().map(|v| format!("?v{}", v.0)).collect(); format!("SUBQUERY SELECT {} {{ ... }}", var_names.join(" ")) } + Pattern::TermComponents(tc) => format!( + "TERM COMPONENTS ?v{} ({:?} {:?} {:?})", + tc.term.0, tc.subject, tc.predicate, tc.object + ), Pattern::PropertyPath(pp) => format!( "PROPERTY PATH {} {:?}", format_ref(&pp.subject), diff --git a/fluree-db-query/src/ir.rs b/fluree-db-query/src/ir.rs index 533d7494fe..528299cd8e 100644 --- a/fluree-db-query/src/ir.rs +++ b/fluree-db-query/src/ir.rs @@ -41,6 +41,7 @@ pub mod pattern; pub mod projection; pub mod query; pub mod reasoning; +pub mod term_components; pub mod triple; pub use adapters::{ @@ -61,4 +62,5 @@ pub use projection::{ }; pub use query::{ConstructTemplate, Query, QueryOutput, Restriction, TemplateReification}; pub use reasoning::{ReasoningConfig, ReasoningModes}; +pub use term_components::{Component, TermComponentsPattern}; pub use triple::{Ref, Term, TriplePattern}; diff --git a/fluree-db-query/src/ir/pattern.rs b/fluree-db-query/src/ir/pattern.rs index 0ddcfc8f34..cf48c8705a 100644 --- a/fluree-db-query/src/ir/pattern.rs +++ b/fluree-db-query/src/ir/pattern.rs @@ -397,6 +397,10 @@ pub enum Pattern { /// Property path pattern (transitive traversal) PropertyPath(PropertyPathPattern), + /// The components of a triple term (internal: the reified-edge lowering + /// emits it beside the link `?r rdf:reifies ?t`). + TermComponents(super::TermComponentsPattern), + /// Anchored shortest-path pattern (Cypher `shortestPath`/`allShortestPaths`). /// /// Both endpoints must be bound by a preceding pattern; binds a path value @@ -683,6 +687,18 @@ impl Pattern { Pattern::PropertyPath(pp) => { debug_assert!(!pp.referenced_vars().contains(&old), "{UNHANDLED}"); } + Pattern::TermComponents(tc) => { + for v in std::iter::once(&mut tc.term).chain( + [&mut tc.subject, &mut tc.predicate, &mut tc.object] + .into_iter() + .filter_map(|c| match c { + super::Component::Var(v) => Some(v), + _ => None, + }), + ) { + rename(v); + } + } Pattern::ShortestPath(sp) => { debug_assert!(!sp.referenced_vars().contains(&old), "{UNHANDLED}"); debug_assert!(sp.path_var != old, "{UNHANDLED}"); @@ -769,6 +785,7 @@ impl Pattern { inner.iter().flat_map(Pattern::referenced_vars).collect() } Pattern::PropertyPath(pp) => pp.referenced_vars(), + Pattern::TermComponents(tc) => tc.referenced_vars(), Pattern::ShortestPath(sp) => sp.referenced_vars(), Pattern::Subquery(sq) => sq.referenced_vars(), Pattern::IndexSearch(isp) => isp.referenced_vars(), @@ -829,6 +846,7 @@ impl Pattern { Pattern::Values { vars, .. } => vars.clone(), Pattern::Minus(_) | Pattern::Exists(_) | Pattern::NotExists(_) => Vec::new(), Pattern::PropertyPath(pp) => pp.produced_vars(), + Pattern::TermComponents(tc) => tc.produced_vars(), Pattern::ShortestPath(sp) => sp.produced_vars(), Pattern::Subquery(sq) => sq.produced_vars(), Pattern::IndexSearch(isp) => isp.produced_vars(), diff --git a/fluree-db-query/src/ir/term_components.rs b/fluree-db-query/src/ir/term_components.rs new file mode 100644 index 0000000000..703f7e6c03 --- /dev/null +++ b/fluree-db-query/src/ir/term_components.rs @@ -0,0 +1,58 @@ +//! The components of a triple term, as a relation the planner can join on. +//! +//! `TermComponents(?t, s, p, o)` holds when `?t` is the triple term +//! `<<( s p o )>>`. The reified-edge lowering emits it beside the link +//! `?r rdf:reifies ?t`, so a component variable shared with another pattern +//! is a join the planner sees, and a bound subject can drive the term +//! dictionary's subject-first reverse tree instead of reading every link. + +use crate::var_registry::VarId; +use fluree_db_core::{DatatypeConstraint, FlakeValue, Sid}; + +/// One component position of a [`TermComponentsPattern`]. +#[derive(Debug, Clone, PartialEq)] +pub enum Component { + /// Unconstrained: nothing reads it. + Any, + /// Bound by the pattern, or joined on when an earlier pattern bound it. + Var(VarId), + /// A constant node: subject, predicate, or IRI object. + Node(Sid), + /// A constant literal object, with its datatype or language tag. + Literal(FlakeValue, DatatypeConstraint), +} + +impl Component { + pub fn var(&self) -> Option { + match self { + Component::Var(v) => Some(*v), + _ => None, + } + } +} + +/// `TermComponents(term, subject, predicate, object)`. +#[derive(Debug, Clone, PartialEq)] +pub struct TermComponentsPattern { + pub term: VarId, + pub subject: Component, + pub predicate: Component, + pub object: Component, +} + +impl TermComponentsPattern { + pub fn components(&self) -> [&Component; 3] { + [&self.subject, &self.predicate, &self.object] + } + + /// The term variable, then each component variable. + pub fn referenced_vars(&self) -> Vec { + std::iter::once(self.term) + .chain(self.components().into_iter().filter_map(Component::var)) + .collect() + } + + pub fn produced_vars(&self) -> Vec { + self.referenced_vars() + } +} diff --git a/fluree-db-query/src/lib.rs b/fluree-db-query/src/lib.rs index 9c652cbbd8..714a2e759b 100644 --- a/fluree-db-query/src/lib.rs +++ b/fluree-db-query/src/lib.rs @@ -105,6 +105,7 @@ pub(crate) mod stats_cache; pub mod stats_query; pub mod subquery; pub mod temporal_mode; +pub mod term_components; pub mod union; pub mod unwind; pub mod values; diff --git a/fluree-db-query/src/planner.rs b/fluree-db-query/src/planner.rs index b0a011178f..36000f1f28 100644 --- a/fluree-db-query/src/planner.rs +++ b/fluree-db-query/src/planner.rs @@ -1160,6 +1160,25 @@ pub fn estimate_pattern( row_count: estimate_branch_cardinality(patterns, stats), }, + // A bound term is one dictionary decode; a bound subject is a reverse- + // tree prefix range. Otherwise the link scan should bind the term + // first, so this ranks after any real scan. + Pattern::TermComponents(tc) => { + let anchored = match &tc.subject { + crate::ir::Component::Var(v) => bound_vars.contains(v), + crate::ir::Component::Node(_) => true, + _ => false, + }; + let row_count = if bound_vars.contains(&tc.term) { + HIGHLY_SELECTIVE + } else if anchored { + MODERATELY_SELECTIVE + } else { + FULL_SCAN + }; + PatternEstimate::Source { row_count } + } + Pattern::PropertyPath(pp) => { // Anchored at a bound endpoint => a bounded closure from a fixed node, // not a full predicate scan. Estimating it as a world scan made reorder diff --git a/fluree-db-query/src/r2rml/rewrite.rs b/fluree-db-query/src/r2rml/rewrite.rs index 60fab6dc94..82afce6a05 100644 --- a/fluree-db-query/src/r2rml/rewrite.rs +++ b/fluree-db-query/src/r2rml/rewrite.rs @@ -128,6 +128,7 @@ fn collect_unsupported_outside_graph_scopes(patterns: &[Pattern], kinds: &mut Ve // The three kinds the rewriter flags, named identically so one // query reads the same whichever guard refuses it. Pattern::PropertyPath(_) => kinds.push("property path"), + Pattern::TermComponents(_) => kinds.push("triple-term components"), Pattern::ShortestPath(_) => kinds.push("shortest path"), Pattern::Subquery(_) => kinds.push("subquery"), // Containers whose bodies evaluate against this view. @@ -343,6 +344,11 @@ pub fn rewrite_patterns_for_r2rml( result_patterns.push(pattern.clone()); origins.push(i); } + Pattern::TermComponents(_) => { + unsupported.push("triple-term components"); + result_patterns.push(pattern.clone()); + origins.push(i); + } Pattern::ShortestPath(_) => { unsupported.push("shortest path"); result_patterns.push(pattern.clone()); diff --git a/fluree-db-query/src/rewrite.rs b/fluree-db-query/src/rewrite.rs index f98715d727..daabcb4d01 100644 --- a/fluree-db-query/src/rewrite.rs +++ b/fluree-db-query/src/rewrite.rs @@ -279,6 +279,7 @@ fn rewrite_single_pattern( | Pattern::Unwind { .. } | Pattern::Values { .. } | Pattern::PropertyPath(_) + | Pattern::TermComponents(_) | Pattern::ShortestPath(_) | Pattern::Subquery(_) | Pattern::IndexSearch(_) diff --git a/fluree-db-query/src/rewrite_owl_ql.rs b/fluree-db-query/src/rewrite_owl_ql.rs index 8a9c384d46..c21abfefe3 100644 --- a/fluree-db-query/src/rewrite_owl_ql.rs +++ b/fluree-db-query/src/rewrite_owl_ql.rs @@ -565,6 +565,7 @@ fn rewrite_owl_ql_single_pattern( | Pattern::Unwind { .. } | Pattern::Values { .. } | Pattern::PropertyPath(_) + | Pattern::TermComponents(_) | Pattern::ShortestPath(_) | Pattern::Subquery(_) | Pattern::IndexSearch(_) diff --git a/fluree-db-query/src/term_components.rs b/fluree-db-query/src/term_components.rs new file mode 100644 index 0000000000..9ca0bcecb4 --- /dev/null +++ b/fluree-db-query/src/term_components.rs @@ -0,0 +1,498 @@ +//! `TermComponents(?t, s, p, o)`: a triple term's components as a relation. +//! +//! Per input row the operator finds the candidate terms, then checks and binds +//! each component: +//! +//! - the term variable is bound: its dictionary key, one forward lookup, which +//! supplies every component at once; +//! - else the subject is bound (a constant, or a value an earlier pattern +//! bound): the reverse tree's `(s_id[, p_id])` prefix range, which yields +//! keys and handles together; distinct prefixes are read once per batch; +//! - else every term of the bound predicate, or of the whole dictionary. +//! +//! A component variable already bound is joined on with the same equality a +//! `BIND` clobber check applies. Dictionary membership says nothing about a +//! live, visible annotation: the link `?r rdf:reifies ?t` that follows keeps +//! the graph, time, policy and novelty checks. + +use crate::binding::{Batch, Binding}; +use crate::context::ExecutionContext; +use crate::error::{QueryError, Result}; +use crate::ir::{Component, TermComponentsPattern}; +use crate::operator::{BoxedOperator, Operator, OperatorState}; +use crate::var_registry::VarId; +use async_trait::async_trait; +use fluree_db_binary_index::BinaryIndexStore; +use fluree_db_core::o_type::OType; +use fluree_db_core::triple_term::TermKey; +use fluree_db_core::value_id::ObjKind; +use fluree_db_core::{DatatypeConstraint, FlakeValue, Sid, TripleTermValue}; +use std::collections::HashMap; +use std::io::ErrorKind; +use std::sync::Arc; + +/// One candidate term: its dictionary key with its handle, or a materialized +/// term a row carried. +enum Candidate { + Encoded { handle: u64, key: TermKey }, + Materialized(Box), +} + +/// The pattern's constant components in the store's encoding. `None` when a +/// dictionary cannot name one, so no interned term can match. +#[derive(Clone, Copy)] +struct EncodedConstants { + s_id: Option, + p_id: Option, + o: Option<(u16, u64)>, +} + +pub struct TermComponentsOperator { + child: BoxedOperator, + pattern: TermComponentsPattern, + schema: Arc<[VarId]>, + child_width: usize, + state: OperatorState, + store: Option>, + constants: Option, + reifies_p_id: u32, +} + +impl TermComponentsOperator { + pub fn new(child: BoxedOperator, pattern: TermComponentsPattern) -> Self { + let mut schema = child.schema().to_vec(); + let child_width = schema.len(); + for v in pattern.produced_vars() { + if !schema.contains(&v) { + schema.push(v); + } + } + Self { + child, + pattern, + schema: Arc::from(schema.into_boxed_slice()), + child_width, + state: OperatorState::Created, + store: None, + constants: None, + reifies_p_id: 0, + } + } + + fn encode_constants( + &self, + store: &BinaryIndexStore, + ctx: &ExecutionContext<'_>, + ) -> std::io::Result> { + let dict_novelty = ctx.dict_novelty.as_ref(); + let s_id = match &self.pattern.subject { + Component::Node(sid) => { + match missing(crate::binary_scan::resolve_subject_v3( + sid, + store, + dict_novelty, + ))? { + Some(id) => Some(id), + None => return Ok(None), + } + } + _ => None, + }; + let p_id = match &self.pattern.predicate { + Component::Node(sid) => match store.sid_to_p_id(sid) { + Some(id) => Some(id), + None => return Ok(None), + }, + _ => None, + }; + let o = match &self.pattern.object { + Component::Node(sid) => { + match missing(crate::binary_scan::resolve_subject_v3( + sid, + store, + dict_novelty, + ))? { + Some(id) => Some((OType::IRI_REF.as_u16(), id)), + None => return Ok(None), + } + } + Component::Literal(value, dtc) => { + let (dt, lang) = match dtc { + DatatypeConstraint::Explicit(dt) => (dt.clone(), None), + DatatypeConstraint::LangTag(tag) => ( + Sid::new( + fluree_vocab::namespaces::RDF, + fluree_vocab::rdf_names::LANG_STRING, + ), + Some(tag.as_ref()), + ), + }; + match missing(crate::binary_scan::value_to_otype_okey( + value, + &dt, + lang, + store, + dict_novelty, + None, + ))? { + Some((ot, key)) => Some((ot.as_u16(), key)), + None => return Ok(None), + } + } + _ => None, + }; + Ok(Some(EncodedConstants { s_id, p_id, o })) + } + + /// The subject's `s_id` for an anchored lookup, from a constant or a + /// bound value; `Err(())` when the bound value names no interned subject. + fn anchor_subject( + &self, + row: &[Binding], + store: &BinaryIndexStore, + ctx: &ExecutionContext<'_>, + ) -> Result>> { + match &self.pattern.subject { + Component::Node(_) => Ok(self.constants.and_then(|c| c.s_id).map(Ok)), + Component::Var(v) => match self.value(row, *v) { + Some(binding) => Ok(Some(subject_id(binding, store, ctx)?.ok_or(()))), + None => Ok(None), + }, + _ => Ok(None), + } + } + + /// The predicate's `p_id` when it is fixed for this row. + fn fixed_predicate(&self, row: &[Binding], store: &BinaryIndexStore) -> Option { + match &self.pattern.predicate { + Component::Node(_) => self.constants.and_then(|c| c.p_id), + Component::Var(v) => match self.value(row, *v)? { + Binding::EncodedPid { p_id } => Some(*p_id), + Binding::Sid { sid, .. } => store.sid_to_p_id(sid), + _ => None, + }, + _ => None, + } + } + + /// The row's value for `var`, if bound. + fn value<'r>(&self, row: &'r [Binding], var: VarId) -> Option<&'r Binding> { + let pos = self.schema.iter().position(|v| *v == var)?; + row.get(pos) + .filter(|b| !matches!(b, Binding::Unbound | Binding::Poisoned)) + } + + /// Check the candidate against the pattern and fill in its variables on + /// `row` (a copy of the input row, widened to the output schema). + fn apply( + &self, + candidate: &Candidate, + t: i64, + row: &mut [Binding], + ctx: &ExecutionContext<'_>, + ) -> Result { + // Constants first: they need no binding built. + if let Candidate::Encoded { key, .. } = candidate { + let Some(c) = self.constants else { + return Ok(false); + }; + if c.s_id.is_some_and(|s| s != key.s_id) + || c.p_id.is_some_and(|p| p != key.p_id) + || c.o + .is_some_and(|(ot, ok)| ot != key.o_type.as_u16() || ok != key.o_key) + { + return Ok(false); + } + } + for (position, component) in self.pattern.components().into_iter().enumerate() { + let computed = match (component, candidate) { + (Component::Any, _) => continue, + (Component::Node(_) | Component::Literal(..), Candidate::Encoded { .. }) => { + continue + } + (Component::Node(_) | Component::Literal(..), Candidate::Materialized(term)) => { + if !materialized_matches(component, position, term) { + return Ok(false); + } + continue; + } + (Component::Var(_), Candidate::Encoded { key, .. }) => match position { + 0 => Binding::encoded_sid(key.s_id), + 1 => Binding::EncodedPid { p_id: key.p_id }, + _ => crate::eval::rdf::term_object_binding(key, t, ctx)?, + }, + (Component::Var(_), Candidate::Materialized(term)) => match position { + 0 => Binding::sid(term.s.clone()), + 1 => Binding::sid(term.p.clone()), + _ => crate::eval::rdf::materialized_term_object(term), + }, + }; + let Component::Var(v) = component else { + continue; + }; + let pos = self + .schema + .iter() + .position(|x| x == v) + .expect("component variable is in the schema"); + match &row[pos] { + Binding::Unbound | Binding::Poisoned => row[pos] = computed, + existing => { + if !crate::object_binding::bind_unifies(existing, &computed, Some(ctx)) { + return Ok(false); + } + } + } + } + // A materialized candidate comes only from a bound term variable. + if let Candidate::Encoded { handle, .. } = candidate { + let term_pos = self + .schema + .iter() + .position(|x| *x == self.pattern.term) + .expect("term variable is in the schema"); + if matches!(row[term_pos], Binding::Unbound | Binding::Poisoned) { + row[term_pos] = Binding::EncodedLit { + o_kind: ObjKind::TRIPLE_TERM.as_u8(), + o_key: *handle, + p_id: self.reifies_p_id, + dt_id: 0, + lang_id: 0, + i_val: i32::MIN, + t, + }; + } + } + Ok(true) + } +} + +/// A component no dictionary can name is a miss, not an error. +fn missing(r: std::io::Result) -> std::io::Result> { + match r { + Ok(v) => Ok(Some(v)), + Err(e) if matches!(e.kind(), ErrorKind::NotFound | ErrorKind::Unsupported) => Ok(None), + Err(e) => Err(e), + } +} + +/// A materialized term's component against a constant. +fn materialized_matches(component: &Component, position: usize, term: &TripleTermValue) -> bool { + match (component, position) { + (Component::Node(sid), 0) => &term.s == sid, + (Component::Node(sid), 1) => &term.p == sid, + (Component::Node(sid), _) => matches!(&term.o, FlakeValue::Ref(o) if o == sid), + (Component::Literal(value, dtc), _) => { + &term.o == value + && match dtc { + DatatypeConstraint::Explicit(dt) => term.lang.is_none() && &term.dt == dt, + DatatypeConstraint::LangTag(tag) => term.lang.as_deref() == Some(tag.as_ref()), + } + } + _ => false, + } +} + +/// The interned `s_id` a subject binding names, or `None` when it names none +/// (a literal, or an IRI no dictionary holds). +fn subject_id( + binding: &Binding, + store: &BinaryIndexStore, + ctx: &ExecutionContext<'_>, +) -> Result> { + let sid = match binding { + Binding::EncodedSid { s_id, .. } => return Ok(Some(*s_id)), + Binding::Sid { sid, .. } => sid.clone(), + Binding::IriMatch { primary_sid, .. } => primary_sid.clone(), + Binding::Iri(iri) => { + return store + .find_subject_id(iri) + .map_err(|e| QueryError::from_io("term components: subject", e)) + } + _ => return Ok(None), + }; + match crate::binary_scan::resolve_subject_v3(&sid, store, ctx.dict_novelty.as_ref()) { + Ok(id) => Ok(Some(id)), + Err(e) if e.kind() == ErrorKind::NotFound => Ok(None), + Err(e) => Err(QueryError::from_io("term components: subject", e)), + } +} + +#[async_trait] +impl Operator for TermComponentsOperator { + fn plan_children(&self) -> Vec> { + vec![crate::plan_node::PlanChild::child(self.child.as_ref())] + } + + fn schema(&self) -> &[VarId] { + &self.schema + } + + fn plan_details(&self) -> serde_json::Map { + let bound = self.child.schema(); + let access = if bound.contains(&self.pattern.term) { + "term" + } else if match &self.pattern.subject { + Component::Node(_) => true, + Component::Var(v) => bound.contains(v), + _ => false, + } { + "subject" + } else { + "scan" + }; + let mut m = serde_json::Map::new(); + m.insert("term".into(), format!("?v{}", self.pattern.term.0).into()); + m.insert("access".into(), access.into()); + m + } + + async fn open(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { + self.child.open(ctx).await?; + self.store = ctx.binary_store.clone(); + if let Some(store) = self.store.clone() { + self.constants = self + .encode_constants(&store, ctx) + .map_err(|e| QueryError::from_io("term components: constants", e))?; + self.reifies_p_id = store + .sid_to_p_id(&Sid::new( + fluree_vocab::namespaces::RDF, + fluree_vocab::rdf_names::REIFIES, + )) + .unwrap_or(0); + } + self.state = OperatorState::Open; + Ok(()) + } + + async fn next_batch(&mut self, ctx: &ExecutionContext<'_>) -> Result> { + if self.state != OperatorState::Open { + return Ok(None); + } + loop { + let input = match self.child.next_batch(ctx).await? { + Some(b) if !b.is_empty() => b, + Some(_) => continue, + None => { + self.state = OperatorState::Exhausted; + return Ok(None); + } + }; + let mut columns: Vec> = vec![Vec::new(); self.schema.len()]; + // Distinct prefixes and predicates are read once per batch. + let mut by_prefix: HashMap<(u64, Option), Arc>> = + HashMap::new(); + let mut by_predicate: HashMap, Arc>> = HashMap::new(); + for row_idx in 0..input.len() { + let mut row: Vec = (0..self.schema.len()) + .map(|col| { + if col < self.child_width { + input.get_by_col(row_idx, col).clone() + } else { + Binding::Unbound + } + }) + .collect(); + let (candidates, t): (Vec, i64) = + match self.value(&row, self.pattern.term).cloned() { + Some(Binding::EncodedLit { + o_kind, o_key, t, .. + }) if o_kind == ObjKind::TRIPLE_TERM.as_u8() => { + let Some(store) = &self.store else { + continue; + }; + let key = store + .resolve_term_key(o_key) + .map_err(|e| QueryError::from_io("resolve_term_key", e))? + .ok_or_else(|| { + QueryError::Internal(format!( + "triple-term handle {o_key:#x} has no dictionary entry" + )) + })?; + (vec![Candidate::Encoded { handle: o_key, key }], t) + } + Some(Binding::Lit { + val: FlakeValue::TripleTerm(term), + .. + }) => (vec![Candidate::Materialized(term)], 0), + // Bound to something that is not a term. + Some(_) => continue, + None => { + let Some(store) = self.store.clone() else { + continue; + }; + let Some(terms) = store.term_dict() else { + continue; + }; + let p_id = self.fixed_predicate(&row, &store); + let found = match self.anchor_subject(&row, &store, ctx)? { + Some(Ok(s_id)) => match by_prefix.get(&(s_id, p_id)) { + Some(found) => Arc::clone(found), + None => { + let found = Arc::new( + terms.terms_with_subject(s_id, p_id).map_err(|e| { + QueryError::from_io("term components: subject", e) + })?, + ); + by_prefix.insert((s_id, p_id), Arc::clone(&found)); + found + } + }, + Some(Err(())) => continue, + None => match by_predicate.get(&p_id) { + Some(found) => Arc::clone(found), + None => { + let predicates: Vec = match p_id { + Some(p) => vec![p], + None => terms.predicates().collect(), + }; + let mut all = Vec::new(); + for p in predicates { + all.extend(terms.terms_of_predicate(p).map_err( + |e| QueryError::from_io("term components: scan", e), + )?); + } + let found = Arc::new(all); + by_predicate.insert(p_id, Arc::clone(&found)); + found + } + }, + }; + ( + found + .iter() + .map(|(key, handle)| Candidate::Encoded { + handle: *handle, + key: *key, + }) + .collect(), + 0, + ) + } + }; + for candidate in &candidates { + let mut out = row.clone(); + if self.apply(candidate, t, &mut out, ctx)? { + for (col, value) in out.into_iter().enumerate() { + columns[col].push(value); + } + } + } + row.clear(); + } + if columns.first().is_some_and(Vec::is_empty) { + continue; + } + return Ok(Some(Batch::new(Arc::clone(&self.schema), columns)?)); + } + } + + fn close(&mut self) { + self.child.close(); + self.state = OperatorState::Closed; + } + + fn estimated_rows(&self) -> Option { + None + } +} diff --git a/fluree-db-sparql/src/lower/annotation.rs b/fluree-db-sparql/src/lower/annotation.rs index aa95773480..f3855c8806 100644 --- a/fluree-db-sparql/src/lower/annotation.rs +++ b/fluree-db-sparql/src/lower/annotation.rs @@ -280,15 +280,16 @@ pub(super) fn link_terms_enabled() -> bool { impl LoweringContext<'_, E> { /// Lower a reified-triple pattern to the link form: one /// `annotation rdf:reifies ?__term` triple whose object is a triple-term - /// handle, then each component of the reified edge either bound from the - /// term (`BIND(SUBJECT(?__term) AS ?s)`) or constrained against it - /// (`FILTER(sameTerm(PREDICATE(?__term),

))`). A variable is always a - /// `BIND`: when the variable is already bound, `BIND` keeps the row only - /// if the values agree, which is the join, and that holds in every scope - /// (a UNION branch, an OPTIONAL, a subquery seed) without the lowering - /// tracking who binds what. A predicate constraint is what the planner - /// turns into one handle interval; a fully constant edge composes to a - /// constant term the scan looks up directly. + /// handle, and `TermComponents(?__term, s, p, o)` relating the term to its + /// components. A variable component is bound by that relation, which joins + /// on it when another pattern bound it first, in every scope, without the + /// lowering tracking who binds what; a bound subject lets the planner + /// drive the relation through the dictionary's subject prefix. A constant + /// component is also a filter on the link + /// (`FILTER(sameTerm(PREDICATE(?__term),

))`), which the planner turns + /// into the scan's handle interval and key check when the link leads. A + /// fully constant edge composes to a constant term the scan looks up + /// directly. pub(super) fn lower_reified_link( &mut self, annotation_ref: Ref, @@ -297,7 +298,7 @@ impl LoweringContext<'_, E> { ) { use fluree_db_core::FlakeValue; use fluree_db_query::binding::Binding; - use fluree_db_query::ir::{Expression, Function}; + use fluree_db_query::ir::{Component, Expression, Function, TermComponentsPattern}; let reifies = self.encoder.encode_ref(fluree_vocab::rdf::REIFIES); @@ -330,54 +331,71 @@ impl LoweringContext<'_, E> { )) }; let IrTriplePattern { s, p, o, dtc } = edge; + // Constant positions are filters on the link (the scan narrows on + // them) and constants of the components relation (a constant subject + // anchors it). Variable positions are bound by the components relation, + // which joins on them when another pattern bound them first. let component = |func: Function, term: IrTerm, out: &mut Vec| match term { - IrTerm::Var(v) => out.push(Pattern::Bind { - var: v, - expr: accessor(func), - }), - IrTerm::Sid(sid) => out.push(same_term(func, Expression::Const(FlakeValue::Ref(sid)))), + IrTerm::Var(v) => Component::Var(v), + IrTerm::Sid(sid) => { + out.push(same_term( + func, + Expression::Const(FlakeValue::Ref(sid.clone())), + )); + Component::Node(sid) + } IrTerm::Iri(iri) => match self.encoder.encode_iri(&iri) { - Some(sid) => out.push(same_term(func, Expression::Const(FlakeValue::Ref(sid)))), + Some(sid) => { + out.push(same_term( + func, + Expression::Const(FlakeValue::Ref(sid.clone())), + )); + Component::Node(sid) + } // An IRI in no registered namespace names nothing in this // ledger, so the pattern cannot match. - None => out.push(Pattern::Filter(Expression::Const(FlakeValue::Boolean( - false, - )))), + None => { + out.push(Pattern::Filter(Expression::Const(FlakeValue::Boolean( + false, + )))); + Component::Any + } }, // A literal matches by term identity: its datatype or language // tag is part of what `<< ?s :p "chat"@fr >>` asks for. IrTerm::Value(v) => match &dtc { - Some(dtc) => out.push(same_term( - func, - Expression::Resolved(Box::new(Binding::Lit { - val: v, - dtc: dtc.clone(), - t: None, - op: None, - p_id: None, - })), - )), - None => out.push(Pattern::Filter(Expression::call( - Function::Eq, - vec![accessor(func), Expression::Const(v)], - ))), + Some(dtc) => { + out.push(same_term( + func, + Expression::Resolved(Box::new(Binding::Lit { + val: v.clone(), + dtc: dtc.clone(), + t: None, + op: None, + p_id: None, + })), + )); + Component::Literal(v, dtc.clone()) + } + None => { + out.push(Pattern::Filter(Expression::call( + Function::Eq, + vec![accessor(func), Expression::Const(v)], + ))); + Component::Any + } }, }; - // Constraints first, so the planner's filter lookahead over the link - // triple reaches every one of them before a BIND intervenes. - let mut binds = Vec::new(); - for (func, term) in [ - (Function::TripleSubject, IrTerm::from(s)), - (Function::TriplePredicate, IrTerm::from(p)), - (Function::TripleObject, o), - ] { - match term { - IrTerm::Var(_) => binds.push((func, term)), - constant => component(func, constant, out), - } - } - for (func, term) in binds { - component(func, term, out); + let tc = TermComponentsPattern { + term: t, + subject: component(Function::TripleSubject, IrTerm::from(s), out), + predicate: component(Function::TriplePredicate, IrTerm::from(p), out), + object: component(Function::TripleObject, o, out), + }; + if tc.components().into_iter().any(|c| c.var().is_some()) + || matches!(tc.subject, Component::Node(_)) + { + out.push(Pattern::TermComponents(tc)); } } } diff --git a/fluree-db-sparql/src/lower/construct.rs b/fluree-db-sparql/src/lower/construct.rs index 8bc3c13375..bcb6d035db 100644 --- a/fluree-db-sparql/src/lower/construct.rs +++ b/fluree-db-sparql/src/lower/construct.rs @@ -220,6 +220,7 @@ impl LoweringContext<'_, E> { | Pattern::Unwind { .. } | Pattern::Values { .. } | Pattern::PropertyPath(_) + | Pattern::TermComponents(_) | Pattern::ShortestPath(_) | Pattern::Subquery(_) | Pattern::IndexSearch(_) diff --git a/fluree-db-sparql/src/lower/mod.rs b/fluree-db-sparql/src/lower/mod.rs index c322e32cff..da69ac3e1e 100644 --- a/fluree-db-sparql/src/lower/mod.rs +++ b/fluree-db-sparql/src/lower/mod.rs @@ -316,6 +316,13 @@ fn reject_direct_reifies_in_patterns(patterns: &[Pattern]) -> Result<()> { return Err(reject_predicate_string(format!("{reserved}"))); } } + Pattern::TermComponents(tc) => { + if let fluree_db_query::ir::Component::Node(p) = &tc.predicate { + if fluree_db_core::is_reserved_reifies_predicate(p) { + return Err(reject_predicate_string(format!("{p}"))); + } + } + } // ShortestPath also carries a predicate Sid; apply the same // firewall (no SPARQL surface produces it today, but stay safe). // The wildcard form (`predicate: None`) excludes `f:reifies*` From da50cf0cf9529eae9ef73f1de3dddeeb5188437c Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 10:27:11 -0400 Subject: [PATCH 28/92] feat(novelty): derive reification links for commits the index has not covered Links existed only in the index, so a link query missed every annotation added, retracted or re-pointed since the last build. Novelty now derives them from each commit's f:reifies* slot ops, replaying a reifier from the index's attachment plus its earlier novelty ops, the rule the index build applies. The index's attachments come from a base the ledger installs: an empty one when the index never held an annotation, the binary store's otherwise (at load, at each publish, for historical views). Commits applied before the base is known wait and are derived when it arrives; a trim past the base drops it. Terms only novelty names take the raw-flake lane, and TermComponents offers them beside the dictionary's, also on a ledger with no index. Graph management and the lifecycle probe skip derived links: no transaction writes one. --- fluree-db-api/src/ledger_manager.rs | 34 +- fluree-db-api/src/view/fluree_ext.rs | 22 +- fluree-db-api/tests/it_policy_verbs.rs | 63 ++++ fluree-db-api/tests/it_triple_term_links.rs | 286 ++++++++++++++++ fluree-db-core/src/lib.rs | 1 + fluree-db-core/src/link.rs | 357 ++++++++++++++++++++ fluree-db-ledger/src/historical.rs | 1 + fluree-db-ledger/src/lib.rs | 35 +- fluree-db-novelty/src/lib.rs | 44 +++ fluree-db-novelty/src/links.rs | 216 ++++++++++++ fluree-db-query/src/binary_range.rs | 58 ++++ fluree-db-query/src/binary_scan.rs | 13 +- fluree-db-query/src/lib.rs | 2 +- fluree-db-query/src/term_components.rs | 291 +++++++++++----- fluree-db-transact/src/stage.rs | 8 + 15 files changed, 1334 insertions(+), 97 deletions(-) create mode 100644 fluree-db-core/src/link.rs create mode 100644 fluree-db-novelty/src/links.rs diff --git a/fluree-db-api/src/ledger_manager.rs b/fluree-db-api/src/ledger_manager.rs index 649a682efc..f30298cf58 100644 --- a/fluree-db-api/src/ledger_manager.rs +++ b/fluree-db-api/src/ledger_manager.rs @@ -773,6 +773,15 @@ impl LedgerHandle { let sync_ns_us = phase.elapsed().as_micros() as u64; let phase = Instant::now(); let arc_store = Arc::new(store); + let links = install_link_base(&mut state, &arc_store)?; + if !links.is_empty() { + fluree_db_binary_index::dict_novelty_safe::populate_dict_novelty_safe( + Arc::make_mut(&mut state.dict_novelty), + Some(&arc_store), + links.iter(), + ) + .map_err(|e| ApiError::internal(format!("populate_dict_novelty_safe: {e}")))?; + } crate::runtime_dicts::reseed_runtime_small_dicts_from_previous(&mut state, &arc_store); let reseed_us = phase.elapsed().as_micros() as u64; @@ -1230,6 +1239,9 @@ pub(crate) async fn load_and_attach_binary_store( // Sync namespace codes between store and snapshot (bimap validation). crate::ns_helpers::sync_store_and_snapshot_ns(&mut store, Arc::make_mut(&mut state.snapshot))?; + let arc_store = Arc::new(store); + install_link_base(state, &arc_store)?; + // Re-populate DictNovelty from already-loaded novelty flakes, but *only* for // entries not present in the persisted dictionaries (canonical IDs must win). // @@ -1239,13 +1251,12 @@ pub(crate) async fn load_and_attach_binary_store( let dn = Arc::make_mut(&mut state.dict_novelty); fluree_db_binary_index::dict_novelty_safe::populate_dict_novelty_safe( dn, - Some(&store), + Some(&arc_store), novelty.iter_flakes(fluree_db_core::IndexType::Post), ) .map_err(|e| ApiError::internal(format!("populate_dict_novelty_safe: {e}")))?; } - let arc_store = Arc::new(store); crate::runtime_dicts::reseed_runtime_small_dicts(state, &arc_store); let ns_fallback = Some(state.snapshot.shared_namespaces()); let provider = BinaryRangeProvider::new( @@ -1278,6 +1289,25 @@ pub(crate) async fn load_and_attach_binary_store( Ok(Some(arc_store)) } +/// Point novelty's reification links at the index `store` serves, deriving +/// the links of commits applied before it was known. A ledger whose index +/// never held an annotation already has its base. +fn install_link_base( + state: &mut LedgerState, + store: &Arc, +) -> Result> { + if !state.snapshot.has_annotations { + return Ok(Vec::new()); + } + let base = fluree_db_novelty::LinkBase::new( + Arc::new(fluree_db_query::IndexAttachments::new(Arc::clone(store))), + state.snapshot.t, + ); + Arc::make_mut(&mut state.novelty) + .set_attachment_base(base) + .map_err(|e| ApiError::internal(format!("derive reification links: {e}"))) +} + // ============================================================================ // LedgerManager - Connection-level cache // ============================================================================ diff --git a/fluree-db-api/src/view/fluree_ext.rs b/fluree-db-api/src/view/fluree_ext.rs index de0a1fe3db..decd69635b 100644 --- a/fluree-db-api/src/view/fluree_ext.rs +++ b/fluree-db-api/src/view/fluree_ext.rs @@ -425,6 +425,26 @@ impl Fluree { .augment_namespace_codes(view.snapshot.namespaces()) .map_err(|e| ApiError::internal(format!("augment namespace codes: {e}")))?; store.set_ns_split_mode(view.snapshot.ns_split_mode()); + let store = Arc::new(store); + if view.snapshot.has_annotations { + if let Some(novelty) = view.novelty.as_mut() { + let base = fluree_db_novelty::LinkBase::new( + Arc::new(fluree_db_query::IndexAttachments::new(Arc::clone( + &store, + ))), + view.snapshot.t, + ); + Arc::make_mut(novelty) + .set_attachment_base(base) + .map_err(|e| { + ApiError::internal(format!("derive reification links: {e}")) + })?; + // The view reads novelty through `overlay`, which + // still holds the copy without links. + view.overlay = + Arc::clone(novelty) as Arc; + } + } // Populate dict novelty safely (persisted dict wins). populate_dict_novelty_from_view( @@ -433,7 +453,7 @@ impl Fluree { view.novelty.as_ref(), )?; view.dict_novelty = Some(Arc::new(dict_novelty)); - view.binary_store = Some(Arc::new(store)); + view.binary_store = Some(store); // Historical views loaded from an index root are metadata-only by default // (`LedgerSnapshot::from_root_bytes` sets `range_provider = None`). diff --git a/fluree-db-api/tests/it_policy_verbs.rs b/fluree-db-api/tests/it_policy_verbs.rs index a2e9bca426..6b8a131712 100644 --- a/fluree-db-api/tests/it_policy_verbs.rs +++ b/fluree-db-api/tests/it_policy_verbs.rs @@ -862,3 +862,66 @@ async fn scaffolder_write_profile_grants_class_ownership_only() { "Lead write profile must NOT edit a Person" ); } + +/// Removing a reifier outright is a delete although its derived +/// `rdf:reifies` link outlives the transaction: no transaction writes a link, +/// so it must not count toward the subject persisting. +#[tokio::test] +async fn removing_a_reifier_is_a_delete_despite_its_link() { + assert_index_defaults(); + let fluree = FlureeBuilder::memory().build_memory(); + let ledger0 = genesis_ledger(&fluree, "verbs_delete_reifier"); + let ledger = fluree + .insert( + ledger0, + &json!({ + "@context": {"ex": "http://example.org/ns/"}, + "@id": "ex:alice", + "@type": "ex:Person", + "ex:knows": { + "@id": "ex:bob", + "@annotation": { + "@id": "ex:claim1", + "@type": "ex:Claim", + "ex:source": {"@id": "ex:hr"} + } + } + }), + ) + .await + .expect("seed") + .ledger; + + let policies = json!([ + view_all(), + { + "@id": "ex:personEditors", + "f:onClass": [{"@id": "http://example.org/ns/Person"}], + "f:action": "f:update", + "f:allow": true + }, + { + "@id": "ex:claimReapers", + "f:onClass": [{"@id": "http://example.org/ns/Claim"}], + "f:action": "f:delete", + "f:allow": true + } + ]); + let ctx = policy_ctx(&ledger, policies).await; + + // The base edge's retract cascades the attachment; the body goes + // explicitly. + let remove = json!({ + "@context": {"ex": "http://example.org/ns/"}, + "where": {"@id": "ex:claim1", "?p": "?o"}, + "delete": [ + {"@id": "ex:claim1", "?p": "?o"}, + {"@id": "ex:alice", "ex:knows": {"@id": "ex:bob"}} + ] + }); + let result = try_txn(&fluree, ledger, TxnType::Update, &remove, &ctx).await; + assert!( + result.is_ok(), + "a delete grant must permit removing the reifier: {result:?}" + ); +} diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 41df65d23c..8ee3fd084d 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -967,3 +967,289 @@ async fn link_lowering_drives_chained_edges_through_their_subjects() { assert_eq!(components["details"]["access"], "subject", "{plan:#}"); } } + +/// What every link query below must answer once the claims are imported and +/// three unindexed transactions have changed them: a new annotated edge, a +/// partial re-point of `ex:claim2`'s object, a second edge out of `ex:bob`, +/// and a retract of `ex:alice ex:knows ex:carol`, whose annotation the +/// retract cascades away. +async fn assert_novelty_links(fluree: &fluree_db_api::Fluree, ledger: &LedgerState) { + let got = links(fluree, ledger).await; + assert_eq!(got.len(), 5, "{got:#?}"); + let term = |reifier: &str| -> String { + got.iter() + .find(|r| r[0].ends_with(reifier)) + .unwrap_or_else(|| panic!("no link for {reifier}: {got:#?}"))[1] + .clone() + }; + assert!( + term("claim2").contains("43") && !term("claim2").contains("42"), + "the link follows an unindexed re-point: {got:#?}" + ); + assert!( + term("claim9").contains("erin") && term("claim9").contains("frank"), + "{got:#?}" + ); + assert!( + !got.iter() + .any(|r| r[1].contains("carol") && r[1].contains("knows")), + "the cascade retracts the link: {got:#?}" + ); + + let got = run_link_query( + fluree, + ledger, + "SELECT ?s ?o WHERE { << ?s ex:knows ?o >> ex:source ?src } ORDER BY ?s ?o".to_string(), + ) + .await; + let pairs: Vec<(String, String)> = got.iter().map(|r| (r[0].clone(), r[1].clone())).collect(); + assert_eq!( + pairs, + [ + ("ex:alice", "ex:bob"), + ("ex:bob", "ex:dave"), + ("ex:bob", "ex:erin"), + ("ex:erin", "ex:frank"), + ] + .map(|(s, o)| (s.to_string(), o.to_string())), + "{got:#?}" + ); + + let got = run_link_query( + fluree, + ledger, + "SELECT ?o WHERE { << ex:erin ex:knows ?o >> ex:source ?src }".to_string(), + ) + .await; + assert_eq!(got, vec![vec!["ex:frank".to_string()]], "{got:#?}"); + + let got = run_link_query( + fluree, + ledger, + "SELECT ?age WHERE { ?r rdf:reifies <<( ex:carol ex:age ?age )>> }".to_string(), + ) + .await; + assert_eq!(got, vec![vec!["43".to_string()]], "{got:#?}"); + + // The second edge's subject is bound by the first: one indexed term and + // one only novelty holds. + let got = run_link_query( + fluree, + ledger, + "SELECT ?x WHERE { << ex:alice ex:knows ?y >> ex:source ?s1 . \ + << ?y ex:knows ?x >> ex:source ?s2 } ORDER BY ?x" + .to_string(), + ) + .await; + assert_eq!( + got, + vec![vec!["ex:dave".to_string()], vec!["ex:erin".to_string()]], + "{got:#?}" + ); +} + +async fn change_claims_without_indexing( + fluree: &fluree_db_api::Fluree, + ledger: LedgerState, +) -> LedgerState { + let turtle = + |body: &str| format!("VERSION \"1.2\"\n@prefix ex: .\n{body}\n"); + let ledger = fluree + .upsert_turtle( + ledger, + &turtle( + "ex:erin ex:knows ex:frank ~ ex:claim9 {| ex:source ex:web |} .\n\ + ex:carol ex:age 43 ~ ex:claim2 .", + ), + ) + .await + .expect("new claim and re-point") + .ledger; + let ledger = fluree + .insert_turtle( + ledger, + &turtle("ex:bob ex:knows ex:erin {| ex:source ex:web |} ."), + ) + .await + .expect("second edge out of bob") + .ledger; + fluree + .update( + ledger, + &json!({ + "@context": { "ex": "http://example.org/" }, + "delete": { "@id": "ex:alice", "ex:knows": { "@id": "ex:carol" } } + }), + ) + .await + .expect("retract an annotated edge") + .ledger +} + +/// Links for commits no index covers come from novelty: new annotations, a +/// partial re-point and a cascaded retract show in link queries before any +/// index build, and again after a reload rebuilds novelty from the commits, +/// where the links wait for the index store before they can be derived. +#[tokio::test] +async fn novelty_carries_links_until_the_index_covers_them() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let alias = "it/triple-term-links:novelty"; + let db_dir = tempfile::tempdir().expect("db tmpdir"); + let data_dir = tempfile::tempdir().expect("data tmpdir"); + std::fs::write(data_dir.path().join("claims.ttl"), CLAIMS).expect("write fixture"); + let open = || { + FlureeBuilder::file(db_dir.path().to_string_lossy().to_string()) + .build() + .expect("build file-backed Fluree") + }; + let fluree = open(); + fluree + .create(alias) + .import(data_dir.path()) + .threads(1) + .memory_budget_mb(256) + .cleanup(false) + .execute() + .await + .expect("import"); + let ledger = fluree.ledger(alias).await.expect("load ledger"); + let ledger = change_claims_without_indexing(&fluree, ledger).await; + assert!( + ledger.t() > ledger.index_t(), + "the changes must stay unindexed" + ); + assert_novelty_links(&fluree, &ledger).await; + drop(ledger); + drop(fluree); + + let fluree = open(); + let ledger = fluree.ledger(alias).await.expect("reload"); + assert!( + ledger.t() > ledger.index_t(), + "the reload must replay novelty" + ); + assert_novelty_links(&fluree, &ledger).await; + + // Time travel to the first change replays novelty on its own. + let view = fluree + .db_at_t(alias, ledger.index_t() + 1) + .await + .expect("historical view"); + let result = fluree + .query( + &view, + "PREFIX rdf: \n\ + SELECT ?r ?t WHERE { ?r rdf:reifies ?t } ORDER BY ?r", + ) + .await + .expect("historical link query"); + let got = rows( + &result + .to_jsonld_async(view.as_graph_db_ref()) + .await + .expect("format"), + ); + assert_eq!(got.len(), 5, "{got:#?}"); + assert!( + got.iter() + .any(|r| r[1].contains("carol") && r[1].contains("knows")), + "the retract comes later: {got:#?}" + ); + assert!( + got.iter() + .any(|r| r[0].ends_with("claim2") && r[1].contains("43")), + "{got:#?}" + ); +} + +/// A ledger that was never indexed holds every link in novelty and has no +/// term dictionary at all. +#[tokio::test] +async fn novelty_carries_links_on_a_ledger_never_indexed() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let fluree = FlureeBuilder::memory().build_memory(); + let ledger = support::genesis_ledger(&fluree, "it/triple-term-links:never-indexed"); + let ledger = fluree + .insert_turtle(ledger, CLAIMS) + .await + .expect("claims") + .ledger; + assert_eq!(ledger.index_t(), 0); + let ledger = change_claims_without_indexing(&fluree, ledger).await; + assert_novelty_links(&fluree, &ledger).await; +} + +/// An index publish in the middle of unindexed re-points: novelty's links +/// must then derive from the new index. Against the old one, the next +/// re-point would retract a term the new index never linked and leave the +/// published link live beside the new one. Every write and read goes through +/// the cached handle, which is what the publish updates in place. +#[tokio::test] +async fn novelty_links_follow_the_index_they_were_published_over() { + use fluree_db_indexer::IndexerConfig; + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + + let fluree = FlureeBuilder::memory() + .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) + .build_memory(); + let ledger_id = "it/triple-term-links:novelty-over-publish"; + let (local, indexer) = + support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let claim = |age: u32| { + format!( + "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:alice ex:age {age} ~ ex:claim1 {{| ex:source ex:hr |}} .\n" + ) + }; + + local + .run_until(async { + fluree.create_ledger(ledger_id).await.expect("create"); + let handle = fluree.ledger_cached(ledger_id).await.expect("cache"); + let upsert = |age: u32| { + let fluree = &fluree; + let handle = &handle; + let turtle = claim(age); + async move { + fluree + .stage(handle) + .upsert_turtle(&turtle) + .execute() + .await + .expect("upsert through the cached handle") + .receipt + .t + } + }; + let live = || async { + let state = handle.snapshot().await.to_ledger_state(); + let got = links(&fluree, &state).await; + (state.t(), state.index_t(), got) + }; + let assert_age = |got: &Vec>, age: u32| { + assert_eq!(got.len(), 1, "one live link: {got:#?}"); + assert!(got[0][1].contains(&age.to_string()), "{got:#?}"); + }; + + let t1 = upsert(41).await; + support::trigger_index_and_wait(&indexer, ledger_id, t1).await; + support::wait_for_index_application(&fluree, ledger_id, t1).await; + + let t2 = upsert(42).await; + let (t, index_t, got) = live().await; + assert_eq!((t, index_t), (t2, t1)); + assert_age(&got, 42); + support::trigger_index_and_wait(&indexer, ledger_id, t2).await; + support::wait_for_index_application(&fluree, ledger_id, t2).await; + + let t3 = upsert(43).await; + let (t, index_t, got) = live().await; + assert_eq!( + (t, index_t), + (t3, t2), + "the publish must reach the cached handle" + ); + assert_age(&got, 43); + }) + .await; +} diff --git a/fluree-db-core/src/lib.rs b/fluree-db-core/src/lib.rs index 5b2bf7d108..24cd8dcd35 100644 --- a/fluree-db-core/src/lib.rs +++ b/fluree-db-core/src/lib.rs @@ -58,6 +58,7 @@ pub mod index_stats; pub mod io_stats; pub mod ledger_config; pub mod ledger_id; +pub mod link; pub mod namespaces; pub mod nonempty; pub mod ns_encoding; diff --git a/fluree-db-core/src/link.rs b/fluree-db-core/src/link.rs new file mode 100644 index 0000000000..679ae39c09 --- /dev/null +++ b/fluree-db-core/src/link.rs @@ -0,0 +1,357 @@ +//! Reification links derived from attachment ops. +//! +//! Commits carry an annotation's attachment as `f:reifies*` slot flakes; the +//! RDF 1.2 link `_:r rdf:reifies <<( s p o )>>` is derived from them. The index +//! build derives it from commit blobs, and novelty derives it for commits the +//! index has not covered yet. Both replay a reifier's slot ops in `t` order, +//! retracts before asserts within a `t`, and emit a link retract and assert at +//! each `t` where the attachment changes from or to a complete triple; this +//! module is the novelty side of that rule. + +use crate::flake::Flake; +use crate::ids::GraphId; +use crate::namespaces::{ + is_reifies_object, is_reifies_predicate, is_reifies_subject, rdf_reifies_sid, + triple_term_datatype_sid, +}; +use crate::sid::Sid; +use crate::value::{FlakeValue, TripleTermValue}; +use std::fmt::Debug; +use std::io; + +/// The three slots that name a reifier's triple, as an index or novelty holds +/// them for one reifier in one graph. +#[derive(Clone, Debug, Default, PartialEq)] +pub struct AttachmentSlots { + pub subject: Option, + pub predicate: Option, + /// The base object with its datatype and language tag. + pub object: Option<(FlakeValue, Sid, Option)>, +} + +/// One slot value of an attachment op. +#[derive(Clone, Debug, PartialEq)] +enum SlotValue { + Subject(Sid), + Predicate(Sid), + Object(FlakeValue, Sid, Option), +} + +impl SlotValue { + /// The slot a flake writes, or `None` for anything but the subject, + /// predicate and object slots (a non-reference subject or predicate value + /// included, which the index build ignores too). + fn of(flake: &Flake) -> Option { + if is_reifies_subject(&flake.p) { + match &flake.o { + FlakeValue::Ref(s) => Some(SlotValue::Subject(s.clone())), + _ => None, + } + } else if is_reifies_predicate(&flake.p) { + match &flake.o { + FlakeValue::Ref(p) => Some(SlotValue::Predicate(p.clone())), + _ => None, + } + } else if is_reifies_object(&flake.p) { + let lang = flake.m.as_ref().and_then(|m| m.lang.clone()); + Some(SlotValue::Object(flake.o.clone(), flake.dt.clone(), lang)) + } else { + None + } + } + + fn rank(&self) -> u8 { + match self { + SlotValue::Subject(_) => 0, + SlotValue::Predicate(_) => 1, + SlotValue::Object(..) => 2, + } + } +} + +/// True for the flakes that write an attachment slot. +#[inline] +pub fn is_attachment_slot(p: &Sid) -> bool { + is_reifies_subject(p) || is_reifies_predicate(p) || is_reifies_object(p) +} + +/// Whether a term with this object can be interned. Objects keyed by a +/// per-(graph, predicate) arena have no graph-independent identity, so the +/// index build gives their attachments no link, and novelty must not either. +#[inline] +pub fn term_object_is_internable(o: &FlakeValue) -> bool { + match o { + FlakeValue::Decimal(_) | FlakeValue::Vector(_) => false, + FlakeValue::BigInt(b) => num_traits::ToPrimitive::to_i64(b.as_ref()).is_some(), + _ => true, + } +} + +impl AttachmentSlots { + /// The triple term a complete attachment names. + pub fn term(&self) -> Option { + let (Some(s), Some(p), Some((o, dt, lang))) = + (&self.subject, &self.predicate, &self.object) + else { + return None; + }; + if !term_object_is_internable(o) { + return None; + } + Some(TripleTermValue { + s: s.clone(), + p: p.clone(), + o: o.clone(), + dt: dt.clone(), + lang: lang.clone(), + }) + } + + /// Fold an index or novelty row into the slots; the last row per slot + /// wins, as in the index build's base lookup. + pub fn observe(&mut self, flake: &Flake) { + if let Some(value) = SlotValue::of(flake) { + self.set(value); + } + } + + fn set(&mut self, value: SlotValue) { + match value { + SlotValue::Subject(s) => self.subject = Some(s), + SlotValue::Predicate(p) => self.predicate = Some(p), + SlotValue::Object(o, dt, lang) => self.object = Some((o, dt, lang)), + } + } + + /// A retract clears its slot only when the slot holds the retracted value. + fn clear(&mut self, value: &SlotValue) { + match value { + SlotValue::Subject(s) if self.subject.as_ref() == Some(s) => self.subject = None, + SlotValue::Predicate(p) if self.predicate.as_ref() == Some(p) => self.predicate = None, + SlotValue::Object(o, dt, lang) + if self + .object + .as_ref() + .is_some_and(|(so, sdt, slang)| so == o && sdt == dt && slang == lang) => + { + self.object = None; + } + _ => {} + } + } +} + +/// The attachments an index holds, read when novelty first meets a reifier. +pub trait AttachmentBase: Send + Sync + Debug { + /// The slots of each reifier in `g_id`, in input order. A reifier the + /// index does not hold has empty slots. + fn attachments(&self, g_id: GraphId, reifiers: &[Sid]) -> io::Result>; +} + +/// Replay one reifier's slot ops from `state` and append its link flakes to +/// `out`. `ops` may arrive in any order; ops with `t <= emit_after` only +/// advance the state (they are history whose links already exist). +pub fn replay_links( + state: &mut AttachmentSlots, + ops: &mut [&Flake], + emit_after: i64, + out: &mut Vec, +) { + let mut keyed: Vec<(i64, bool, u8, SlotValue, &Flake)> = ops + .iter() + .filter_map(|f| SlotValue::of(f).map(|v| (f.t, f.op, v.rank(), v, *f))) + .collect(); + // Within one `t`, retracts apply before asserts so a re-pointed slot + // passes through its old value on the way to the new one. + keyed.sort_by_key(|k| (k.0, k.1, k.2)); + let mut i = 0; + while i < keyed.len() { + let t = keyed[i].0; + let before = state.term(); + let anchor = keyed[i].4; + while i < keyed.len() && keyed[i].0 == t { + let (_, op, _, value, _) = &keyed[i]; + if *op { + state.set(value.clone()); + } else { + state.clear(value); + } + i += 1; + } + if t <= emit_after { + continue; + } + let after = state.term(); + if before != after { + if let Some(old) = before { + out.push(link_flake(anchor, old, t, false)); + } + if let Some(new) = after { + out.push(link_flake(anchor, new, t, true)); + } + } + } +} + +/// The link flake for `term`, in the graph and on the reifier of `slot`. +fn link_flake(slot: &Flake, term: TripleTermValue, t: i64, op: bool) -> Flake { + let o = FlakeValue::TripleTerm(Box::new(term)); + let dt = triple_term_datatype_sid().clone(); + let p = rdf_reifies_sid().clone(); + match &slot.g { + Some(g) => Flake::new_in_graph(g.clone(), slot.s.clone(), p, o, dt, t, op, None), + None => Flake::new(slot.s.clone(), p, o, dt, t, op, None), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::flake::FlakeMeta; + use crate::namespaces::{reifies_object_sid, reifies_predicate_sid, reifies_subject_sid}; + + fn sid(name: &str) -> Sid { + Sid::new(100, name) + } + + fn slot(p: &Sid, o: FlakeValue, t: i64, op: bool) -> Flake { + let dt = if matches!(o, FlakeValue::Ref(_)) { + crate::edge::id_datatype_sid() + } else { + crate::edge::xsd_string_datatype_sid() + }; + Flake::new(sid("r"), p.clone(), o, dt, t, op, None) + } + + fn bundle(s: &str, p: &str, o: &str, t: i64, op: bool) -> Vec { + vec![ + slot(reifies_subject_sid(), FlakeValue::Ref(sid(s)), t, op), + slot(reifies_predicate_sid(), FlakeValue::Ref(sid(p)), t, op), + slot(reifies_object_sid(), FlakeValue::String(o.into()), t, op), + ] + } + + fn links( + state: &mut AttachmentSlots, + flakes: &[Flake], + emit_after: i64, + ) -> Vec<(i64, bool, String)> { + let mut ops: Vec<&Flake> = flakes.iter().collect(); + let mut out = Vec::new(); + replay_links(state, &mut ops, emit_after, &mut out); + out.iter() + .map(|f| match &f.o { + FlakeValue::TripleTerm(t) => { + (f.t, f.op, format!("{} {} {:?}", t.s.name, t.p.name, t.o)) + } + other => panic!("link object {other:?}"), + }) + .collect() + } + + #[test] + fn a_partial_repoint_retracts_the_old_term_and_asserts_the_new() { + let mut state = AttachmentSlots::default(); + let mut flakes = bundle("a", "p", "x", 1, true); + flakes.push(slot( + reifies_object_sid(), + FlakeValue::String("x".into()), + 2, + false, + )); + flakes.push(slot( + reifies_object_sid(), + FlakeValue::String("y".into()), + 2, + true, + )); + assert_eq!( + links(&mut state, &flakes, 0), + vec![ + (1, true, "a p String(\"x\")".to_string()), + (2, false, "a p String(\"x\")".to_string()), + (2, true, "a p String(\"y\")".to_string()), + ] + ); + } + + #[test] + fn history_at_or_before_the_watermark_only_seeds_the_state() { + let mut state = AttachmentSlots::default(); + let mut flakes = bundle("a", "p", "x", 1, true); + flakes.push(slot( + reifies_subject_sid(), + FlakeValue::Ref(sid("a")), + 3, + false, + )); + assert_eq!( + links(&mut state, &flakes, 1), + vec![(3, false, "a p String(\"x\")".to_string())] + ); + } + + #[test] + fn a_base_attachment_is_retracted_by_a_full_retract() { + let mut state = AttachmentSlots { + subject: Some(sid("a")), + predicate: Some(sid("p")), + object: Some(( + FlakeValue::String("x".into()), + crate::edge::xsd_string_datatype_sid(), + None, + )), + }; + let flakes = bundle("a", "p", "x", 5, false); + assert_eq!( + links(&mut state, &flakes, 4), + vec![(5, false, "a p String(\"x\")".to_string())] + ); + assert_eq!(state, AttachmentSlots::default()); + } + + #[test] + fn a_retract_of_a_value_the_slot_does_not_hold_changes_nothing() { + let mut state = AttachmentSlots::default(); + let mut flakes = bundle("a", "p", "x", 1, true); + flakes.push(slot( + reifies_object_sid(), + FlakeValue::String("z".into()), + 2, + false, + )); + assert_eq!( + links(&mut state, &flakes, 0), + vec![(1, true, "a p String(\"x\")".to_string())] + ); + } + + #[test] + fn language_tags_distinguish_object_slots() { + let mut state = AttachmentSlots::default(); + let mut flakes = bundle("a", "p", "x", 1, true); + let mut tagged = slot( + reifies_object_sid(), + FlakeValue::String("x".into()), + 2, + false, + ); + tagged.m = Some(FlakeMeta::with_lang("en")); + flakes.push(tagged); + assert_eq!(links(&mut state, &flakes, 0).len(), 1); + } + + #[test] + fn arena_kind_objects_get_no_term() { + let state = AttachmentSlots { + subject: Some(sid("a")), + predicate: Some(sid("p")), + object: Some(( + FlakeValue::Decimal(Box::new("1.5".parse().unwrap())), + Sid::new(2, "decimal"), + None, + )), + }; + assert!(state.term().is_none()); + } +} diff --git a/fluree-db-ledger/src/historical.rs b/fluree-db-ledger/src/historical.rs index eb42426810..3f3f02cdb9 100644 --- a/fluree-db-ledger/src/historical.rs +++ b/fluree-db-ledger/src/historical.rs @@ -378,6 +378,7 @@ impl HistoricalLedgerView { } novelty.apply_commit(flakes, commit_t, &reverse_graph)?; } + crate::link_base_without_annotations(&mut novelty, snapshot)?; tracing::trace!( commit_count, diff --git a/fluree-db-ledger/src/lib.rs b/fluree-db-ledger/src/lib.rs index 0742e5b442..1efbb8d96d 100644 --- a/fluree-db-ledger/src/lib.rs +++ b/fluree-db-ledger/src/lib.rs @@ -46,6 +46,20 @@ use fluree_db_novelty::{ use futures::StreamExt; use std::sync::Arc; +/// Install an empty attachment base when the index never held an annotation, +/// so novelty derives reification links without waiting for the binary store. +/// An index with annotations gets its base when the store attaches. Returns +/// the links of commits that were waiting on it. +pub(crate) fn link_base_without_annotations( + novelty: &mut Novelty, + snapshot: &LedgerSnapshot, +) -> Result> { + if snapshot.has_annotations { + return Ok(Vec::new()); + } + Ok(novelty.set_attachment_base(fluree_db_novelty::LinkBase::empty(snapshot.t))?) +} + /// Type-erased binary index store for query engine access. /// /// Allows `LedgerState` to carry a `BinaryIndexStore` without @@ -334,7 +348,7 @@ impl LedgerState { // Load novelty from commits since index_t let head_commit_id = match &record.commit_head_id { Some(head_cid) if record.commit_t > snapshot.t => { - let (novelty_overlay, head_id, head_temporal) = Self::load_novelty( + let (mut novelty_overlay, head_id, head_temporal) = Self::load_novelty( store, head_cid, snapshot.t, @@ -343,6 +357,8 @@ impl LedgerState { &mut dict_novelty, ) .await?; + let links = link_base_without_annotations(&mut novelty_overlay, &snapshot)?; + dict_novelty.populate_from_flakes(&links); let head_index_id = record.index_head_id.clone(); let mut runtime_small_dicts = RuntimeSmallDicts::new(); runtime_small_dicts.populate_from_flakes_iter( @@ -369,10 +385,11 @@ impl LedgerState { }; let head_index_id = record.index_head_id.clone(); - let novelty_t = snapshot.t; + let mut novelty = Novelty::new(snapshot.t); + link_base_without_annotations(&mut novelty, &snapshot)?; Ok(Self { snapshot: Arc::new(snapshot), - novelty: Arc::new(Novelty::new(novelty_t)), + novelty: Arc::new(novelty), dict_novelty: Arc::new(dict_novelty), schema_hierarchy_cache: Arc::new(fluree_db_core::SchemaHierarchyCache::default()), shacl_compile_cache: Arc::new(parking_lot::RwLock::new(None)), @@ -529,11 +546,14 @@ impl LedgerState { } /// Create a new ledger state from components - pub fn new(snapshot: LedgerSnapshot, novelty: Novelty) -> Self { - let dict_novelty = DictNovelty::with_watermarks( + pub fn new(snapshot: LedgerSnapshot, mut novelty: Novelty) -> Self { + let links = link_base_without_annotations(&mut novelty, &snapshot) + .expect("an empty attachment base derives links without reading anything"); + let mut dict_novelty = DictNovelty::with_watermarks( snapshot.subject_watermarks.clone(), snapshot.string_watermark, ); + dict_novelty.populate_from_flakes(&links); let mut runtime_small_dicts = RuntimeSmallDicts::new(); runtime_small_dicts .populate_from_flakes_iter(novelty.iter_flakes(fluree_db_core::IndexType::Post)); @@ -680,6 +700,7 @@ impl LedgerState { // Clear novelty up to new index_t let mut new_novelty = (*self.novelty).clone(); new_novelty.clear_up_to(new_snapshot.t); + link_base_without_annotations(&mut new_novelty, &new_snapshot)?; // Reset dict_novelty with new watermarks from the index root let mut new_dict_novelty = DictNovelty::with_watermarks( @@ -756,6 +777,7 @@ impl LedgerState { // Clear novelty up to new index_t let mut new_novelty = (*self.novelty).clone(); new_novelty.clear_up_to(new_snapshot.t); + let links = link_base_without_annotations(&mut new_novelty, &new_snapshot)?; // Note: use `size > 0` not `is_empty()` — after clear_up_to the arena still // holds dead flakes, but `size` tracks only active bytes. let has_remaining_novelty = new_novelty.size > 0; @@ -831,6 +853,9 @@ impl LedgerState { &self.snapshot.subject_watermarks, self.snapshot.string_watermark, ); + if retired.is_some() && !links.is_empty() { + Arc::make_mut(&mut self.dict_novelty).populate_from_flakes(&links); + } let new_runtime_small_dicts = match retired { Some(_) => Arc::clone(&self.runtime_small_dicts), None => { diff --git a/fluree-db-novelty/src/lib.rs b/fluree-db-novelty/src/lib.rs index 8823a05fe2..132b7f91ca 100644 --- a/fluree-db-novelty/src/lib.rs +++ b/fluree-db-novelty/src/lib.rs @@ -39,6 +39,7 @@ mod commit_flakes; pub mod delta; mod error; mod fact_state; +mod links; mod runtime_stats; mod stats; @@ -61,6 +62,7 @@ pub use fluree_db_core::commit::codec::envelope::{MAX_GRAPH_DELTA_ENTRIES, MAX_G pub use fluree_db_core::commit::codec::format::{CommitSignature, ALGO_ED25519}; pub use fluree_db_core::commit::codec::verify_commit_blob; pub use fluree_db_credential::SigningKey; +pub use links::LinkBase; pub use runtime_stats::{ assemble_fast_stats, assemble_fast_stats_with, assemble_full_stats, assemble_full_stats_with, assemble_planner_stats, merge_is_identity, resolve_runtime_predicate_id, stats_merge_site, @@ -569,6 +571,12 @@ pub struct Novelty { /// identity, per graph, within this novelty window). Persistent map, so it /// clones in O(1). The dedup oracle behind the seam; see [`fact_state`]. fact_state: NoveltyFactState, + + /// The index's attachments, which reification links are derived against + /// (see [`links`]); `None` until the ledger installs it. + link_base: Option, + /// Every commit at or before this `t` has its links in novelty. + links_through: i64, } #[inline] @@ -592,6 +600,8 @@ impl Novelty { config_write_t: 0, attachments: AttachmentNovelty::new(), fact_state: NoveltyFactState::new(), + link_base: None, + links_through: t, } } @@ -901,8 +911,21 @@ impl Novelty { } } + // Links read the index, so they are derived before any mutation too. + let links_current = self.current_link_base().is_some(); + if let Some(base) = self.current_link_base() { + let touched = links::touched_reifiers(routed.iter().map(|(f, g)| (*g, f))); + if !touched.is_empty() { + let links = self.derive_links(base, &touched, self.t)?; + routed.extend(links.into_iter().map(|(g_id, f)| (f, g_id))); + } + } + // From here on every step is infallible. self.t = self.t.max(commit_t); + if links_current { + self.links_through = self.t; + } self.epoch += 1; // Bump epoch once per commit self.refresh_content_version(); @@ -1046,6 +1069,24 @@ impl Novelty { } } + let links_current = self.current_link_base().is_some(); + if let Some(base) = self.current_link_base() { + let touched = links::touched_reifiers( + per_graph + .iter() + .flat_map(|(g_id, flakes)| flakes.iter().map(move |f| (*g_id, f))), + ); + if !touched.is_empty() { + let links = self.derive_links(base, &touched, self.t)?; + for (g_id, flake) in links { + per_graph.entry(g_id).or_default().push(flake); + } + } + } + if links_current { + self.links_through = max_t; + } + if per_graph.is_empty() { self.t = max_t; self.epoch += 1; @@ -1182,6 +1223,9 @@ impl Novelty { /// after each index rebuild rather than mutated in-place, so this is rarely /// the hot path. pub fn clear_up_to(&mut self, cutoff_t: i64) { + // Commits at or before the cutoff have their links in the index now. + self.trim_link_base(cutoff_t); + self.links_through = self.links_through.max(cutoff_t.min(self.t)); if self.flake_count == 0 { return; } diff --git a/fluree-db-novelty/src/links.rs b/fluree-db-novelty/src/links.rs new file mode 100644 index 0000000000..367843d3af --- /dev/null +++ b/fluree-db-novelty/src/links.rs @@ -0,0 +1,216 @@ +//! Reification links for commits the index has not covered. +//! +//! The index derives `_:r rdf:reifies <<( s p o )>>` from each commit's +//! `f:reifies*` slot ops; novelty derives the same links for its own commits, +//! so a link query sees an annotation as soon as it commits. Deriving a +//! reifier's link needs its attachment before the commit: the index's slots +//! (an [`AttachmentBase`]) plus the reifier's earlier novelty ops. +//! +//! The base is known once the ledger has attached its index (or knows it has +//! none). Until then commits apply without links and `links_through` stays +//! behind; [`Novelty::set_attachment_base`] derives the missing links. A trim +//! past the base's `t` drops the base, since the trimmed ops then live only in +//! the next index. + +use crate::error::{NoveltyError, Result}; +use crate::{Novelty, Segment}; +use fluree_db_core::flake::FlakeMeta; +use fluree_db_core::link::{is_attachment_slot, replay_links, AttachmentBase, AttachmentSlots}; +use fluree_db_core::{Flake, FlakeValue, GraphId, IndexType, Sid}; +use std::collections::BTreeMap; +use std::sync::Arc; + +/// An index's attachments, and the `t` it covers. +#[derive(Clone, Debug)] +pub struct LinkBase { + base: Arc, + t: i64, +} + +/// The base of a ledger whose index holds no attachments. +#[derive(Debug)] +struct NoAttachments; + +impl AttachmentBase for NoAttachments { + fn attachments( + &self, + _g_id: GraphId, + reifiers: &[Sid], + ) -> std::io::Result> { + Ok(vec![AttachmentSlots::default(); reifiers.len()]) + } +} + +impl LinkBase { + /// The attachments `base` holds, as of index `t`. + pub fn new(base: Arc, t: i64) -> Self { + Self { base, t } + } + + /// A base with no attachments, as of index `t`: no index, or one that + /// never held an annotation. + pub fn empty(t: i64) -> Self { + Self::new(Arc::new(NoAttachments), t) + } +} + +/// Reifiers with attachment ops, keyed `(g_id, reifier)`, each with the ops a +/// batch brings that novelty does not hold yet. +pub(crate) type Touched<'a> = BTreeMap<(GraphId, Sid), Vec<&'a Flake>>; + +/// Group a batch's attachment ops by reifier. +pub(crate) fn touched_reifiers<'a>( + batch: impl IntoIterator, +) -> Touched<'a> { + let mut touched = Touched::new(); + for (g_id, flake) in batch { + if is_attachment_slot(&flake.p) { + touched + .entry((g_id, flake.s.clone())) + .or_default() + .push(flake); + } + } + touched +} + +impl Novelty { + /// Install the index's attachments and derive the links of every commit + /// applied without them. Returns the derived links, which the caller adds + /// to its dictionary novelty. + pub fn set_attachment_base(&mut self, base: LinkBase) -> Result> { + let derived = if self.links_through < self.t { + let mut touched = Touched::new(); + for g_id in 0..self.graphs.len() as GraphId { + for flake in self.graph_flakes(g_id) { + if flake.t > self.links_through && is_attachment_slot(&flake.p) { + touched.entry((g_id, flake.s.clone())).or_default(); + } + } + } + self.derive_links(&base, &touched, self.links_through)? + } else { + Vec::new() + }; + self.link_base = Some(base); + self.links_through = self.t; + if derived.is_empty() { + return Ok(Vec::new()); + } + let mut per_graph: BTreeMap> = BTreeMap::new(); + for (g_id, flake) in &derived { + per_graph.entry(*g_id).or_default().push(flake.clone()); + } + for (g_id, batch) in per_graph { + self.check_segment_capacity(g_id)?; + for flake in &batch { + self.fact_state.record(g_id, flake); + } + self.push_segment(g_id, Segment::build(batch, false)); + } + self.epoch += 1; + self.refresh_content_version(); + Ok(derived.into_iter().map(|(_, f)| f).collect()) + } + + /// The base links are derived from, when every earlier commit has its + /// links; `None` leaves the next commit's links pending. + pub(crate) fn current_link_base(&self) -> Option<&LinkBase> { + self.link_base + .as_ref() + .filter(|_| self.links_through == self.t) + } + + /// Links for each touched reifier: its index slots, replayed through every + /// op novelty holds for it and then its new ops. Only ops after + /// `emit_after` produce links. + pub(crate) fn derive_links( + &self, + base: &LinkBase, + touched: &Touched<'_>, + emit_after: i64, + ) -> Result> { + let mut out: Vec<(GraphId, Flake)> = Vec::new(); + let mut links: Vec = Vec::new(); + let mut graphs: BTreeMap> = BTreeMap::new(); + for (g_id, reifier) in touched.keys() { + graphs.entry(*g_id).or_default().push(reifier); + } + for (g_id, reifiers) in graphs { + let owned: Vec = reifiers.iter().map(|s| (*s).clone()).collect(); + let slots = base + .base + .attachments(g_id, &owned) + .map_err(|e| NoveltyError::Storage(format!("index attachments: {e}")))?; + if slots.len() != owned.len() { + return Err(NoveltyError::Storage(format!( + "index attachments: {} answers for {} reifiers", + slots.len(), + owned.len() + ))); + } + for (reifier, mut state) in owned.into_iter().zip(slots) { + let mut ops = self.reifier_slot_history(g_id, &reifier); + ops.extend(touched[&(g_id, reifier)].iter().copied()); + replay_links(&mut state, &mut ops, emit_after, &mut links); + out.extend(links.drain(..).map(|f| (g_id, f))); + } + } + Ok(out) + } + + /// Every slot op novelty holds for `reifier` in `g_id`. + fn reifier_slot_history(&self, g_id: GraphId, reifier: &Sid) -> Vec<&Flake> { + let first = Flake::new( + reifier.clone(), + Sid::min(), + FlakeValue::min(), + Sid::min(), + i64::MIN, + false, + None, + ); + let rhs = Flake::new( + reifier.clone(), + Sid::max(), + FlakeValue::max(), + Sid::max(), + i64::MAX, + true, + Some(FlakeMeta::max()), + ); + let mut ops = Vec::new(); + if let Some(Some(segs)) = self.graphs.get(g_id as usize) { + for seg in segs { + if !seg.may_overlap(IndexType::Spot, Some(&first), Some(&rhs), false) { + continue; + } + for &local in seg.range(IndexType::Spot, Some(&first), Some(&rhs), false) { + let flake = &seg.flakes[local as usize]; + if flake.s == *reifier && is_attachment_slot(&flake.p) { + ops.push(flake); + } + } + } + } + ops + } + + /// Every flake of `g_id`, segment by segment. + fn graph_flakes(&self, g_id: GraphId) -> impl Iterator { + self.graphs + .get(g_id as usize) + .and_then(Option::as_ref) + .into_iter() + .flatten() + .flat_map(|seg| seg.flakes.iter()) + } + + /// Drop the base once a trim passes it: the trimmed ops now live only in + /// a newer index, which the ledger installs next. + pub(crate) fn trim_link_base(&mut self, cutoff_t: i64) { + if self.link_base.as_ref().is_some_and(|b| b.t < cutoff_t) { + self.link_base = None; + } + } +} diff --git a/fluree-db-query/src/binary_range.rs b/fluree-db-query/src/binary_range.rs index f064ce9be4..e9b2670199 100644 --- a/fluree-db-query/src/binary_range.rs +++ b/fluree-db-query/src/binary_range.rs @@ -393,6 +393,64 @@ impl RangeProvider for BinaryRangeProvider { } } +/// The attachments an index holds, read without novelty: what novelty +/// derives its reification links against. Holds only the store, so it never +/// pins the ledger's dictionaries. +pub struct IndexAttachments { + store: Arc, + dict_novelty: Arc, + runtime_small_dicts: Arc, +} + +impl IndexAttachments { + pub fn new(store: Arc) -> Self { + Self { + store, + dict_novelty: Arc::new(DictNovelty::new_uninitialized()), + runtime_small_dicts: Arc::new(RuntimeSmallDicts::new()), + } + } +} + +impl std::fmt::Debug for IndexAttachments { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("IndexAttachments") + .field("max_t", &self.store.max_t()) + .finish() + } +} + +impl fluree_db_core::link::AttachmentBase for IndexAttachments { + fn attachments( + &self, + g_id: GraphId, + reifiers: &[Sid], + ) -> std::io::Result> { + let opts = RangeOptions::new(); + reifiers + .iter() + .map(|reifier| { + let rows = binary_range_eq_v3( + &self.store, + &self.dict_novelty, + &self.runtime_small_dicts, + g_id, + IndexType::Spot, + &RangeMatch::new().with_subject(reifier.clone()), + &opts, + &fluree_db_core::NoOverlay, + None, + )?; + let mut slots = fluree_db_core::link::AttachmentSlots::default(); + for row in &rows { + slots.observe(row); + } + Ok(slots) + }) + .collect() + } +} + /// The slice of `raw` — sorted in `index` order — that can hold a flake /// matching `match_val`: the span between [`overlay_eq_bounds`]' sentinels, /// or all of `raw` when the match binds no prefix of the order. diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index 04bb576668..2f70f439cf 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -3736,6 +3736,8 @@ pub(crate) fn translate_one_flake_v3_pub( } // Object value → (o_type, o_key), using flake.dt + lang for proper OType. + // A term novelty derived has no handle until the index interns it; it is + // an ordinary novelty-only value, so it takes the raw-flake lane. let (o_type, o_key) = value_to_otype_okey( &flake.o, &flake.dt, @@ -3743,7 +3745,16 @@ pub(crate) fn translate_one_flake_v3_pub( store, dict_novelty, Some((g_id, p_id)), - )?; + ) + .map_err(|e| match (&flake.o, e.kind()) { + (fluree_db_core::FlakeValue::TripleTerm(_), std::io::ErrorKind::NotFound) => { + std::io::Error::new( + std::io::ErrorKind::Unsupported, + "triple term not interned (novelty-only); use raw flake path", + ) + } + _ => e, + })?; // List index let o_i = flake diff --git a/fluree-db-query/src/lib.rs b/fluree-db-query/src/lib.rs index 714a2e759b..8db0d62eb4 100644 --- a/fluree-db-query/src/lib.rs +++ b/fluree-db-query/src/lib.rs @@ -115,7 +115,7 @@ pub mod vector; // Re-exports pub use aggregate::AggregateOperator; pub use binary_history::BinaryHistoryScanOperator; -pub use binary_range::BinaryRangeProvider; +pub use binary_range::{BinaryRangeProvider, IndexAttachments}; pub use binary_scan::BinaryScanOperator; pub use bind::BindOperator; pub use binding::{ diff --git a/fluree-db-query/src/term_components.rs b/fluree-db-query/src/term_components.rs index 9ca0bcecb4..e7e6ba7bd1 100644 --- a/fluree-db-query/src/term_components.rs +++ b/fluree-db-query/src/term_components.rs @@ -14,6 +14,9 @@ //! `BIND` clobber check applies. Dictionary membership says nothing about a //! live, visible annotation: the link `?r rdf:reifies ?t` that follows keeps //! the graph, time, policy and novelty checks. +//! +//! Terms the index has not interned yet appear only in novelty's links; they +//! are collected once per open and offered alongside the dictionary's. use crate::binding::{Batch, Binding}; use crate::context::ExecutionContext; @@ -27,7 +30,7 @@ use fluree_db_core::o_type::OType; use fluree_db_core::triple_term::TermKey; use fluree_db_core::value_id::ObjKind; use fluree_db_core::{DatatypeConstraint, FlakeValue, Sid, TripleTermValue}; -use std::collections::HashMap; +use std::collections::{HashMap, HashSet}; use std::io::ErrorKind; use std::sync::Arc; @@ -47,6 +50,13 @@ struct EncodedConstants { o: Option<(u16, u64)>, } +/// A term only novelty's links name, with its subject's id when a dictionary +/// holds one. +struct NoveltyTerm { + term: Box, + s_id: Option, +} + pub struct TermComponentsOperator { child: BoxedOperator, pattern: TermComponentsPattern, @@ -56,6 +66,7 @@ pub struct TermComponentsOperator { store: Option>, constants: Option, reifies_p_id: u32, + novelty_terms: Vec, } impl TermComponentsOperator { @@ -76,6 +87,85 @@ impl TermComponentsOperator { store: None, constants: None, reifies_p_id: 0, + novelty_terms: Vec::new(), + } + } + + /// The terms novelty's links assert that the dictionary does not hold. + fn collect_novelty_terms(&self, ctx: &ExecutionContext<'_>) -> Result> { + let reifies = fluree_db_core::rdf_reifies_sid(); + let (first, rhs) = crate::fast_path_common::predicate_walk_bounds(reifies); + let mut seen: HashSet = HashSet::new(); + let mut visit = |overlay: &dyn fluree_db_core::OverlayProvider, g_id, to_t| { + overlay.for_each_overlay_flake( + g_id, + fluree_db_core::IndexType::Post, + Some(&first), + Some(&rhs), + false, + to_t, + &mut |f| { + if f.op && f.p == *reifies { + if let FlakeValue::TripleTerm(term) = &f.o { + seen.insert((**term).clone()); + } + } + }, + ); + }; + match ctx.active_graphs() { + crate::dataset::ActiveGraphs::Single => { + visit(ctx.overlay(), ctx.binary_g_id, ctx.to_t); + } + crate::dataset::ActiveGraphs::Many(graphs) => { + for g in graphs { + visit(g.overlay, g.g_id, g.to_t); + } + } + } + let dict_novelty = ctx.dict_novelty.as_ref(); + let mut out = Vec::new(); + for term in seen { + let s_id = match &self.store { + Some(store) => { + match missing(crate::binary_scan::compose_term_handle( + &term, + store, + dict_novelty, + )) + .map_err(|e| QueryError::from_io("term components: novelty", e))? + { + // The dictionary offers it already. + Some(_) => continue, + None => missing(crate::binary_scan::resolve_subject_v3( + &term.s, + store, + dict_novelty, + )) + .map_err(|e| QueryError::from_io("term components: novelty", e))?, + } + } + None => None, + }; + out.push(NoveltyTerm { + term: Box::new(term), + s_id, + }); + } + Ok(out) + } + + /// Whether a novelty term can match the row's subject anchor. A cheap + /// filter only: `apply` still checks every component. + fn novelty_anchor_matches(&self, row: &[Binding], nt: &NoveltyTerm) -> bool { + match &self.pattern.subject { + Component::Node(sid) => nt.term.s == *sid, + Component::Var(v) => match self.value(row, *v) { + Some(Binding::EncodedSid { s_id, .. }) => nt.s_id == Some(*s_id), + Some(Binding::Sid { sid, .. }) => nt.term.s == *sid, + _ => true, + }, + _ => true, } } @@ -244,15 +334,14 @@ impl TermComponentsOperator { } } } - // A materialized candidate comes only from a bound term variable. - if let Candidate::Encoded { handle, .. } = candidate { - let term_pos = self - .schema - .iter() - .position(|x| *x == self.pattern.term) - .expect("term variable is in the schema"); - if matches!(row[term_pos], Binding::Unbound | Binding::Poisoned) { - row[term_pos] = Binding::EncodedLit { + let term_pos = self + .schema + .iter() + .position(|x| *x == self.pattern.term) + .expect("term variable is in the schema"); + if matches!(row[term_pos], Binding::Unbound | Binding::Poisoned) { + row[term_pos] = match candidate { + Candidate::Encoded { handle, .. } => Binding::EncodedLit { o_kind: ObjKind::TRIPLE_TERM.as_u8(), o_key: *handle, p_id: self.reifies_p_id, @@ -260,8 +349,18 @@ impl TermComponentsOperator { lang_id: 0, i_val: i32::MIN, t, - }; - } + }, + // As a scan binds a link object novelty holds. + Candidate::Materialized(term) => Binding::Lit { + val: FlakeValue::TripleTerm(term.clone()), + dtc: DatatypeConstraint::Explicit( + fluree_db_core::triple_term_datatype_sid().clone(), + ), + t: None, + op: None, + p_id: None, + }, + }; } Ok(true) } @@ -361,6 +460,7 @@ impl Operator for TermComponentsOperator { )) .unwrap_or(0); } + self.novelty_terms = self.collect_novelty_terms(ctx)?; self.state = OperatorState::Open; Ok(()) } @@ -393,83 +493,100 @@ impl Operator for TermComponentsOperator { } }) .collect(); - let (candidates, t): (Vec, i64) = - match self.value(&row, self.pattern.term).cloned() { - Some(Binding::EncodedLit { - o_kind, o_key, t, .. - }) if o_kind == ObjKind::TRIPLE_TERM.as_u8() => { - let Some(store) = &self.store else { - continue; - }; - let key = store - .resolve_term_key(o_key) - .map_err(|e| QueryError::from_io("resolve_term_key", e))? - .ok_or_else(|| { - QueryError::Internal(format!( - "triple-term handle {o_key:#x} has no dictionary entry" - )) - })?; - (vec![Candidate::Encoded { handle: o_key, key }], t) - } - Some(Binding::Lit { - val: FlakeValue::TripleTerm(term), - .. - }) => (vec![Candidate::Materialized(term)], 0), - // Bound to something that is not a term. - Some(_) => continue, - None => { - let Some(store) = self.store.clone() else { - continue; - }; - let Some(terms) = store.term_dict() else { - continue; - }; - let p_id = self.fixed_predicate(&row, &store); - let found = match self.anchor_subject(&row, &store, ctx)? { - Some(Ok(s_id)) => match by_prefix.get(&(s_id, p_id)) { - Some(found) => Arc::clone(found), - None => { - let found = Arc::new( - terms.terms_with_subject(s_id, p_id).map_err(|e| { - QueryError::from_io("term components: subject", e) - })?, - ); - by_prefix.insert((s_id, p_id), Arc::clone(&found)); - found - } - }, - Some(Err(())) => continue, - None => match by_predicate.get(&p_id) { - Some(found) => Arc::clone(found), - None => { - let predicates: Vec = match p_id { - Some(p) => vec![p], - None => terms.predicates().collect(), - }; - let mut all = Vec::new(); - for p in predicates { - all.extend(terms.terms_of_predicate(p).map_err( - |e| QueryError::from_io("term components: scan", e), - )?); + let (candidates, t): (Vec, i64) = match self + .value(&row, self.pattern.term) + .cloned() + { + Some(Binding::EncodedLit { + o_kind, o_key, t, .. + }) if o_kind == ObjKind::TRIPLE_TERM.as_u8() => { + let Some(store) = &self.store else { + continue; + }; + let key = store + .resolve_term_key(o_key) + .map_err(|e| QueryError::from_io("resolve_term_key", e))? + .ok_or_else(|| { + QueryError::Internal(format!( + "triple-term handle {o_key:#x} has no dictionary entry" + )) + })?; + (vec![Candidate::Encoded { handle: o_key, key }], t) + } + Some(Binding::Lit { + val: FlakeValue::TripleTerm(term), + .. + }) => (vec![Candidate::Materialized(term)], 0), + // Bound to something that is not a term. + Some(_) => continue, + None => { + let mut candidates = Vec::new(); + if let Some(store) = self.store.clone() { + if let Some(terms) = store.term_dict() { + let p_id = self.fixed_predicate(&row, &store); + let found = match self.anchor_subject(&row, &store, ctx)? { + Some(Ok(s_id)) => match by_prefix.get(&(s_id, p_id)) { + Some(found) => Some(Arc::clone(found)), + None => { + let found = Arc::new( + terms.terms_with_subject(s_id, p_id).map_err( + |e| { + QueryError::from_io( + "term components: subject", + e, + ) + }, + )?, + ); + by_prefix.insert((s_id, p_id), Arc::clone(&found)); + Some(found) + } + }, + // No interned subject: only novelty can match. + Some(Err(())) => None, + None => match by_predicate.get(&p_id) { + Some(found) => Some(Arc::clone(found)), + None => { + let predicates: Vec = match p_id { + Some(p) => vec![p], + None => terms.predicates().collect(), + }; + let mut all = Vec::new(); + for p in predicates { + all.extend(terms.terms_of_predicate(p).map_err( + |e| { + QueryError::from_io( + "term components: scan", + e, + ) + }, + )?); + } + let found = Arc::new(all); + by_predicate.insert(p_id, Arc::clone(&found)); + Some(found) + } + }, + }; + if let Some(found) = found { + candidates.extend(found.iter().map(|(key, handle)| { + Candidate::Encoded { + handle: *handle, + key: *key, } - let found = Arc::new(all); - by_predicate.insert(p_id, Arc::clone(&found)); - found - } - }, - }; - ( - found - .iter() - .map(|(key, handle)| Candidate::Encoded { - handle: *handle, - key: *key, - }) - .collect(), - 0, - ) + })); + } + } } - }; + candidates.extend( + self.novelty_terms + .iter() + .filter(|nt| self.novelty_anchor_matches(&row, nt)) + .map(|nt| Candidate::Materialized(nt.term.clone())), + ); + (candidates, 0) + } + }; for candidate in &candidates { let mut out = row.clone(); if self.apply(candidate, t, &mut out, ctx)? { diff --git a/fluree-db-transact/src/stage.rs b/fluree-db-transact/src/stage.rs index 966e973e8f..8b51f22e9b 100644 --- a/fluree-db-transact/src/stage.rs +++ b/fluree-db-transact/src/stage.rs @@ -1372,6 +1372,8 @@ async fn scan_graph_flakes( return Err(TransactError::WholeGraphScanTooLarge { limit: l }); } } + // Links are derived from the bundles this rewrites, never written. + flakes.retain(|f| !fluree_db_core::is_rdf_reifies(&f.p)); for f in &mut flakes { f.g = g_sid.cloned(); } @@ -2491,6 +2493,8 @@ async fn classify_subject_lifecycle( delta: &SubjectDelta, pre_classes: &[Sid], ) -> Result { + // A reifier's link is derived and never retracted by a transaction, so + // it would make every full delete of a reifier look partial. let scan_pre_state = || async move { let rm = fluree_db_core::RangeMatch::new().with_subject(subject.clone()); let opts = fluree_db_core::RangeOptions::new().with_to_t(ledger.t()); @@ -2504,6 +2508,10 @@ async fn classify_subject_lifecycle( opts, ) .await + .map(|mut flakes| { + flakes.retain(|f| !fluree_db_core::is_rdf_reifies(&f.p)); + flakes + }) }; if delta.has_assert { From 82e4d26bbec1d522764c1ed5c1214829499622c1 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 11:50:45 -0400 Subject: [PATCH 29/92] feat(query): provisional handles for terms only novelty holds A link whose term the index has not interned took the raw-flake lane, so any annotation committed since the last build took a link count off the count plan (P2 on the dev slice: 25 ms indexed, 2.8 s with 10k such links). Dictionary novelty now keeps a term table, populated from the links novelty derives, and a term there gets a provisional handle in its inner predicate's interval, above every indexed sequence: 2^31 plus its index. The term-dictionary builder stops below that. Composition tries the persisted dictionary first. A provisional handle's key is computed from its novelty term, the novelty-aware graph view and the batched join decode it, and publish retires the entries the index now interns. apply_commit returns the links it derived so each write path registers them. P2 with the same novelty: 69 ms, on the count plan. --- fluree-db-api/src/commit_transfer.rs | 10 +- fluree-db-api/tests/it_triple_term_links.rs | 29 +++ fluree-db-binary-index/src/dict/term_dict.rs | 7 +- .../src/dict_novelty_safe.rs | 43 ++-- .../src/read/binary_index_store.rs | 29 +++ fluree-db-core/src/dict_novelty.rs | 194 +++++++++++++++++- fluree-db-core/src/triple_term.rs | 38 ++++ fluree-db-ledger/src/lib.rs | 8 +- fluree-db-novelty/src/lib.rs | 15 +- fluree-db-query/src/binary_scan.rs | 95 +++++++-- fluree-db-query/src/eval/rdf.rs | 31 ++- fluree-db-query/src/join.rs | 11 + fluree-db-query/src/term_components.rs | 70 ++++--- fluree-db-transact/src/commit.rs | 6 +- 14 files changed, 505 insertions(+), 81 deletions(-) diff --git a/fluree-db-api/src/commit_transfer.rs b/fluree-db-api/src/commit_transfer.rs index 8e3c5debba..169c512332 100644 --- a/fluree-db-api/src/commit_transfer.rs +++ b/fluree-db-api/src/commit_transfer.rs @@ -1479,11 +1479,19 @@ fn apply_pushed_commits_to_state( })?; Arc::make_mut(&mut runtime_small_dicts).populate_from_flakes(flakes); // Apply to novelty. - novelty + let links = novelty .apply_commit(flakes.clone(), *t, &reverse_graph) .map_err(|e| { PushError::Internal(format!("novelty apply_commit failed at t={t}: {e}")) })?; + fluree_db_binary_index::dict_novelty_safe::populate_dict_novelty_safe( + Arc::make_mut(&mut dict_novelty), + store_opt, + links.iter(), + ) + .map_err(|e| { + PushError::Internal(format!("populate_dict_novelty_safe failed at t={t}: {e}")) + })?; } base.novelty = Arc::new(novelty); diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 8ee3fd084d..88029556a4 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -1253,3 +1253,32 @@ async fn novelty_links_follow_the_index_they_were_published_over() { }) .await; } + +/// A count over links whose terms only novelty holds stays on the count plan: +/// those terms get provisional handles, so their links join the encoded +/// overlay instead of the raw-flake lane the plan declines on. +#[tokio::test(flavor = "current_thread")] +async fn novelty_terms_keep_link_counts_on_the_count_plan() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let (fluree, ledger) = import( + &[("claims.ttl", CLAIMS)], + "it/triple-term-links:novelty-count", + ) + .await; + let ledger = change_claims_without_indexing(&fluree, ledger).await; + let count = || "SELECT (COUNT(*) AS ?n) WHERE { << ?s ?p ?o >> ex:source ?src }".to_string(); + // Register the stamp callsite before this thread's subscriber reads it. + run_link_query(&fluree, &ledger, count()).await; + let (store, _guard) = support::span_capture::init_test_tracing(); + tracing::callsite::rebuild_interest_cache(); + + let got = run_link_query(&fluree, &ledger, count()).await; + assert_eq!(got, vec![vec!["4".to_string()]], "{got:#?}"); + let outcomes: Vec = store + .find_events("fast-path outcome") + .iter() + .filter(|e| e.fields.get("site").map(String::as_str) == Some("count-plan")) + .filter_map(|e| e.fields.get("outcome").cloned()) + .collect(); + assert_eq!(outcomes, ["proceed"], "{outcomes:?}"); +} diff --git a/fluree-db-binary-index/src/dict/term_dict.rs b/fluree-db-binary-index/src/dict/term_dict.rs index f2f8794849..216128f6bb 100644 --- a/fluree-db-binary-index/src/dict/term_dict.rs +++ b/fluree-db-binary-index/src/dict/term_dict.rs @@ -267,11 +267,12 @@ impl TermDictBuilder { return Ok(*h); } let next = self.next_seq.entry(key.p_id).or_insert(0); - if *next == u32::MAX { + // The sequences above are novelty's provisional handles. + if *next >= fluree_db_core::triple_term::NOVELTY_TERM_SEQ_BASE { return Err(io::Error::other(format!( - "triple-term dictionary: predicate {} exhausted its {}-bit sequence space", + "triple-term dictionary: predicate {} exhausted its {} indexed sequences", key.p_id, - fluree_db_core::triple_term::TERM_SEQ_BITS + fluree_db_core::triple_term::NOVELTY_TERM_SEQ_BASE ))); } let handle = term_handle(key.p_id, *next); diff --git a/fluree-db-binary-index/src/dict_novelty_safe.rs b/fluree-db-binary-index/src/dict_novelty_safe.rs index f20d200c12..d500916265 100644 --- a/fluree-db-binary-index/src/dict_novelty_safe.rs +++ b/fluree-db-binary-index/src/dict_novelty_safe.rs @@ -67,25 +67,40 @@ pub fn populate_dict_novelty_safe<'a>( Ok(()) }; + let mut string = |dict_novelty: &mut DictNovelty, s: &'a str, t: i64| -> io::Result<()> { + if dict_novelty.strings.find_string(s).is_some() || persisted_strings.contains(s) { + return Ok(()); + } + let persisted = match store { + Some(store) => string_is_persisted(store, s)?, + None => false, + }; + if persisted { + persisted_strings.insert(s); + } else { + dict_novelty.strings.assign_or_lookup_at(s, t); + } + Ok(()) + }; + for flake in flakes { subject(dict_novelty, &flake.s, flake.t)?; match &flake.o { FlakeValue::Ref(sid) => subject(dict_novelty, sid, flake.t)?, - FlakeValue::String(s) | FlakeValue::Json(s) => { - if dict_novelty.strings.find_string(s).is_some() - || persisted_strings.contains(s.as_str()) - { - continue; - } - let persisted = match store { - Some(store) => string_is_persisted(store, s)?, - None => false, - }; - if persisted { - persisted_strings.insert(s); - } else { - dict_novelty.strings.assign_or_lookup_at(s, flake.t); + FlakeValue::String(s) | FlakeValue::Json(s) => string(dict_novelty, s, flake.t)?, + // A term's components need ids for its provisional handle's key. + // The term itself is registered whether or not the index holds + // it: readers try the persisted dictionary first. + FlakeValue::TripleTerm(term) => { + subject(dict_novelty, &term.s, flake.t)?; + match &term.o { + FlakeValue::Ref(sid) => subject(dict_novelty, sid, flake.t)?, + FlakeValue::String(s) | FlakeValue::Json(s) => { + string(dict_novelty, s, flake.t)?; + } + _ => {} } + dict_novelty.terms.assign_or_lookup_at(term, flake.t); } _ => {} } diff --git a/fluree-db-binary-index/src/read/binary_index_store.rs b/fluree-db-binary-index/src/read/binary_index_store.rs index dd640b1dd3..f461c86b5a 100644 --- a/fluree-db-binary-index/src/read/binary_index_store.rs +++ b/fluree-db-binary-index/src/read/binary_index_store.rs @@ -2937,6 +2937,26 @@ impl BinaryGraphView { self.namespace_codes_fallback.clone() } + /// A provisional triple-term handle's term, from dictionary novelty; + /// `None` for an indexed handle. + fn novelty_term( + dn: &fluree_db_core::dict_novelty::DictNovelty, + handle: u64, + ) -> Option> { + let index = fluree_db_core::triple_term::novelty_term_index(handle)?; + Some( + dn.terms + .resolve(index) + .map(|term| FlakeValue::TripleTerm(Box::new(term.clone()))) + .ok_or_else(|| { + io::Error::new( + io::ErrorKind::NotFound, + format!("provisional triple-term handle {handle:#x} not in novelty"), + ) + }), + ) + } + /// Decode a value from `(o_type, o_key)`. Novelty-aware when `dict_novelty` /// is set: dict-backed types (IriRef, StringDict, JsonArena) route through /// watermark checks; all other types delegate directly to the store. @@ -2961,6 +2981,11 @@ impl BinaryGraphView { return Ok(FlakeValue::Json(s)); } } + DecodeKind::TripleTermDict => { + if let Some(term) = Self::novelty_term(dn, o_key) { + return term; + } + } DecodeKind::Duration => { // Generic durations are string-dict-backed; an overlay // row can carry a novelty string id (the canonical @@ -3021,6 +3046,10 @@ impl BinaryGraphView { if let Some(s) = self.resolve_novel_string(dn, o_key as u32) { return Ok(FlakeValue::Json(s)); } + } else if o_kind == ObjKind::TRIPLE_TERM.as_u8() { + if let Some(term) = Self::novelty_term(dn, o_key) { + return term; + } } } } diff --git a/fluree-db-core/src/dict_novelty.rs b/fluree-db-core/src/dict_novelty.rs index 98db8ab35c..f4c8192dae 100644 --- a/fluree-db-core/src/dict_novelty.rs +++ b/fluree-db-core/src/dict_novelty.rs @@ -1,4 +1,4 @@ -//! Dictionary novelty overlay for subjects and strings. +//! Dictionary novelty overlay for subjects, strings and triple terms. //! //! `DictNovelty` is a LedgerState-scoped layer that tracks novel dictionary //! entries (subjects and strings) introduced by commits since the last index @@ -42,7 +42,8 @@ use std::sync::Arc; use crate::ns_vec_bi_dict::{lookup_key, NsVecBiDict}; use crate::vec_bi_dict::VecBiDict; -use crate::{Flake, FlakeValue}; +use crate::{Flake, FlakeValue, TripleTermValue}; +use std::collections::HashMap; /// Does not match `namespaces::OVERFLOW` (0xFFFE); no production path assigns this /// code, so the special case below is never taken (#1843). @@ -76,6 +77,7 @@ pub fn subject_reverse_key(ns_code: u16, suffix: &str) -> Box<[u8]> { pub struct DictNovelty { pub subjects: SubjectDictNovelty, pub strings: StringDictNovelty, + pub terms: TermDictNovelty, initialized: bool, } @@ -88,6 +90,7 @@ impl DictNovelty { Self { subjects: SubjectDictNovelty::default(), strings: StringDictNovelty::default(), + terms: TermDictNovelty::default(), initialized: true, } } @@ -101,6 +104,7 @@ impl DictNovelty { Self { subjects: SubjectDictNovelty::default(), strings: StringDictNovelty::default(), + terms: TermDictNovelty::default(), initialized: false, } } @@ -134,6 +138,7 @@ impl DictNovelty { watermark: string_wm, parent: None, }, + terms: TermDictNovelty::default(), initialized: true, } } @@ -171,6 +176,7 @@ impl DictNovelty { .retire_seen_through(t, trimmed_wm, overflow_wm); let strings = self.strings.inner.retire_seen_through(t, string_wm + 1); self.strings.watermark = string_wm; + self.terms.retire_seen_through(t); Some((subjects, strings)) } @@ -186,6 +192,11 @@ impl DictNovelty { watermark: parent.strings.watermark, parent: Some(Arc::clone(&parent)), }, + terms: TermDictNovelty { + base: parent.terms.next_index(), + parent: Some(Arc::clone(&parent)), + ..TermDictNovelty::default() + }, initialized: parent.initialized, } } @@ -243,6 +254,24 @@ impl DictNovelty { FlakeValue::String(s) | FlakeValue::Json(s) => { self.strings.assign_or_lookup_at(s, flake.t); } + FlakeValue::TripleTerm(term) => { + self.subjects + .assign_or_lookup_at(term.s.namespace_code, &term.s.name, flake.t); + match &term.o { + FlakeValue::Ref(sid) => { + self.subjects.assign_or_lookup_at( + sid.namespace_code, + &sid.name, + flake.t, + ); + } + FlakeValue::String(s) | FlakeValue::Json(s) => { + self.strings.assign_or_lookup_at(s, flake.t); + } + _ => {} + } + self.terms.assign_or_lookup_at(term, flake.t); + } _ => {} } } @@ -483,6 +512,114 @@ impl StringDictNovelty { // Tests // =========================================================================== +// --------------------------------------------------------------------------- +// TermDictNovelty +// --------------------------------------------------------------------------- + +/// Triple-term dictionary novelty: the terms novelty's links name, each with +/// an index that, under the term's inner predicate id, forms a provisional +/// handle (`triple_term::novelty_term_handle`). Readers try the persisted +/// dictionary first, so a term the index interned keeps its handle. +#[derive(Clone, Debug, Default)] +pub struct TermDictNovelty { + /// `(term, earliest t seen)`, by index minus `base`. + entries: Vec<(Arc, i64)>, + by_term: HashMap, u32>, + /// Index of this layer's first entry: a layer allocates above its parent. + base: u32, + parent: Option>, +} + +impl TermDictNovelty { + /// Look up or assign the index of `term`, here or in a parent. + pub fn assign_or_lookup_at(&mut self, term: &TripleTermValue, t: i64) -> u32 { + if let Some(&local) = self.by_term.get(term) { + let entry = &mut self.entries[(local - self.base) as usize]; + entry.1 = entry.1.min(t); + return local; + } + if let Some(index) = self.parent.as_ref().and_then(|p| p.terms.find(term)) { + return index; + } + let index = self.next_index(); + let term = Arc::new(term.clone()); + self.by_term.insert(Arc::clone(&term), index); + self.entries.push((term, t)); + index + } + + /// The index of `term`, here or in a parent. + pub fn find(&self, term: &TripleTermValue) -> Option { + let mut dict = self; + loop { + if let Some(&index) = dict.by_term.get(term) { + return Some(index); + } + dict = &dict.parent.as_ref()?.terms; + } + } + + /// The term at `index`, here or in a parent. + pub fn resolve(&self, index: u32) -> Option<&TripleTermValue> { + let mut dict = self; + loop { + if index >= dict.base { + return dict + .entries + .get((index - dict.base) as usize) + .map(|(term, _)| term.as_ref()); + } + dict = &dict.parent.as_ref()?.terms; + } + } + + /// Every term, parents first, with its index. + pub fn iter(&self) -> impl Iterator + '_ { + let mut chain = vec![self]; + let mut dict = self; + while let Some(parent) = dict.parent.as_ref() { + dict = &parent.terms; + chain.push(dict); + } + chain.reverse(); + chain.into_iter().flat_map(|dict| { + dict.entries + .iter() + .enumerate() + .map(move |(i, (term, _))| (dict.base + i as u32, term.as_ref())) + }) + } + + fn next_index(&self) -> u32 { + self.base + self.entries.len() as u32 + } + + /// Drop terms first seen at or before `t` (the index now interns them) + /// and renumber the rest from zero. Returns the number dropped. + fn retire_seen_through(&mut self, t: i64) -> usize { + let before = self.entries.len(); + self.entries.retain(|(_, seen)| *seen > t); + self.by_term = self + .entries + .iter() + .enumerate() + .map(|(i, (term, _))| (Arc::clone(term), i as u32)) + .collect(); + self.base = 0; + before - self.entries.len() + } + + /// Number of terms, parents included. + pub fn len(&self) -> usize { + self.entries.len() + self.parent.as_ref().map_or(0, |p| p.terms.len()) + } + + /// True when no term has been registered here or in a parent. + pub fn is_empty(&self) -> bool { + self.len() == 0 + } +} + #[cfg(test)] mod tests { use super::*; @@ -753,7 +890,7 @@ mod tests { let layer = DictNovelty::layered_over(Arc::clone(&parent)); assert!(Arc::ptr_eq(layer.parent().unwrap(), &parent)); // One `Arc` per sub-dictionary; nothing is copied. - assert_eq!(Arc::strong_count(&parent), 3); + assert_eq!(Arc::strong_count(&parent), 4); assert!(layer.is_initialized()); let alice = parent.subjects.find_subject(2, "alice").unwrap(); @@ -838,8 +975,8 @@ mod tests { drop(a); assert_eq!( Arc::strong_count(&parent), - 3, - "only b's two references remain" + 4, + "only b's three references remain" ); assert_eq!(parent.subjects.len(), 3); assert_eq!(parent.subjects.find_subject(2, "from-a"), None); @@ -917,4 +1054,51 @@ mod tests { "a layer never retires" ); } + + fn term(s: &str, o: i64) -> TripleTermValue { + TripleTermValue { + s: crate::Sid::new(9, s), + p: crate::Sid::new(9, "p"), + o: FlakeValue::Long(o), + dt: crate::Sid::new(2, "integer"), + lang: None, + } + } + + /// Link flakes register their term and its components; a layer numbers + /// above its parent and reads through it; a publish drops the terms it + /// covered and renumbers the rest. + #[test] + fn terms_register_layer_and_retire() { + let mut d = DictNovelty::with_watermarks(vec![], 0); + let link = |t: i64, term: TripleTermValue| { + Flake::new( + crate::Sid::new(9, "r"), + crate::namespaces::rdf_reifies_sid().clone(), + FlakeValue::TripleTerm(Box::new(term)), + crate::namespaces::triple_term_datatype_sid().clone(), + t, + true, + None, + ) + }; + d.populate_from_flakes(&[link(1, term("a", 1)), link(2, term("b", 2))]); + assert_eq!(d.terms.find(&term("a", 1)), Some(0)); + assert_eq!(d.terms.find(&term("b", 2)), Some(1)); + assert!(d.subjects.find_subject(9, "a").is_some()); + + let parent = Arc::new(d.clone()); + let mut layer = DictNovelty::layered_over(Arc::clone(&parent)); + assert_eq!(layer.terms.assign_or_lookup_at(&term("a", 1), 3), 0); + assert_eq!(layer.terms.assign_or_lookup_at(&term("c", 3), 3), 2); + assert_eq!(layer.terms.resolve(1), Some(&term("b", 2))); + assert_eq!(layer.terms.resolve(2), Some(&term("c", 3))); + assert_eq!(parent.terms.find(&term("c", 3)), None); + assert_eq!(layer.terms.len(), 3); + + d.retire_seen_through(1, &[], 0).unwrap(); + assert_eq!(d.terms.find(&term("a", 1)), None); + assert_eq!(d.terms.find(&term("b", 2)), Some(0)); + assert_eq!(d.terms.resolve(0), Some(&term("b", 2))); + } } diff --git a/fluree-db-core/src/triple_term.rs b/fluree-db-core/src/triple_term.rs index 249e38e47a..641da9ab93 100644 --- a/fluree-db-core/src/triple_term.rs +++ b/fluree-db-core/src/triple_term.rs @@ -35,6 +35,31 @@ pub const fn term_handle_seq(handle: u64) -> u32 { (handle & TERM_SEQ_MASK) as u32 } +/// First per-predicate sequence of a provisional handle: one dictionary +/// novelty assigns to a term the index has not interned. Indexed sequences +/// stay below it, so a provisional handle still falls in its predicate's +/// interval and never collides with an indexed one. +pub const NOVELTY_TERM_SEQ_BASE: u32 = 1 << 31; + +/// The provisional handle for dictionary-novelty term `index` under +/// `inner_p_id`. +#[inline] +pub const fn novelty_term_handle(inner_p_id: u32, index: u32) -> u64 { + term_handle(inner_p_id, NOVELTY_TERM_SEQ_BASE | index) +} + +/// The dictionary-novelty index of a provisional handle; `None` for an +/// indexed one. +#[inline] +pub const fn novelty_term_index(handle: u64) -> Option { + let seq = term_handle_seq(handle); + if seq >= NOVELTY_TERM_SEQ_BASE { + Some(seq - NOVELTY_TERM_SEQ_BASE) + } else { + None + } +} + /// Inclusive `o_key` interval holding every term under `inner_p_id`. #[inline] pub const fn term_handle_range(inner_p_id: u32) -> (u64, u64) { @@ -108,6 +133,19 @@ mod tests { assert!(term_handle(6, u32::MAX) < lo); } + #[test] + fn provisional_handles_stay_in_their_predicate_interval() { + let h = novelty_term_handle(7, 3); + let (lo, hi) = term_handle_range(7); + assert!(lo <= h && h <= hi); + assert_eq!(term_handle_p_id(h), 7); + assert_eq!(novelty_term_index(h), Some(3)); + assert_eq!( + novelty_term_index(term_handle(7, NOVELTY_TERM_SEQ_BASE - 1)), + None + ); + } + #[test] fn key_roundtrip_orders_subject_first() { let k = TermKey { diff --git a/fluree-db-ledger/src/lib.rs b/fluree-db-ledger/src/lib.rs index 1efbb8d96d..455b8ca702 100644 --- a/fluree-db-ledger/src/lib.rs +++ b/fluree-db-ledger/src/lib.rs @@ -941,7 +941,9 @@ impl LedgerState { Arc::make_mut(&mut self.dict_novelty).populate_from_flakes(&flakes); Arc::make_mut(&mut self.runtime_small_dicts).populate_from_flakes(&flakes); - Arc::make_mut(&mut self.novelty).apply_commit(flakes, next_t, &reverse_graph)?; + let links = + Arc::make_mut(&mut self.novelty).apply_commit(flakes, next_t, &reverse_graph)?; + Arc::make_mut(&mut self.dict_novelty).populate_from_flakes(&links); Ok(()) } @@ -1071,7 +1073,9 @@ impl LedgerState { // avoids the unconditional deep clone the previous clone-then-swap forced. Arc::make_mut(&mut self.dict_novelty).populate_from_flakes(&all_flakes); Arc::make_mut(&mut self.runtime_small_dicts).populate_from_flakes(&all_flakes); - Arc::make_mut(&mut self.novelty).apply_commit(all_flakes, commit_t, &reverse_graph)?; + let links = + Arc::make_mut(&mut self.novelty).apply_commit(all_flakes, commit_t, &reverse_graph)?; + Arc::make_mut(&mut self.dict_novelty).populate_from_flakes(&links); // Update state self.head_commit_id = Some(commit_id.clone()); diff --git a/fluree-db-novelty/src/lib.rs b/fluree-db-novelty/src/lib.rs index 132b7f91ca..9efb1c9f7d 100644 --- a/fluree-db-novelty/src/lib.rs +++ b/fluree-db-novelty/src/lib.rs @@ -863,16 +863,19 @@ impl Novelty { /// Unknown graph Sids cause an error — no silent fallback to the default /// graph. /// - /// Atomic: graph routing (the only fallible step) is resolved before any - /// mutation, so an error leaves novelty untouched. + /// Atomic: graph routing and link derivation (the fallible steps) run + /// before any mutation, so an error leaves novelty untouched. + /// + /// Returns the reification links derived for the commit (see [`links`]), + /// which the caller's dictionary novelty must register. pub fn apply_commit( &mut self, flakes: Vec, commit_t: i64, reverse_graph: &HashMap, - ) -> Result<()> { + ) -> Result> { if flakes.is_empty() { - return Ok(()); + return Ok(Vec::new()); } let span = tracing::debug_span!( @@ -913,10 +916,12 @@ impl Novelty { // Links read the index, so they are derived before any mutation too. let links_current = self.current_link_base().is_some(); + let mut derived: Vec = Vec::new(); if let Some(base) = self.current_link_base() { let touched = links::touched_reifiers(routed.iter().map(|(f, g)| (*g, f))); if !touched.is_empty() { let links = self.derive_links(base, &touched, self.t)?; + derived = links.iter().map(|(_, f)| f.clone()).collect(); routed.extend(links.into_iter().map(|(g_id, f)| (f, g_id))); } } @@ -1009,7 +1014,7 @@ impl Novelty { self.push_segment(g_id, seg); } - Ok(()) + Ok(derived) } /// Bulk-apply many commits' flakes in a single pass (first-load / catch-up). diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index 2f70f439cf..5230df9fc1 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -1483,8 +1483,7 @@ impl BinaryScanOperator { if o_type != OType::TRIPLE_TERM.as_u16() { continue; } - let Some(key) = store_arc - .resolve_term_key(o_key) + let Some(key) = term_key_for_handle(o_key, &store_arc, dict_novelty_arc.as_ref()) .map_err(|e| QueryError::from_io("resolve_term_key", e))? else { continue; @@ -4529,20 +4528,46 @@ fn encode_term_key_filter( Ok(TermKeyFilter { s_id, o }) } -/// A constant triple term's handle, composed through the term dictionary. -/// A subject, predicate, object or term the dictionary does not hold is -/// `NotFound`: the link index only ever carries interned handles, so the -/// pattern cannot match a base row and the caller may skip to novelty. +/// A constant triple term's handle: the term dictionary's, else the +/// provisional handle dictionary novelty gives a term the index has not +/// interned. A term neither holds is `NotFound`: no link names it, so the +/// pattern cannot match. pub(crate) fn compose_term_handle( term: &fluree_db_core::TripleTermValue, store: &BinaryIndexStore, dict_novelty: Option<&Arc>, ) -> std::io::Result<(OType, u64)> { use std::io::{Error, ErrorKind}; + match persisted_term_handle(term, store, dict_novelty) { + Ok(Some(handle)) => return Ok((OType::TRIPLE_TERM, handle)), + Ok(None) => {} + Err(e) if matches!(e.kind(), ErrorKind::NotFound | ErrorKind::Unsupported) => {} + Err(e) => return Err(e), + } + let provisional = dict_novelty + .filter(|dn| dn.is_initialized()) + .and_then(|dn| dn.terms.find(term)) + .zip(store.sid_to_p_id(&term.p)) + .map(|(index, p_id)| fluree_db_core::triple_term::novelty_term_handle(p_id, index)); + match provisional { + Some(handle) => Ok((OType::TRIPLE_TERM, handle)), + None => Err(Error::new( + ErrorKind::NotFound, + "triple term is not interned", + )), + } +} + +/// The handle the persisted term dictionary holds for `term`. +fn persisted_term_handle( + term: &fluree_db_core::TripleTermValue, + store: &BinaryIndexStore, + dict_novelty: Option<&Arc>, +) -> std::io::Result> { + let Some(p_id) = store.sid_to_p_id(&term.p) else { + return Ok(None); + }; let s_id = resolve_subject_v3(&term.s, store, dict_novelty)?; - let p_id = store - .sid_to_p_id(&term.p) - .ok_or_else(|| Error::new(ErrorKind::NotFound, "triple term predicate is not known"))?; let (o_type, o_key) = value_to_otype_okey( &term.o, &term.dt, @@ -4551,18 +4576,56 @@ pub(crate) fn compose_term_handle( dict_novelty, None, )?; - let key = fluree_db_core::triple_term::TermKey { + store.find_term_handle(&fluree_db_core::triple_term::TermKey { s_id, p_id, o_type, o_key, + }) +} + +/// The encoded base edge behind any term handle: the dictionary's key, or for +/// a provisional handle the key its novelty term encodes to. `None` when +/// neither dictionary can name it. +pub(crate) fn term_key_for_handle( + handle: u64, + store: &BinaryIndexStore, + dict_novelty: Option<&Arc>, +) -> std::io::Result> { + use fluree_db_core::triple_term::{novelty_term_index, term_handle_p_id, TermKey}; + let Some(index) = novelty_term_index(handle) else { + return store.resolve_term_key(handle); }; - match store.find_term_handle(&key)? { - Some(handle) => Ok((OType::TRIPLE_TERM, handle)), - None => Err(Error::new( - ErrorKind::NotFound, - "triple term is not interned", - )), + let Some(term) = dict_novelty.and_then(|dn| dn.terms.resolve(index)) else { + return Ok(None); + }; + let encoded = resolve_subject_v3(&term.s, store, dict_novelty).and_then(|s_id| { + value_to_otype_okey( + &term.o, + &term.dt, + term.lang.as_deref(), + store, + dict_novelty, + None, + ) + .map(|(o_type, o_key)| TermKey { + s_id, + p_id: term_handle_p_id(handle), + o_type, + o_key, + }) + }); + match encoded { + Ok(key) => Ok(Some(key)), + Err(e) + if matches!( + e.kind(), + std::io::ErrorKind::NotFound | std::io::ErrorKind::Unsupported + ) => + { + Ok(None) + } + Err(e) => Err(e), } } diff --git a/fluree-db-query/src/eval/rdf.rs b/fluree-db-query/src/eval/rdf.rs index dc59867163..52d26642fb 100644 --- a/fluree-db-query/src/eval/rdf.rs +++ b/fluree-db-query/src/eval/rdf.rs @@ -515,14 +515,18 @@ fn encoded_term<'c, R: RowAccess>( let Some(store) = ctx.binary_store.as_deref() else { return Ok(None); }; - let key = store - .resolve_term_key(*o_key) - .map_err(|e| QueryError::from_io("resolve_term_key", e))? - .ok_or_else(|| { - QueryError::Internal(format!( - "triple-term handle {o_key:#x} has no dictionary entry" - )) - })?; + let key = crate::binary_scan::term_key_for_handle(*o_key, store, ctx.dict_novelty.as_ref()) + .map_err(|e| QueryError::from_io("resolve_term_key", e))?; + // A provisional handle whose components no dictionary encodes takes the + // value path through its novelty term. + if key.is_none() && fluree_db_core::triple_term::novelty_term_index(*o_key).is_some() { + return Ok(None); + } + let key = key.ok_or_else(|| { + QueryError::Internal(format!( + "triple-term handle {o_key:#x} has no dictionary entry" + )) + })?; Ok(Some((key, *t, ctx))) } @@ -590,6 +594,17 @@ pub(crate) fn term_component_binding( val: fluree_db_core::FlakeValue::TripleTerm(term), .. }) => term.as_ref().clone(), + Some(Binding::EncodedLit { o_kind, o_key, .. }) + if *o_kind == fluree_db_core::value_id::ObjKind::TRIPLE_TERM.as_u8() => + { + match fluree_db_core::triple_term::novelty_term_index(*o_key) + .zip(ctx.and_then(|c| c.dict_novelty.as_ref())) + .and_then(|(index, dn)| dn.terms.resolve(index)) + { + Some(term) => term.clone(), + None => return Ok(None), + } + } _ => return Ok(None), }, _ => match triple_term_arg(args, row, ctx, name)? { diff --git a/fluree-db-query/src/join.rs b/fluree-db-query/src/join.rs index f879a62e14..d06d3382a3 100644 --- a/fluree-db-query/src/join.rs +++ b/fluree-db-query/src/join.rs @@ -3243,6 +3243,17 @@ fn decode_overlay_object( ov.resolve_string_value(o_key as u32) .map_err(|e| decode_err("resolve_string_value", &e))?, ), + (DecodeKind::TripleTermDict, _) + if fluree_db_core::triple_term::novelty_term_index(o_key).is_some() => + { + let term = fluree_db_core::triple_term::novelty_term_index(o_key) + .zip(ctx.dict_novelty.as_ref()) + .and_then(|(index, dn)| dn.terms.resolve(index)) + .ok_or_else(|| { + decode_err("novelty term", &format!("no term for handle {o_key:#x}")) + })?; + FlakeValue::TripleTerm(Box::new(term.clone())) + } _ => store .decode_value_v3(o_type, o_key, p_id, ctx.binary_g_id) .map_err(|e| decode_err("decode_value_v3", &e))?, diff --git a/fluree-db-query/src/term_components.rs b/fluree-db-query/src/term_components.rs index e7e6ba7bd1..fab433478b 100644 --- a/fluree-db-query/src/term_components.rs +++ b/fluree-db-query/src/term_components.rs @@ -27,7 +27,7 @@ use crate::var_registry::VarId; use async_trait::async_trait; use fluree_db_binary_index::BinaryIndexStore; use fluree_db_core::o_type::OType; -use fluree_db_core::triple_term::TermKey; +use fluree_db_core::triple_term::{novelty_term_index, TermKey}; use fluree_db_core::value_id::ObjKind; use fluree_db_core::{DatatypeConstraint, FlakeValue, Sid, TripleTermValue}; use std::collections::{HashMap, HashSet}; @@ -50,11 +50,11 @@ struct EncodedConstants { o: Option<(u16, u64)>, } -/// A term only novelty's links name, with its subject's id when a dictionary -/// holds one. +/// A term only novelty's links name: encoded under its provisional handle +/// when dictionary novelty gives it one, else materialized. struct NoveltyTerm { term: Box, - s_id: Option, + encoded: Option<(u64, TermKey)>, } pub struct TermComponentsOperator { @@ -126,30 +126,29 @@ impl TermComponentsOperator { let dict_novelty = ctx.dict_novelty.as_ref(); let mut out = Vec::new(); for term in seen { - let s_id = match &self.store { + let encoded = match &self.store { Some(store) => { - match missing(crate::binary_scan::compose_term_handle( + let handle = missing(crate::binary_scan::compose_term_handle( &term, store, dict_novelty, )) .map_err(|e| QueryError::from_io("term components: novelty", e))? - { + .map(|(_, handle)| handle); + match handle { // The dictionary offers it already. - Some(_) => continue, - None => missing(crate::binary_scan::resolve_subject_v3( - &term.s, - store, - dict_novelty, - )) - .map_err(|e| QueryError::from_io("term components: novelty", e))?, + Some(h) if novelty_term_index(h).is_none() => continue, + Some(h) => crate::binary_scan::term_key_for_handle(h, store, dict_novelty) + .map_err(|e| QueryError::from_io("term components: novelty", e))? + .map(|key| (h, key)), + None => None, } } None => None, }; out.push(NoveltyTerm { term: Box::new(term), - s_id, + encoded, }); } Ok(out) @@ -161,7 +160,9 @@ impl TermComponentsOperator { match &self.pattern.subject { Component::Node(sid) => nt.term.s == *sid, Component::Var(v) => match self.value(row, *v) { - Some(Binding::EncodedSid { s_id, .. }) => nt.s_id == Some(*s_id), + Some(Binding::EncodedSid { s_id, .. }) => { + nt.encoded.is_none_or(|(_, key)| key.s_id == *s_id) + } Some(Binding::Sid { sid, .. }) => nt.term.s == *sid, _ => true, }, @@ -503,15 +504,29 @@ impl Operator for TermComponentsOperator { let Some(store) = &self.store else { continue; }; - let key = store - .resolve_term_key(o_key) - .map_err(|e| QueryError::from_io("resolve_term_key", e))? - .ok_or_else(|| { - QueryError::Internal(format!( - "triple-term handle {o_key:#x} has no dictionary entry" - )) - })?; - (vec![Candidate::Encoded { handle: o_key, key }], t) + let key = crate::binary_scan::term_key_for_handle( + o_key, + store, + ctx.dict_novelty.as_ref(), + ) + .map_err(|e| QueryError::from_io("resolve_term_key", e))?; + // A provisional handle whose components no dictionary + // encodes is offered as its novelty term. + let candidate = match key { + Some(key) => Candidate::Encoded { handle: o_key, key }, + None => match novelty_term_index(o_key) + .zip(ctx.dict_novelty.as_ref()) + .and_then(|(index, dn)| dn.terms.resolve(index)) + { + Some(term) => Candidate::Materialized(Box::new(term.clone())), + None => { + return Err(QueryError::Internal(format!( + "triple-term handle {o_key:#x} has no dictionary entry" + ))) + } + }, + }; + (vec![candidate], t) } Some(Binding::Lit { val: FlakeValue::TripleTerm(term), @@ -582,7 +597,10 @@ impl Operator for TermComponentsOperator { self.novelty_terms .iter() .filter(|nt| self.novelty_anchor_matches(&row, nt)) - .map(|nt| Candidate::Materialized(nt.term.clone())), + .map(|nt| match nt.encoded { + Some((handle, key)) => Candidate::Encoded { handle, key }, + None => Candidate::Materialized(nt.term.clone()), + }), ); (candidates, 0) } diff --git a/fluree-db-transact/src/commit.rs b/fluree-db-transact/src/commit.rs index e3377982a3..abed0dd8a1 100644 --- a/fluree-db-transact/src/commit.rs +++ b/fluree-db-transact/src/commit.rs @@ -1146,7 +1146,11 @@ fn finalize_state_with_base( if let Some(g_sid) = snapshot.encode_iri(&txn_meta_iri) { reverse_graph.entry(g_sid).or_insert(TXN_META_GRAPH_ID); } - Arc::make_mut(&mut new_novelty).apply_commit(all_flakes, new_t, &reverse_graph)?; + let links = + Arc::make_mut(&mut new_novelty).apply_commit(all_flakes, new_t, &reverse_graph)?; + if !links.is_empty() { + populate_dict_novelty(Arc::make_mut(&mut dict_novelty), store.as_deref(), &links)?; + } } // Rebuild + reattach the provider with the UPDATED dicts so reads through the From 39b928645fc8be927366062e10b4203ef2dad533 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 11:50:45 -0400 Subject: [PATCH 30/92] perf(query): bracket the novelty walk of a scan bound to a triple term A scan whose bound object was a term walked every novelty flake of its predicate, so each chained link lookup re-translated all of novelty's links, composing each term (P19b over 10k unindexed annotations: 12 s). Terms order by base subject and predicate first, so the terms sharing the bound term's form one run, which the walk now brackets. --- fluree-db-query/src/binary_scan.rs | 108 ++++++++++++++++++++++++----- 1 file changed, 92 insertions(+), 16 deletions(-) diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index 5230df9fc1..d86e277315 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -1744,7 +1744,7 @@ impl BinaryScanOperator { /// unconditional), and on the scan filter's match set never exceeding /// the comparator's equal-value run. Subject/predicate brackets compare /// `Sid`s only and need none of that reasoning. Reference objects use - /// that same exact Sid ordering in `bounded_ref_object_walk` below. + /// that same exact Sid ordering in `bounded_object_walk` below. /// /// The walk order is a property of the bracketed term, NOT of `self.index`: /// the *set* of novelty flakes for a subject is the same however it is @@ -1803,15 +1803,33 @@ impl BinaryScanOperator { /// Reference values form an exact, contiguous run in OPST order, or /// within one predicate in POST order. Keep every datatype, subject, /// timestamp and metadata value in that run so a base assertion never - /// loses its cancelling overlay op. + /// loses its cancelling overlay op. A triple term orders by its base + /// subject and predicate first, so the terms sharing those form a run that + /// holds every flake the scan's handle compare can match. /// Literal objects keep the existing whole-graph fallback: their scan /// match semantics can exceed a single comparator equality class. - fn bounded_ref_object_walk( + fn bounded_object_walk( object: Option<&FlakeValue>, predicate: Option<&Sid>, ) -> Option { - let FlakeValue::Ref(sid) = object? else { - return None; + let (lo, hi) = match object? { + FlakeValue::Ref(sid) => (FlakeValue::Ref(sid.clone()), FlakeValue::Ref(sid.clone())), + FlakeValue::TripleTerm(term) => { + let bound = |o: FlakeValue, dt: Sid| { + FlakeValue::TripleTerm(Box::new(fluree_db_core::TripleTermValue { + s: term.s.clone(), + p: term.p.clone(), + o, + dt, + lang: None, + })) + }; + ( + bound(FlakeValue::min(), Sid::min()), + bound(FlakeValue::max(), Sid::max()), + ) + } + _ => return None, }; Some(BoundedOverlayWalk { // POST leads with predicate, then object value/datatype; OPST @@ -1825,7 +1843,7 @@ impl BinaryScanOperator { first: Flake::new( Sid::min(), predicate.cloned().unwrap_or_else(Sid::min), - FlakeValue::Ref(sid.clone()), + lo, Sid::min(), i64::MIN, false, @@ -1834,7 +1852,7 @@ impl BinaryScanOperator { rhs: Flake::new( Sid::max(), predicate.cloned().unwrap_or_else(Sid::max), - FlakeValue::Ref(sid.clone()), + hi, Sid::max(), i64::MAX, true, @@ -2723,7 +2741,7 @@ impl Operator for BinaryScanOperator { // subject, use both predicate and reference object when available: // translating a whole predicate discards the selective object bound. let bounded = if s_sid.is_none() { - Self::bounded_ref_object_walk(self.bound_o.as_ref(), p_sid.as_ref()) + Self::bounded_object_walk(self.bound_o.as_ref(), p_sid.as_ref()) } else { None } @@ -4921,7 +4939,7 @@ mod bounded_overlay_walk_tests { } novelty.apply_commit(flakes, t, &graphs).expect("commit"); } - let bound = BinaryScanOperator::bounded_ref_object_walk(Some(&target), None) + let bound = BinaryScanOperator::bounded_object_walk(Some(&target), None) .expect("reference bracket"); assert_eq!(bound.index, IndexType::Opst); let events = |flakes: Vec| { @@ -4944,8 +4962,7 @@ mod bounded_overlay_walk_tests { // datatype range. This must be the intersection, not either whole run. for predicate in [sid(102, "p0"), sid(102, "p3"), sid(102, "absent")] { let bound = - BinaryScanOperator::bounded_ref_object_walk(Some(&target), Some(&predicate)) - .unwrap(); + BinaryScanOperator::bounded_object_walk(Some(&target), Some(&predicate)).unwrap(); assert_eq!(bound.index, IndexType::Post); for to_t in [1, 2, 3, 6, i64::MAX] { let expected = walk(&novelty, None, to_t) @@ -4956,10 +4973,69 @@ mod bounded_overlay_walk_tests { } } let missing = FlakeValue::Ref(sid(100, "absent")); - let bound = BinaryScanOperator::bounded_ref_object_walk(Some(&missing), None).unwrap(); + let bound = BinaryScanOperator::bounded_object_walk(Some(&missing), None).unwrap(); assert!(walk(&novelty, Some(&bound), i64::MAX).is_empty()); } + /// A term object's window is the run of terms sharing its base subject + /// and predicate: every flake on the target term, nothing from another + /// base edge. + #[test] + fn term_object_window_holds_the_terms_of_its_base_edge() { + let mut novelty = Novelty::new(0); + let graphs = HashMap::new(); + let term = |s: &str, p: &str, o: i64| { + FlakeValue::TripleTerm(Box::new(fluree_db_core::TripleTermValue { + s: sid(100, s), + p: sid(100, p), + o: FlakeValue::Long(o), + dt: sid(2, "integer"), + lang: None, + })) + }; + let reifies = sid(3, "reifies"); + let target = term("a", "p", 1); + let objects = [ + target.clone(), + term("a", "p", 2), + term("a", "q", 1), + term("b", "p", 1), + FlakeValue::Ref(sid(100, "a")), + ]; + for t in 1..=4 { + let flakes = (0..20) + .map(|i| { + Flake::new( + sid(101, &format!("r{i}")), + reifies.clone(), + objects[i % objects.len()].clone(), + sid(103, "tripleTerm"), + t, + t % 2 == 1, + None, + ) + }) + .collect(); + novelty.apply_commit(flakes, t, &graphs).expect("commit"); + } + let bound = BinaryScanOperator::bounded_object_walk(Some(&target), Some(&reifies)) + .expect("term bracket"); + assert_eq!(bound.index, IndexType::Post); + let in_run = |f: &Flake| match &f.o { + FlakeValue::TripleTerm(t) => t.s == sid(100, "a") && t.p == sid(100, "p"), + _ => false, + }; + for to_t in [1, 2, 4] { + let all = walk(&novelty, None, to_t); + let window = sorted(walk(&novelty, Some(&bound), to_t)); + assert_eq!( + window, + sorted(all.iter().filter(|f| in_run(f)).cloned().collect()) + ); + assert!(window.iter().any(|f| f.o == target)); + } + } + #[test] fn object_window_does_not_narrow_literal_match_semantics() { for value in [ @@ -4968,9 +5044,9 @@ mod bounded_overlay_walk_tests { FlakeValue::Double(f64::NAN), FlakeValue::String("5".into()), ] { - assert!(BinaryScanOperator::bounded_ref_object_walk(Some(&value), None).is_none()); + assert!(BinaryScanOperator::bounded_object_walk(Some(&value), None).is_none()); } - assert!(BinaryScanOperator::bounded_ref_object_walk(None, None).is_none()); + assert!(BinaryScanOperator::bounded_object_walk(None, None).is_none()); } /// The bracket must not be sensitive to a bound object: on SPOT with a bound @@ -5528,9 +5604,9 @@ mod tests { let object = FlakeValue::Ref(Sid::new(7, "target")); for predicate in [None, Some(&p)] { - let walk = BinaryScanOperator::bounded_ref_object_walk(Some(&object), predicate) + let walk = BinaryScanOperator::bounded_object_walk(Some(&object), predicate) .expect("reference must produce a bracketed walk"); - assert_pinned("bounded_ref_object_walk", &walk.rhs); + assert_pinned("bounded_object_walk", &walk.rhs); } } } From c66dd639a4fa952ff971ef71b854e933b98c28f9 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 16:57:24 -0400 Subject: [PATCH 31/92] fix(query): a BIND's agreement check decodes arena-keyed numbers An indexed decimal or big integer is keyed by a handle into its own predicate's NumBig arena, so bind_unifies, which normalized without a graph view, left it encoded: it never equalled a constant, a decoded value or the same number stored under another predicate. `BIND(1.5 AS ?o) ?s ex:p ?o` returned nothing over indexed data, as did a JSON-LD rebind across two decimal predicates. The check now decodes either side that is NUM_BIG through the graph view, as DISTINCT and MINUS already do. --- .../tests/it_query_expression_semantics.rs | 42 +++++++++++++++++++ fluree-db-query/src/object_binding.rs | 7 +++- 2 files changed, 47 insertions(+), 2 deletions(-) diff --git a/fluree-db-api/tests/it_query_expression_semantics.rs b/fluree-db-api/tests/it_query_expression_semantics.rs index d99d1ebbed..7a1da6dc71 100644 --- a/fluree-db-api/tests/it_query_expression_semantics.rs +++ b/fluree-db-api/tests/it_query_expression_semantics.rs @@ -2283,3 +2283,45 @@ async fn jsonld_bind_into_a_bound_variable_checks_it_when_only_counted() { json!([["ex:alice"]]) ); } + +/// Indexed decimals are keyed by a handle into their own predicate's arena, +/// so the check must compare decoded values: against a constant, and against +/// a decimal another predicate stored under a different handle. +#[tokio::test] +async fn bind_into_a_bound_variable_compares_indexed_decimals_by_value() { + let tx = json!({ + "@context": ctx(), + "@graph": [ + { "@id": "ex:acct", "ex:balance": { "@value": "1.50", "@type": "xsd:decimal" } }, + { "@id": "ex:other", "ex:balance": { "@value": "2.5", "@type": "xsd:decimal" } }, + { "@id": "ex:cap", "ex:limit": { "@value": "9", "@type": "xsd:decimal" } }, + { "@id": "ex:floor", "ex:limit": { "@value": "1.5", "@type": "xsd:decimal" } } + ] + }); + let fluree = FlureeBuilder::memory().build_memory(); + let ledger = seed_indexed(&fluree, "exprsem/bind-check:decimal", &tx).await; + + assert_eq!( + sparql_rows( + &fluree, + &ledger, + "PREFIX ex: \ + SELECT ?s WHERE { BIND(1.5 AS ?o) ?s ex:balance ?o }", + ) + .await, + json!([["ex:acct"]]) + ); + let q = json!({ + "@context": ctx(), + "select": ["?s", "?t"], + "where": [ + { "@id": "?s", "ex:balance": "?a" }, + { "@id": "?t", "ex:limit": "?v" }, + ["bind", "?a", "?v"] + ] + }); + assert_eq!( + jsonld_rows(&fluree, &ledger, &q).await, + json!([["ex:acct", "ex:floor"]]) + ); +} diff --git a/fluree-db-query/src/object_binding.rs b/fluree-db-query/src/object_binding.rs index 48fa28039b..2c179b53df 100644 --- a/fluree-db-query/src/object_binding.rs +++ b/fluree-db-query/src/object_binding.rs @@ -485,8 +485,11 @@ pub(crate) fn bind_unifies( let b = as_sid(computed); let a = a.as_ref().unwrap_or(existing); let b = b.as_ref().unwrap_or(computed); - normalize_for_key_cow(a, Some(store), None).as_ref() - == normalize_for_key_cow(b, Some(store), None).as_ref() + let gv = (is_numbig_encoded(a) || is_numbig_encoded(b)) + .then(|| ctx.graph_view()) + .flatten(); + normalize_for_key_cow(a, Some(store), gv.as_ref()).as_ref() + == normalize_for_key_cow(b, Some(store), gv.as_ref()).as_ref() } /// True if this is an arena-backed (NUM_BIG) encoded literal. From 99fb4bfe1d1d77fe41108ef77bf7cfe2be533a23 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 16:57:56 -0400 Subject: [PATCH 32/92] feat(index): link annotations whose object is a decimal, big integer or vector The main index keys these objects by arena handles scoped to one graph and predicate, so a graph-free term key could not carry them: import interned them as such, rebuild and novelty skipped them, and a link query never saw those annotations. A term now keys such an object by the string-dictionary id of its canonical form (a normalized decimal, the integer's digits, the vector at the f32 precision ingest stores), under the main index's o_type for the kind, so equal values in any graph or predicate share one term. - Rebuild and incremental builds intern the form when they observe an f:reifiesObject op; an incremental build decodes a base attachment's arena handle and joins its form to the window's strings, so a re-point retracts the term the base linked and keeps a base object it does not change. - Import writes the form into the chunk term table and remaps it with the chunk's strings. - Novelty links these attachments as it does every other. - Compose looks the form up in the persisted string dictionary; decompose parses it, so the object no longer decodes through the default graph's arena, and binds decoded. --- fluree-db-api/tests/it_triple_term_links.rs | 254 ++++++++++++++++++ .../src/read/binary_index_store.rs | 38 ++- fluree-db-core/src/commit/codec/raw_reader.rs | 1 + fluree-db-core/src/link.rs | 29 -- fluree-db-core/src/triple_term.rs | 122 ++++++++- .../run_index/build/incremental_resolve.rs | 71 ++++- .../src/run_index/resolve/link_synth.rs | 55 ++-- fluree-db-indexer/src/run_index/runs/spool.rs | 17 +- fluree-db-query/src/binary_scan.rs | 54 ++-- fluree-db-query/src/eval/rdf.rs | 15 +- fluree-db-query/src/term_components.rs | 3 +- fluree-db-transact/src/import_sink.rs | 34 ++- 12 files changed, 585 insertions(+), 108 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 88029556a4..3b62fc4f8c 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -1282,3 +1282,257 @@ async fn novelty_terms_keep_link_counts_on_the_count_plan() { .collect(); assert_eq!(outcomes, ["proceed"], "{outcomes:?}"); } + +const ARENA_CLAIMS: &str = r#"VERSION "1.2" +@prefix ex: . +@prefix f: . + +ex:acct ex:balance 1.50 ~ ex:c1 {| ex:source ex:ledger |} . +ex:acct ex:serial 123456789012345678901234567890 ~ ex:c2 {| ex:source ex:registry |} . +ex:acct ex:embedding "[0.5, -1.25]"^^f:embeddingVector ~ ex:c3 {| ex:source ex:model |} . +"#; + +/// Decimal, big-integer and vector objects, which the main index keys by +/// per-graph arena handles: each annotated edge has a link, a constant term +/// composes to it, and its object decomposes with its datatype; a decimal's +/// joins the asserted edge. +async fn assert_arena_links(fluree: &fluree_db_api::Fluree, ledger: &LedgerState) { + let got = links(fluree, ledger).await; + let term = |reifier: &str| -> String { + got.iter() + .find(|r| r[0].ends_with(reifier)) + .unwrap_or_else(|| panic!("no link for {reifier}: {got:#?}"))[1] + .clone() + }; + assert!(term("c1").contains("1.5"), "{got:#?}"); + assert!( + term("c2").contains("123456789012345678901234567890"), + "{got:#?}" + ); + assert!( + term("c3").contains("0.5") && term("c3").contains("-1.25"), + "{got:#?}" + ); + + let run = |body: &str| run_link_query(fluree, ledger, body.to_string()); + for (edge, source) in [ + ("ex:acct ex:balance 1.5", "ledger"), + ("?s ex:balance 1.50", "ledger"), + ( + "ex:acct ex:serial 123456789012345678901234567890", + "registry", + ), + ( + "?s ex:embedding \"[0.5, -1.25]\"^^", + "model", + ), + ] { + let got = run(&format!( + "SELECT ?src WHERE {{ << {edge} >> ex:source ?src }}" + )) + .await; + if edge.starts_with("ex:acct ") { + assert_eq!(got.len(), 1, "{edge}: {got:#?}"); + } + assert!( + got.iter().any(|r| r[0].ends_with(source)), + "{edge}: {got:#?}" + ); + } + + for (p, dt) in [ + ("balance", "decimal"), + ("serial", "integer"), + ("embedding", "embeddingVector"), + ] { + let got = run(&format!( + "SELECT (DATATYPE(?o) AS ?dt) WHERE {{ << ex:acct ex:{p} ?o >> ex:source ?src }}" + )) + .await; + assert_eq!(got.len(), 1, "{p}: {got:#?}"); + assert!(got[0][0].ends_with(dt), "{p}: {got:#?}"); + } + for body in [ + "SELECT ?src WHERE { << ex:acct ex:balance ?o >> ex:source ?src . ex:acct ex:balance ?o }", + "SELECT ?src WHERE { ex:acct ex:balance ?o . << ex:acct ex:balance ?o >> ex:source ?src }", + ] { + let got = run(body).await; + assert_eq!(got, vec![vec!["ex:ledger".to_string()]], "{body}: {got:#?}"); + } +} + +/// Import interns these terms through the chunk term table, a rebuild +/// through the resolver's attachment replay. The first file is its own chunk +/// with strings that sort before the forms, so the second chunk's local +/// string ids are not the global ones. +#[tokio::test] +async fn arena_kind_objects_link_through_import_and_reindex() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let alias = "it/triple-term-links:arena"; + let pad = "@prefix ex: .\nex:pad ex:label \"!pad\" , \"0pad\" .\n"; + let (fluree, ledger) = import(&[("a.ttl", pad), ("b.ttl", ARENA_CLAIMS)], alias).await; + assert_arena_links(&fluree, &ledger).await; + + // A term only novelty holds whose object form the index already interned. + let ledger = fluree + .insert_turtle( + ledger, + "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:acct2 ex:balance 1.5 ~ ex:c4 {| ex:source ex:late |} .\n", + ) + .await + .expect("unindexed claim") + .ledger; + let got = run_link_query( + &fluree, + &ledger, + "SELECT ?src WHERE { << ?s ex:balance 1.5 >> ex:source ?src } ORDER BY ?src".to_string(), + ) + .await; + assert_eq!( + got, + vec![vec!["ex:late".to_string()], vec!["ex:ledger".to_string()]], + "{got:#?}" + ); + + fluree + .reindex(alias, fluree_db_api::ReindexOptions::default()) + .await + .expect("reindex"); + let ledger = fluree.ledger(alias).await.expect("reload after reindex"); + assert_eq!(ledger.t(), ledger.index_t()); + assert_arena_links(&fluree, &ledger).await; + assert_eq!(links(&fluree, &ledger).await.len(), 4); +} + +/// A ledger never indexed holds these links in novelty. +#[tokio::test] +async fn arena_kind_objects_link_in_novelty() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let fluree = FlureeBuilder::memory().build_memory(); + let ledger = support::genesis_ledger(&fluree, "it/triple-term-links:arena-novelty"); + let ledger = fluree + .insert_turtle(ledger, ARENA_CLAIMS) + .await + .expect("claims") + .ledger; + assert_eq!(ledger.index_t(), 0); + assert_arena_links(&fluree, &ledger).await; +} + +/// Incremental builds over these objects, in the default graph and a named +/// one. The base index holds each attachment's object as an arena handle; +/// a re-point must still retract the term the base linked (an object +/// change) and carry the base object into the new term (a subject change). +#[tokio::test] +async fn arena_kind_links_follow_incremental_repoints_in_every_graph() { + use fluree_db_indexer::IndexerConfig; + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + + let fluree = FlureeBuilder::memory() + .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) + .build_memory(); + let ledger_id = "it/triple-term-links:arena-incremental"; + let (local, handle) = + support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let trig = |body: &str| { + format!( + "VERSION \"1.2\"\n@prefix ex: .\n\ + @prefix f: .\n{body}\n" + ) + }; + let audit_links = |ledger: LedgerState| { + let fluree = &fluree; + async move { + let result = support::query_sparql_formatted( + fluree, + &ledger, + "PREFIX rdf: \n\ + SELECT ?r ?t WHERE { GRAPH { ?r rdf:reifies ?t } }", + ) + .await + .expect("named-graph link query"); + rows(&result) + } + }; + + local + .run_until(async { + fluree.create_ledger(ledger_id).await.expect("create"); + let commit = |body: String| { + let fluree = &fluree; + async move { + fluree + .graph(ledger_id) + .transact() + .upsert_turtle(&body) + .commit() + .await + .expect("commit claims") + .receipt + .t + } + }; + let first = commit(trig( + "ex:acct ex:balance 1.50 ~ ex:c1 {| ex:source ex:ledger |} .\n\ + ex:acct ex:embedding \"[0.5, -1.25]\"^^f:embeddingVector \ + ~ ex:c3 {| ex:source ex:model |} .\n\ + GRAPH ex:audit { ex:acct ex:balance 1.5 ~ ex:c2 {| ex:source ex:audit |} . }", + )) + .await; + support::trigger_index_and_wait(&handle, ledger_id, first).await; + support::wait_for_index_application(&fluree, ledger_id, first).await; + let ledger1 = fluree.ledger(ledger_id).await.expect("reload"); + assert_eq!(ledger1.t(), ledger1.index_t()); + let got = links(&fluree, &ledger1).await; + assert_eq!(got.len(), 2, "{got:#?}"); + let got = audit_links(ledger1).await; + assert_eq!(got.len(), 1, "{got:#?}"); + assert!(got[0][1].contains("1.5"), "{got:#?}"); + + let second = commit(trig( + "ex:acct ex:balance 2.25 ~ ex:c1 .\n\ + ex:other ex:embedding \"[0.5, -1.25]\"^^f:embeddingVector ~ ex:c3 .\n\ + GRAPH ex:audit { ex:acct ex:balance 3.75 ~ ex:c2 . }", + )) + .await; + support::trigger_index_and_wait(&handle, ledger_id, second).await; + support::wait_for_index_application(&fluree, ledger_id, second).await; + let ledger2 = fluree.ledger(ledger_id).await.expect("reload"); + assert_eq!(ledger2.t(), ledger2.index_t()); + + let got = links(&fluree, &ledger2).await; + assert_eq!(got.len(), 2, "one live link per reifier: {got:#?}"); + let term = |reifier: &str| -> String { + got.iter() + .find(|r| r[0].ends_with(reifier)) + .unwrap_or_else(|| panic!("no link for {reifier}: {got:#?}"))[1] + .clone() + }; + assert!( + term("c1").contains("2.25") && !term("c1").contains("1.5"), + "{got:#?}" + ); + assert!( + term("c3").contains("other") && term("c3").contains("-1.25"), + "{got:#?}" + ); + let got = audit_links(ledger2.clone()).await; + assert_eq!(got.len(), 1, "{got:#?}"); + assert!( + got[0][1].contains("3.75") && !got[0][1].contains("1.5"), + "{got:#?}" + ); + + let got = run_link_query( + &fluree, + &ledger2, + "SELECT ?src WHERE { << ex:other ex:embedding \ + \"[0.5, -1.25]\"^^ >> ex:source ?src }" + .to_string(), + ) + .await; + assert_eq!(got, vec![vec!["ex:model".to_string()]], "{got:#?}"); + }) + .await; +} diff --git a/fluree-db-binary-index/src/read/binary_index_store.rs b/fluree-db-binary-index/src/read/binary_index_store.rs index f461c86b5a..265cd5badf 100644 --- a/fluree-db-binary-index/src/read/binary_index_store.rs +++ b/fluree-db-binary-index/src/read/binary_index_store.rs @@ -2272,11 +2272,36 @@ impl BinaryIndexStore { } } + /// A term key's object. Every kind the main index keys graph-wide decodes + /// as it does there; a decimal, big-integer or vector object is keyed by + /// the string id of its canonical form instead of an arena handle. + pub fn decode_term_object( + &self, + key: &fluree_db_core::triple_term::TermKey, + ) -> io::Result { + use fluree_db_core::triple_term::{is_lexical_term_object, parse_lexical_term_object}; + if !is_lexical_term_object(key.o_type) { + return self.decode_value_v3( + key.o_type.as_u16(), + key.o_key, + key.p_id, + fluree_db_core::DEFAULT_GRAPH_ID, + ); + } + let form = self.resolve_string_value(key.o_key as u32)?; + parse_lexical_term_object(key.o_type, &form).ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + format!( + "term object form {form:?} is not a {:#06x} value", + key.o_type.as_u16() + ), + ) + }) + } + /// Materialize a triple-term handle: the base edge's subject, predicate /// and object as SIDs and a value, with the object's datatype and tag. - /// - /// Terms are graph-independent, so a per-graph object arena (NumBig) is - /// read through the default graph. fn decode_triple_term(&self, handle: u64) -> io::Result { let key = self.resolve_term_key(handle)?.ok_or_else(|| { io::Error::new( @@ -2295,12 +2320,7 @@ impl BinaryIndexStore { ) })?; let o_type = key.o_type.as_u16(); - let o = self.decode_value_v3( - o_type, - key.o_key, - key.p_id, - fluree_db_core::DEFAULT_GRAPH_ID, - )?; + let o = self.decode_term_object(&key)?; let dt = self .resolve_datatype_sid_for_value(o_type, &o) .ok_or_else(|| { diff --git a/fluree-db-core/src/commit/codec/raw_reader.rs b/fluree-db-core/src/commit/codec/raw_reader.rs index 4f02585097..4749448178 100644 --- a/fluree-db-core/src/commit/codec/raw_reader.rs +++ b/fluree-db-core/src/commit/codec/raw_reader.rs @@ -162,6 +162,7 @@ pub struct RawOp<'a> { } /// Object value without allocation. Borrows from ops buffer or dicts. +#[derive(Clone)] pub enum RawObject<'a> { /// IRI reference: namespace code + local name from object_ref dict. Ref { ns_code: u16, name: &'a str }, diff --git a/fluree-db-core/src/link.rs b/fluree-db-core/src/link.rs index 679ae39c09..5423e3745a 100644 --- a/fluree-db-core/src/link.rs +++ b/fluree-db-core/src/link.rs @@ -75,18 +75,6 @@ pub fn is_attachment_slot(p: &Sid) -> bool { is_reifies_subject(p) || is_reifies_predicate(p) || is_reifies_object(p) } -/// Whether a term with this object can be interned. Objects keyed by a -/// per-(graph, predicate) arena have no graph-independent identity, so the -/// index build gives their attachments no link, and novelty must not either. -#[inline] -pub fn term_object_is_internable(o: &FlakeValue) -> bool { - match o { - FlakeValue::Decimal(_) | FlakeValue::Vector(_) => false, - FlakeValue::BigInt(b) => num_traits::ToPrimitive::to_i64(b.as_ref()).is_some(), - _ => true, - } -} - impl AttachmentSlots { /// The triple term a complete attachment names. pub fn term(&self) -> Option { @@ -95,9 +83,6 @@ impl AttachmentSlots { else { return None; }; - if !term_object_is_internable(o) { - return None; - } Some(TripleTermValue { s: s.clone(), p: p.clone(), @@ -340,18 +325,4 @@ mod tests { flakes.push(tagged); assert_eq!(links(&mut state, &flakes, 0).len(), 1); } - - #[test] - fn arena_kind_objects_get_no_term() { - let state = AttachmentSlots { - subject: Some(sid("a")), - predicate: Some(sid("p")), - object: Some(( - FlakeValue::Decimal(Box::new("1.5".parse().unwrap())), - Sid::new(2, "decimal"), - None, - )), - }; - assert!(state.term().is_none()); - } } diff --git a/fluree-db-core/src/triple_term.rs b/fluree-db-core/src/triple_term.rs index 641da9ab93..9a35d7a43e 100644 --- a/fluree-db-core/src/triple_term.rs +++ b/fluree-db-core/src/triple_term.rs @@ -10,7 +10,8 @@ //! The split is a deliberate constant: 32 bits of sequence per predicate, //! 32 bits of predicate id. Both limits are enforced at allocation. -use crate::o_type::OType; +use crate::o_type::{DecodeKind, OType}; +use crate::value::FlakeValue; /// Bits of per-predicate sequence in a handle. pub const TERM_SEQ_BITS: u32 = 32; @@ -69,8 +70,85 @@ pub const fn term_handle_range(inner_p_id: u32) -> (u64, u64) { ) } +/// Whether a term key's object is keyed by [`lexical_term_object`]: an `o_key` +/// that is a string-dictionary id, where the main index holds an arena handle. +#[inline] +pub const fn is_lexical_term_object(o_type: OType) -> bool { + matches!( + o_type.decode_kind(), + DecodeKind::NumBigArena | DecodeKind::VectorArena + ) +} + +/// The `o_type` and canonical form that key a decimal, big-integer or vector +/// object in a term. The main index keys these by arena handles scoped to a +/// graph and predicate, which name nothing across graphs, so a term keys them +/// by the string-dictionary id of this form instead. `None` for every other +/// object, and for an integer that fits `i64` (keyed inline). +/// +/// Equal values share a form: a decimal is normalized as the main index's +/// arena normalizes it, and a vector is read at the `f32` precision ingest +/// quantizes it to. A decimal's form always carries an exponent and a big +/// integer's never does, so the two cannot collide under one `o_type`. +pub fn lexical_term_object(o: &FlakeValue) -> Option<(OType, String)> { + match o { + FlakeValue::Decimal(d) => { + let (unscaled, scale) = d.normalized().as_bigint_and_exponent(); + Some(( + OType::NUM_BIG_OVERFLOW, + format!("{unscaled}e{}", -i128::from(scale)), + )) + } + FlakeValue::BigInt(b) if num_traits::ToPrimitive::to_i64(b.as_ref()).is_none() => { + Some((OType::NUM_BIG_OVERFLOW, b.to_string())) + } + FlakeValue::Vector(v) => { + let elements: Vec = v + .iter() + // -0.0 equals 0.0 as a vector element; one form for both. + .map(|x| (*x as f32 + 0.0).to_string()) + .collect(); + Some((OType::VECTOR, format!("[{}]", elements.join(",")))) + } + _ => None, + } +} + +/// The object a [`lexical_term_object`] form names; `None` for a malformed +/// form or an `o_type` that is not keyed that way. +pub fn parse_lexical_term_object(o_type: OType, form: &str) -> Option { + match o_type.decode_kind() { + DecodeKind::NumBigArena => match form.split_once('e') { + Some((unscaled, exp)) => { + let unscaled: num_bigint::BigInt = unscaled.parse().ok()?; + let exp: i128 = exp.parse().ok()?; + Some(FlakeValue::Decimal(Box::new(bigdecimal::BigDecimal::new( + unscaled, + i64::try_from(-exp).ok()?, + )))) + } + None => Some(FlakeValue::BigInt(Box::new(form.parse().ok()?))), + }, + DecodeKind::VectorArena => { + let inner = form.strip_prefix('[')?.strip_suffix(']')?; + let elements = if inner.is_empty() { + Vec::new() + } else { + inner + .split(',') + .map(|x| x.parse::().ok().map(f64::from)) + .collect::>>()? + }; + Some(FlakeValue::Vector(elements.into())) + } + _ => None, + } +} + /// The encoded identity of a triple term: the base edge\'s `(s_id, p_id, -/// o_type, o_key)` as the main index stores it. No graph, no list index. +/// o_type, o_key)` as the main index stores it, except that a decimal, +/// big-integer or vector object is keyed by [`lexical_term_object`]. No graph, +/// no list index. /// /// Its big-endian byte form is the reverse-tree key, ordered subject-first so /// a subject-bound reified-triple pattern is one key range. @@ -146,6 +224,46 @@ mod tests { ); } + #[test] + fn lexical_forms_identify_values_and_round_trip() { + let decimal = |s: &str| FlakeValue::Decimal(Box::new(s.parse().unwrap())); + let form = |o: &FlakeValue| lexical_term_object(o).expect("arena kind").1; + + assert_eq!(form(&decimal("1.5")), form(&decimal("1.50"))); + assert_ne!(form(&decimal("1.5")), form(&decimal("15"))); + let big: num_bigint::BigInt = "123456789012345678901234567890".parse().unwrap(); + let big = FlakeValue::BigInt(Box::new(big)); + let integral = decimal("123456789012345678901234567890"); + assert_ne!( + form(&big), + form(&integral), + "decimal and integer stay apart" + ); + assert!(lexical_term_object(&FlakeValue::BigInt(Box::new(7.into()))).is_none()); + assert!(lexical_term_object(&FlakeValue::Long(7)).is_none()); + + let vector = |v: &[f64]| FlakeValue::Vector(v.to_vec().into()); + assert_eq!(form(&vector(&[-0.0, 1.5])), form(&vector(&[0.0, 1.5]))); + + for o in [ + decimal("1.50"), + decimal("-0.000123"), + decimal("1E+30"), + decimal("0"), + big, + vector(&[0.1f32 as f64, -2.5, 1e-30f32 as f64]), + ] { + let (o_type, s) = lexical_term_object(&o).unwrap(); + assert!(is_lexical_term_object(o_type)); + let back = parse_lexical_term_object(o_type, &s).unwrap(); + assert_eq!(back, o, "{s}"); + assert_eq!(std::mem::discriminant(&back), std::mem::discriminant(&o)); + assert_eq!(lexical_term_object(&back).unwrap().1, s); + } + assert!(!is_lexical_term_object(OType::XSD_STRING)); + assert!(parse_lexical_term_object(OType::XSD_STRING, "1").is_none()); + } + #[test] fn key_roundtrip_orders_subject_first() { let k = TermKey { diff --git a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs index efa2128f55..7553cd287b 100644 --- a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs +++ b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs @@ -617,7 +617,7 @@ pub async fn resolve_incremental_commits_v6( // 7. Reconcile chunk-local IDs to global IDs (same algorithm as V5). let t_reconcile_start = Instant::now(); - let reconcile = reconcile_chunk_to_global( + let mut reconcile = reconcile_chunk_to_global( &chunk, &subject_tree, &string_tree, @@ -692,6 +692,11 @@ pub async fn resolve_incremental_commits_v6( &shared.predicates, shared.link_synth.slots(), &keys, + &mut WindowStrings { + new: &mut reconcile.new_strings, + watermark: &mut reconcile.updated_string_watermark, + index: None, + }, ) .await .map_err(IncrementalResolveError::Io)? @@ -1287,10 +1292,12 @@ async fn base_attachment_states( predicates: &crate::run_index::resolve::global_dict::PredicateDict, slots: [Option; 3], keys: &[(u16, u64)], + strings: &mut WindowStrings<'_>, ) -> io::Result; 3]>> { use crate::run_index::resolve::link_synth::ObjectId; use fluree_db_binary_index::read::binary_index_store::BinaryIndexStore; use fluree_db_core::o_type::OType; + use fluree_db_core::triple_term::{is_lexical_term_object, lexical_term_object}; let mut states = HashMap::new(); if keys.is_empty() || slots.iter().all(Option::is_none) { @@ -1347,6 +1354,25 @@ async fn base_attachment_states( }; Some(SlotValue::Predicate(p_id)) } + // An arena handle names the value only within this graph + // and slot predicate; the window's ops carry the form. + 2 if is_lexical_term_object(OType::from_u16(o_type)) => { + let value = store.decode_value_v3(o_type, o_key, p_id, g_id)?; + let (_, form) = lexical_term_object(&value).ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + format!("reifier {ann}: arena object {value:?} has no term form"), + ) + })?; + let id = match store.find_string_id(&form)? { + Some(id) => id, + None => strings.intern(form), + }; + Some(SlotValue::Object(ObjectId::Typed { + o_type, + o_key: u64::from(id), + })) + } 2 => Some(SlotValue::Object(ObjectId::Typed { o_type, o_key })), _ => None, }; @@ -1357,6 +1383,33 @@ async fn base_attachment_states( Ok(states) } +/// The strings this window adds to the dictionary, which the canonical +/// object form of a base attachment can still join after reconciliation. +struct WindowStrings<'a> { + new: &'a mut Vec<(u32, Vec)>, + watermark: &'a mut u32, + index: Option, u32>>, +} + +impl WindowStrings<'_> { + /// The id `form` has in this window, allocating one above the watermark. + fn intern(&mut self, form: String) -> u32 { + let new = &*self.new; + let index = self + .index + .get_or_insert_with(|| new.iter().map(|(id, b)| (b.clone(), *id)).collect()); + if let Some(&id) = index.get(form.as_bytes()) { + return id; + } + *self.watermark += 1; + let id = *self.watermark; + let bytes = form.into_bytes(); + index.insert(bytes.clone(), id); + self.new.push((id, bytes)); + id + } +} + /// Fetch one commit blob, honoring the optional artifact cache. async fn fetch_commit_bytes( cs: &dyn ContentStore, @@ -2212,4 +2265,20 @@ mod tests { ]) ); } + + #[test] + fn a_base_object_form_joins_the_window_strings() { + let mut new = vec![(11, b"15e-1".to_vec())]; + let mut watermark = 11; + let mut strings = WindowStrings { + new: &mut new, + watermark: &mut watermark, + index: None, + }; + assert_eq!(strings.intern("15e-1".into()), 11); + assert_eq!(strings.intern("225e-2".into()), 12); + assert_eq!(strings.intern("225e-2".into()), 12); + assert_eq!(watermark, 12); + assert_eq!(new, vec![(11, b"15e-1".to_vec()), (12, b"225e-2".to_vec())]); + } } diff --git a/fluree-db-indexer/src/run_index/resolve/link_synth.rs b/fluree-db-indexer/src/run_index/resolve/link_synth.rs index ff93d12c4f..24b327e383 100644 --- a/fluree-db-indexer/src/run_index/resolve/link_synth.rs +++ b/fluree-db-indexer/src/run_index/resolve/link_synth.rs @@ -20,18 +20,19 @@ use super::global_dict::PredicateDict; use super::resolver::RebuildChunk; use fluree_db_binary_index::format::run_record::{RunRecord, LIST_INDEX_NONE}; use fluree_db_core::commit::codec::raw_reader::{RawObject, RawOp}; -use fluree_db_core::o_type::{DecodeKind, OType}; +use fluree_db_core::o_type::OType; use fluree_db_core::o_type_registry::OTypeRegistry; use fluree_db_core::subject_id::SubjectId; -use fluree_db_core::triple_term::TermKey; +use fluree_db_core::triple_term::{lexical_term_object, TermKey}; use fluree_db_core::value_id::{ObjKey, ObjKind}; -use fluree_db_core::DatatypeDictId; +use fluree_db_core::{DatatypeDictId, FlakeValue}; use fluree_vocab::{db, fluree}; use std::collections::HashMap; use std::io; /// The base edge's object, as the resolver saw it or as the base index -/// stores it. The two compare through [`ObjectId::typed`]. +/// stores it. The two compare through [`ObjectId::typed`]; on both sides an +/// arena kind's key is the string id of its canonical form. #[derive(Debug, Clone, Copy)] pub enum ObjectId { /// Kind, key, datatype and tag of a resolved op; the `o_type` needs the @@ -128,7 +129,8 @@ impl AttachmentOp { let kind = ObjKind::from_u8(*o_kind); if kind == ObjKind::REF_ID { *o_key = subject(*o_key)?; - } else if kind == ObjKind::LEX_ID || kind == ObjKind::JSON_ID { + } else if kind == ObjKind::LEX_ID || kind == ObjKind::JSON_ID || is_arena_kind(kind) + { let local = ObjKey::from_u64(*o_key).decode_u32_id() as usize; let global = *str_remap .get(local) @@ -238,12 +240,25 @@ impl LinkSynth { .unwrap_or(""); SlotValue::Predicate(predicates.get_or_insert_parts(prefix, name)) } - _ => SlotValue::Object(ObjectId::Raw { - o_kind: record.o_kind, - o_key: record.o_key, - dt: record.dt, - lang_id: record.lang_id, - }), + _ => { + let mut o_key = record.o_key; + if is_arena_kind(ObjKind::from_u8(record.o_kind)) { + let Some((_, form)) = FlakeValue::try_from(raw.o.clone()) + .ok() + .and_then(|o| lexical_term_object(&o)) + else { + return; + }; + o_key = ObjKey::encode_u32_id(chunk.strings.get_or_insert(form.as_bytes())) + .as_u64(); + } + SlotValue::Object(ObjectId::Raw { + o_kind: record.o_kind, + o_key, + dt: record.dt, + lang_id: record.lang_id, + }) + } }; if self.link.is_none() { let p_id = predicates.get_or_insert(fluree_vocab::rdf::REIFIES); @@ -268,8 +283,6 @@ impl LinkSynth { } /// The term a complete attachment names, or `None` while a slot is missing. -/// An object in a per-(graph, predicate) arena has no graph-independent -/// identity and gets no term; import skips those edges too. fn term_of(state: &[Option; 3], registry: &OTypeRegistry) -> Option { let ( Some(SlotValue::Subject(s_id)), @@ -280,21 +293,21 @@ fn term_of(state: &[Option; 3], registry: &OTypeRegistry) -> Option bool { + kind == ObjKind::NUM_BIG || kind == ObjKind::VECTOR_ID +} + /// Replay every reifier's attachment ops (ids global) and hand its link /// records to `sink`: at each `t` where the attachment changes from or to a /// complete edge, a retract of the old term's link and an assert of the new diff --git a/fluree-db-indexer/src/run_index/runs/spool.rs b/fluree-db-indexer/src/run_index/runs/spool.rs index c4fa39b7a5..f49e26c300 100644 --- a/fluree-db-indexer/src/run_index/runs/spool.rs +++ b/fluree-db-indexer/src/run_index/runs/spool.rs @@ -1290,10 +1290,14 @@ impl TermRemapCtx { lang_remap, Some(self), )?; + let o_type = fluree_db_core::o_type::OType::from_u16(term.o_type); + if fluree_db_core::triple_term::is_lexical_term_object(o_type) { + term.o_key = u64::from(string_remap.get(term.o_key as usize)?); + } let key = fluree_db_core::triple_term::TermKey { s_id: term.s_id.as_u64(), p_id: term.p_id, - o_type: fluree_db_core::o_type::OType::from_u16(term.o_type), + o_type, o_key: term.o_key, }; self.builder @@ -1815,9 +1819,18 @@ pub fn sort_remap_and_write_sorted_commit( } } // Term-table entries keep their ordinal order (link records address them - // by position), so they are remapped but never sorted. + // by position), so they are remapped but never sorted. A decimal, + // big-integer or vector object there holds a string id, not an arena + // handle. + use fluree_db_core::value_id::{ObjKey, ObjKind}; for term in &mut terms { remap_record(term, &subject_remap, &string_remap)?; + let kind = ObjKind::from_u8(term.o_kind); + if kind == ObjKind::NUM_BIG || kind == ObjKind::VECTOR_ID { + let local = ObjKey::from_u64(term.o_key).decode_u32_id() as usize; + term.o_key = + ObjKey::encode_u32_id(StringRemap::get(&string_remap[..], local)?).as_u64(); + } } // A.2 step 4: Sort records by the V2-native graph-prefixed SPOT key without diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index d86e277315..291573d4a2 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -4538,7 +4538,7 @@ fn encode_term_key_filter( Some(tag.as_ref()), ), }; - let (ot, key) = value_to_otype_okey(value, &dt, lang, store, dict_novelty, None)?; + let (ot, key) = term_object_key(value, &dt, lang, store, dict_novelty)?; Some((ot.as_u16(), key)) } None => None, @@ -4546,6 +4546,28 @@ fn encode_term_key_filter( Ok(TermKeyFilter { s_id, o }) } +/// A term's object as a term key holds it. A decimal, big-integer or vector +/// object is keyed by the persisted string id of its canonical form; a form +/// the index never interned names no indexed term, so it is `NotFound`. +pub(crate) fn term_object_key( + value: &FlakeValue, + dt: &Sid, + lang: Option<&str>, + store: &BinaryIndexStore, + dict_novelty: Option<&Arc>, +) -> std::io::Result<(OType, u64)> { + let Some((o_type, form)) = fluree_db_core::triple_term::lexical_term_object(value) else { + return value_to_otype_okey(value, dt, lang, store, dict_novelty, None); + }; + match store.find_string_id(&form)? { + Some(id) => Ok((o_type, u64::from(id))), + None => Err(std::io::Error::new( + std::io::ErrorKind::NotFound, + "term object form is not interned", + )), + } +} + /// A constant triple term's handle: the term dictionary's, else the /// provisional handle dictionary novelty gives a term the index has not /// interned. A term neither holds is `NotFound`: no link names it, so the @@ -4586,14 +4608,8 @@ fn persisted_term_handle( return Ok(None); }; let s_id = resolve_subject_v3(&term.s, store, dict_novelty)?; - let (o_type, o_key) = value_to_otype_okey( - &term.o, - &term.dt, - term.lang.as_deref(), - store, - dict_novelty, - None, - )?; + let (o_type, o_key) = + term_object_key(&term.o, &term.dt, term.lang.as_deref(), store, dict_novelty)?; store.find_term_handle(&fluree_db_core::triple_term::TermKey { s_id, p_id, @@ -4618,20 +4634,14 @@ pub(crate) fn term_key_for_handle( return Ok(None); }; let encoded = resolve_subject_v3(&term.s, store, dict_novelty).and_then(|s_id| { - value_to_otype_okey( - &term.o, - &term.dt, - term.lang.as_deref(), - store, - dict_novelty, - None, + term_object_key(&term.o, &term.dt, term.lang.as_deref(), store, dict_novelty).map( + |(o_type, o_key)| TermKey { + s_id, + p_id: term_handle_p_id(handle), + o_type, + o_key, + }, ) - .map(|(o_type, o_key)| TermKey { - s_id, - p_id: term_handle_p_id(handle), - o_type, - o_key, - }) }); match encoded { Ok(key) => Ok(Some(key)), diff --git a/fluree-db-query/src/eval/rdf.rs b/fluree-db-query/src/eval/rdf.rs index 52d26642fb..dd9efcf585 100644 --- a/fluree-db-query/src/eval/rdf.rs +++ b/fluree-db-query/src/eval/rdf.rs @@ -539,18 +539,21 @@ pub(crate) fn term_object_binding( ctx: &ExecutionContext<'_>, ) -> Result { let o_type = key.o_type.as_u16(); - if let Some(b) = - late_materialized_object_binding(o_type, key.o_key, key.p_id, t, u32::MAX, None) - { - return Ok(b); + // An arena kind's key is not the main index's, so it binds decoded. + if !fluree_db_core::triple_term::is_lexical_term_object(key.o_type) { + if let Some(b) = + late_materialized_object_binding(o_type, key.o_key, key.p_id, t, u32::MAX, None) + { + return Ok(b); + } } let store = ctx .binary_store .as_deref() .ok_or_else(|| QueryError::Internal("term object decode without a store".into()))?; let val = store - .decode_value_v3(o_type, key.o_key, key.p_id, ctx.binary_g_id) - .map_err(|e| QueryError::from_io("decode_value_v3", e))?; + .decode_term_object(key) + .map_err(|e| QueryError::from_io("decode_term_object", e))?; Ok(materialized_object_binding( store, o_type, diff --git a/fluree-db-query/src/term_components.rs b/fluree-db-query/src/term_components.rs index fab433478b..39616c1caf 100644 --- a/fluree-db-query/src/term_components.rs +++ b/fluree-db-query/src/term_components.rs @@ -218,13 +218,12 @@ impl TermComponentsOperator { Some(tag.as_ref()), ), }; - match missing(crate::binary_scan::value_to_otype_okey( + match missing(crate::binary_scan::term_object_key( value, &dt, lang, store, dict_novelty, - None, ))? { Some((ot, key)) => Some((ot.as_u16(), key)), None => return Ok(None), diff --git a/fluree-db-transact/src/import_sink.rs b/fluree-db-transact/src/import_sink.rs index ee19ef313f..49e73ac713 100644 --- a/fluree-db-transact/src/import_sink.rs +++ b/fluree-db-transact/src/import_sink.rs @@ -577,11 +577,11 @@ mod inner { /// base edge. /// /// The base edge becomes a pseudo-record in the chunk's term table, - /// resolved exactly as the base triple's own record was (the object - /// under the inner predicate, so per-predicate arena handles agree). - /// The link record's `o_key` is that entry's ordinal; the build remaps - /// the entry to global ids and interns it, replacing the ordinal with - /// the term handle. + /// resolved as the base triple's own record was, except that a + /// decimal, big-integer or vector object holds the string id of its + /// canonical form. The link record's `o_key` is that entry's ordinal; + /// the build remaps the entry to global ids and interns it, replacing + /// the ordinal with the term handle. /// /// `object` is the bundle's `f:reifiesObject` flake: its subject is /// the reifier and its object, datatype and tag are the base edge's. @@ -596,17 +596,23 @@ mod inner { let s_id = self.assign_subject_id(s); let p_id = self.assign_predicate_id(p); let dt_id = self.assign_datatype_id(&object.dt)?; - let Some((o_kind, o_key)) = self.resolve_object_value(&object.o, p_id) else { - return Ok(()); + // An arena handle names a value only within one graph and + // predicate; a term keys the object by its canonical form. + let resolved = match fluree_db_core::triple_term::lexical_term_object(&object.o) { + Some((o_type, form)) => { + let o_kind = if o_type == fluree_db_core::o_type::OType::VECTOR { + ObjKind::VECTOR_ID + } else { + ObjKind::NUM_BIG + }; + let id = self.assign_string_id(&form); + Some((o_kind.as_u8(), ObjKey::encode_u32_id(id).as_u64())) + } + None => self.resolve_object_value(&object.o, p_id), }; - // Per-(graph, predicate) arena handles are not a graph-independent - // object identity; rebuild skips these bundles too. - if matches!( - ObjKind::from_u8(o_kind), - ObjKind::NUM_BIG | ObjKind::VECTOR_ID - ) { + let Some((o_kind, o_key)) = resolved else { return Ok(()); - } + }; let lang_id = object .m .as_ref() From 8c7217637f0cccae53f428de940ff5338e06c6e8 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 21:20:27 -0400 Subject: [PATCH 33/92] feat(index): live link counts per inner predicate, for the planner MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A link scan pinned to one inner predicate reads only that predicate's handle interval, but the planner ranked it at the total rdf:reifies count, so a small interval lost to a larger scan: annotation benchmark S7 drove from rdf:type and hash-probed every derives_from row (545 ms against the bundle chain's 200 ms). - The stats hook counts rows whose object is a triple term by predicate and the handle's inner predicate, with the same signed deltas as property counts. Rebuild, import and incremental builds roll the rdf:reifies rows up into IndexStats.links, keyed by the inner predicate's SID. An incremental build seeds them from the base; a base that predates them leaves them unknown. The import's cross-chunk duplicate discount keys link rows by inner predicate too. - The counts travel in a new stats tail section (tag 3), after the historical and class tails, which older readers stop before. - The novelty merge moves them by the inner predicate of each link flake; a ledger never indexed counts from none. - The planner reads `sameTerm(PREDICATE(?t),

)` filters and ranks an unbound link scan on ?t at p's live links; explain shows the same estimate. Dev slice: S7 545 → 184 ms, S8 462 → 111 ms, S9 463 → 102 ms, the rest of the sweep level, every answer unchanged. --- fluree-db-api/src/explain.rs | 11 +- fluree-db-api/src/import.rs | 16 ++ fluree-db-api/tests/it_triple_term_links.rs | 88 +++++++++ .../src/format/index_root.rs | 1 + .../src/format/stats_wire.rs | 57 ++++++ fluree-db-core/src/index_stats.rs | 17 ++ fluree-db-core/src/lib.rs | 2 +- fluree-db-core/src/stats_view.rs | 22 +++ fluree-db-core/src/stats_wire.rs | 58 +++++- fluree-db-indexer/src/build/incremental.rs | 31 ++++ fluree-db-indexer/src/build/rebuild.rs | 11 ++ .../src/run_index/build/build_from_commits.rs | 29 ++- fluree-db-indexer/src/stats/id_hook.rs | 115 +++++++++++- fluree-db-indexer/src/stats/mod.rs | 24 +++ fluree-db-novelty/src/runtime_stats.rs | 129 ++++++++++++- fluree-db-policy/src/index.rs | 2 + fluree-db-query/src/explain.rs | 10 +- fluree-db-query/src/fast_count.rs | 2 + fluree-db-query/src/planner.rs | 173 +++++++++++++++++- fluree-db-query/src/stats_cache.rs | 2 + 20 files changed, 767 insertions(+), 33 deletions(-) diff --git a/fluree-db-api/src/explain.rs b/fluree-db-api/src/explain.rs index b86d58c456..f32c557678 100644 --- a/fluree-db-api/src/explain.rs +++ b/fluree-db-api/src/explain.rs @@ -172,12 +172,13 @@ fn logical_node( vars: &VarRegistry, compactor: &IriCompactor, stats: Option<&StatsView>, + pins: &std::collections::HashMap, bound_vars: &HashSet, ) -> JsonValue { - use fluree_db_query::planner::{estimate_pattern, PatternEstimate}; + use fluree_db_query::planner::{estimate_in_group, PatternEstimate}; let mut node = Map::new(); - let category = match estimate_pattern(p, bound_vars, stats) { + let category = match estimate_in_group(p, pins, bound_vars, stats) { PatternEstimate::Source { row_count } => { node.insert( "estimate".into(), @@ -202,10 +203,11 @@ fn logical_node( // sees it). A fresh local per list means UNION branches each start from `bound_vars`. let children = |ps: &[Pattern]| -> JsonValue { let mut local = bound_vars.clone(); + let pins = fluree_db_query::planner::link_pins(ps); JsonValue::Array( ps.iter() .map(|c| { - let n = logical_node(c, vars, compactor, stats, &local); + let n = logical_node(c, vars, compactor, stats, &pins, &local); local.extend(c.produced_vars()); n }) @@ -591,11 +593,12 @@ fn explain_from_parsed( // Thread the evolving bound-var set through the ordered plan so each node's // estimate is context-aware (a bound-subject scan, not a full predicate scan). let mut bound: HashSet = HashSet::new(); + let pins = fluree_db_query::planner::link_pins(&ordered); JsonValue::Array( ordered .iter() .map(|p| { - let n = logical_node(p, vars, &compactor, stats_view.as_ref(), &bound); + let n = logical_node(p, vars, &compactor, stats_view.as_ref(), &pins, &bound); bound.extend(p.produced_vars()); n }) diff --git a/fluree-db-api/src/import.rs b/fluree-db-api/src/import.rs index 36761d78bb..a5806a524c 100644 --- a/fluree-db-api/src/import.rs +++ b/fluree-db-api/src/import.rs @@ -6883,6 +6883,21 @@ where } } + let reifies = fluree_db_core::rdf_reifies_sid(); + let reifies_p_id = predicate_sids_v6 + .iter() + .position(|(ns, name)| *ns == reifies.namespace_code && *name == *reifies.name) + .map(|p_id| p_id as u32); + let links = fluree_db_indexer::stats::link_stat_entries( + &id_stats.term_rows, + reifies_p_id, + |p_id| { + predicate_sids_v6 + .get(p_id as usize) + .cloned() + .unwrap_or((0u16, String::new())) + }, + ); let mut stats = is::IndexStats { flakes: id_stats.total_flakes, size: 0, @@ -6894,6 +6909,7 @@ where // record at every `t` — historical coverage is complete from // genesis. historical_since_t: Some(0), + links: Some(links), }; // Wire `total_commit_size` into `stats.size` and per-graph sizes, // mirroring `root_assembly::compose_root_v6` for the normal indexing diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 3b62fc4f8c..3e7dc1863a 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -1536,3 +1536,91 @@ async fn arena_kind_links_follow_incremental_repoints_in_every_graph() { }) .await; } + +/// Live links per inner predicate, as the index root's stats carry them. +fn link_counts(ledger: &LedgerState) -> Option> { + let links = ledger.snapshot.stats.as_ref()?.links.as_ref()?; + Some(links.iter().map(|l| (l.sid.1.clone(), l.count)).collect()) +} + +/// The link scan's planner estimate, read from the explain plan. +async fn link_scan_estimate(fluree: &fluree_db_api::Fluree, ledger: &LedgerState) -> i64 { + let db = fluree_db_api::GraphDb::from_ledger_state(ledger); + let plan = fluree + .explain_sparql( + &db, + "PREFIX ex: \n\ + SELECT ?src WHERE { << ?s ex:knows ?o >> ex:source ?src }", + ) + .await + .expect("explain"); + plan["plan"]["logical"] + .as_array() + .expect("logical plan") + .iter() + .find(|n| { + n["pattern"]["property"] + .as_str() + .is_some_and(|p| p.ends_with("#reifies")) + }) + .unwrap_or_else(|| panic!("no link scan: {plan:#}"))["estimate"]["row-count"] + .as_i64() + .expect("row count") +} + +/// Every build publishes live links per inner predicate, and the planner +/// ranks a link scan pinned to an inner predicate by its count: import and +/// reindex count from the whole history, an incremental build moves the +/// base's counts by the window's link asserts and retracts. +#[tokio::test] +async fn link_counts_follow_every_build() { + use fluree_db_indexer::IndexerConfig; + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let expected = |knows: u64| Some(vec![("age".to_string(), 1), ("knows".to_string(), knows)]); + + let alias = "it/triple-term-links:link-counts"; + // A second file restates claim1: the merge collapses the duplicate link + // row, and the count must not keep it. + let restated = "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:alice ex:knows ex:bob ~ ex:claim1 .\n"; + let (fluree, ledger) = import(&[("a.ttl", CLAIMS), ("b.ttl", restated)], alias).await; + assert_eq!(link_counts(&ledger), expected(3)); + assert_eq!(link_scan_estimate(&fluree, &ledger).await, 3); + fluree + .reindex(alias, fluree_db_api::ReindexOptions::default()) + .await + .expect("reindex"); + let ledger = fluree.ledger(alias).await.expect("reload"); + assert_eq!(link_counts(&ledger), expected(3)); + + let fluree = FlureeBuilder::memory() + .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) + .build_memory(); + let ledger_id = "it/triple-term-links:link-counts-incremental"; + let (local, handle) = + support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + local + .run_until(async { + let ledger = fluree + .insert_turtle(support::genesis_ledger(&fluree, ledger_id), CLAIMS) + .await + .expect("claims") + .ledger; + support::trigger_index_and_wait(&handle, ledger_id, ledger.t()).await; + support::wait_for_index_application(&fluree, ledger_id, ledger.t()).await; + let ledger = fluree.ledger(ledger_id).await.expect("reload"); + assert_eq!(link_counts(&ledger), expected(3)); + + // +2 knows (claim9, bob knows erin), -1 knows (the cascaded + // retract), and an age re-point that keeps age at one. + let ledger = change_claims_without_indexing(&fluree, ledger).await; + let t = ledger.t(); + support::trigger_index_and_wait(&handle, ledger_id, t).await; + support::wait_for_index_application(&fluree, ledger_id, t).await; + let ledger = fluree.ledger(ledger_id).await.expect("reload"); + assert_eq!(ledger.index_t(), t); + assert_eq!(link_counts(&ledger), expected(4)); + assert_eq!(link_scan_estimate(&fluree, &ledger).await, 4); + }) + .await; +} diff --git a/fluree-db-binary-index/src/format/index_root.rs b/fluree-db-binary-index/src/format/index_root.rs index bdc838db97..cd631a3a9d 100644 --- a/fluree-db-binary-index/src/format/index_root.rs +++ b/fluree-db-binary-index/src/format/index_root.rs @@ -1669,6 +1669,7 @@ mod tests { }]), }]), historical_since_t: Some(0), + links: None, }); root.schema = Some(IndexSchema { t: 4, diff --git a/fluree-db-binary-index/src/format/stats_wire.rs b/fluree-db-binary-index/src/format/stats_wire.rs index f1f219fed0..7264428373 100644 --- a/fluree-db-binary-index/src/format/stats_wire.rs +++ b/fluree-db-binary-index/src/format/stats_wire.rs @@ -197,6 +197,9 @@ pub fn encode_stats(stats: &IndexStats) -> Vec { if graphs_carry_classes { fluree_db_core::stats_wire::encode_class_tail(&mut buf, &sorted_graphs); } + if let Some(links) = &stats.links { + fluree_db_core::stats_wire::encode_link_tail(&mut buf, links); + } buf } @@ -584,6 +587,7 @@ mod tests { classes: None, graphs: None, historical_since_t: None, + links: None, }; let bytes = encode_stats(&stats); @@ -641,6 +645,7 @@ mod tests { }, ]), historical_since_t: None, + links: None, }; let bytes = encode_stats(&stats); @@ -689,6 +694,7 @@ mod tests { classes: None, graphs: None, historical_since_t: None, + links: None, }; let bytes = encode_stats(&stats); @@ -736,6 +742,7 @@ mod tests { }]), graphs: None, historical_since_t: None, + links: None, }; let bytes = encode_stats(&stats); @@ -859,6 +866,7 @@ mod tests { classes: None, graphs: Some(graphs.clone()), historical_since_t: Some(0), + links: None, }; let bytes = encode_stats(&stats); let (via_core, consumed) = fluree_db_core::stats_wire::decode_stats(&bytes).unwrap(); @@ -894,6 +902,7 @@ mod tests { classes: None, graphs: Some(class_tables()), historical_since_t: Some(0), + links: None, }; let mut without = with.clone(); for g in without.graphs.iter_mut().flatten() { @@ -985,6 +994,7 @@ mod tests { classes: union.clone(), graphs: Some(graphs), historical_since_t: None, + links: None, }; let imported = IndexStats { classes: None, @@ -1034,6 +1044,7 @@ mod tests { classes: None, graphs: None, historical_since_t: None, + links: None, }; let bytes1 = encode_stats(&stats); @@ -1056,6 +1067,7 @@ mod tests { classes: None, }]), historical_since_t: None, + links: None, }; let bytes = encode_stats(&stats); @@ -1096,6 +1108,7 @@ mod tests { classes: None, graphs: None, historical_since_t: None, + links: None, }; let bytes = encode_stats(&stats); let expected = vec![1u8, 7]; @@ -1161,6 +1174,7 @@ mod tests { classes: None, }]), historical_since_t: Some(2), + links: None, } } @@ -1247,6 +1261,49 @@ mod tests { ); } + /// Link counts round-trip in both decoders beside the other tails, and a + /// blob without them reads as unknown. + #[test] + fn link_tail_round_trips_after_the_other_tails() { + let mut stats = stats_with_historical(); + stats.links = Some(vec![ + fluree_db_core::LinkStatEntry { + sid: (15, "SEMNET_PART_OF".to_string()), + count: 231_170, + }, + fluree_db_core::LinkStatEntry { + sid: (15, "SEMNET_CAUSES".to_string()), + count: 0, + }, + ]); + let bytes = encode_stats(&stats); + let expected = vec![ + fluree_db_core::LinkStatEntry { + sid: (15, "SEMNET_CAUSES".to_string()), + count: 0, + }, + fluree_db_core::LinkStatEntry { + sid: (15, "SEMNET_PART_OF".to_string()), + count: 231_170, + }, + ]; + let (decoded, consumed) = decode_stats_with_len(&bytes).unwrap(); + assert_eq!(consumed, bytes.len()); + assert_eq!(decoded.links.as_ref(), Some(&expected)); + assert_eq!(decoded.historical_since_t, Some(2), "historical tail kept"); + let (via_core, _) = fluree_db_core::stats_wire::decode_stats(&bytes).unwrap(); + assert_eq!(via_core.links, Some(expected)); + + let without = encode_stats(&stats_with_historical()); + assert_eq!( + &bytes[..without.len()], + &without[..], + "appended, not interleaved" + ); + let (decoded, _) = decode_stats_with_len(&without).unwrap(); + assert_eq!(decoded.links, None); + } + /// Forward evolution: a tail carrying an unknown future tag reads as /// absent (conservative), and the remainder is consumed so the section /// length still accounts for every byte. diff --git a/fluree-db-core/src/index_stats.rs b/fluree-db-core/src/index_stats.rs index 6c4a5ad2a7..8db8f0b78b 100644 --- a/fluree-db-core/src/index_stats.rs +++ b/fluree-db-core/src/index_stats.rs @@ -100,6 +100,20 @@ impl PropertyStatEntry { } } +// === Reification Link Statistics === + +/// Live `rdf:reifies` links whose triple term has one inner predicate, +/// ledger-wide. A predicate's handle interval holds its terms, not its live +/// links: one term can have many reifiers, and a term outlives its retracted +/// links, so the planner reads this count instead. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct LinkStatEntry { + /// The inner predicate's SID as (namespace_code, name). + pub sid: (u16, String), + /// Live links. + pub count: u64, +} + // === Index Statistics === /// Index statistics (fast estimates). @@ -158,6 +172,9 @@ pub struct IndexStats { /// unknown — permanently for that chain, until a full rebuild resets /// the boundary to genesis. pub historical_since_t: Option, + /// Live `rdf:reifies` links per inner predicate. `None` when the stats + /// predate the count, which consumers must read as unknown. + pub links: Option>, } impl IndexStats { diff --git a/fluree-db-core/src/lib.rs b/fluree-db-core/src/lib.rs index 24cd8dcd35..16bbff31ba 100644 --- a/fluree-db-core/src/lib.rs +++ b/fluree-db-core/src/lib.rs @@ -139,7 +139,7 @@ pub use ids::{ pub use index_schema::{IndexSchema, SchemaPredicateInfo, SchemaPredicates}; pub use index_stats::{ ClassPropertyUsage, ClassRefCount, ClassStatEntry, GraphPropertyStatEntry, GraphStatsEntry, - IndexStats, PropertyStatEntry, + IndexStats, LinkStatEntry, PropertyStatEntry, }; pub use ledger_id::{ format_ledger_id, normalize_ledger_id, parse_ledger_id_with_time, parse_time_travel_spec, diff --git a/fluree-db-core/src/stats_view.rs b/fluree-db-core/src/stats_view.rs index 18f669ed8d..178468da5c 100644 --- a/fluree-db-core/src/stats_view.rs +++ b/fluree-db-core/src/stats_view.rs @@ -101,6 +101,9 @@ pub struct StatsView { /// query stats-cache builder; defaults `false` so any caller that does not /// explicitly vouch for current-state exactness never triggers elision. pub class_coverage_trustworthy: bool, + /// Inner predicate SID -> live `rdf:reifies` links whose triple term has + /// it. `None` when the stats carry no link counts. + pub links: Option>, } /// Per-property statistics within a graph, keyed by numeric IDs. @@ -219,6 +222,13 @@ impl StatsView { } } + view.links = stats.links.as_ref().map(|links| { + links + .iter() + .map(|l| (Sid::new(l.sid.0, &l.sid.1), l.count)) + .collect() + }); + if let Some(ref graphs) = stats.graphs { for g_entry in graphs { let mut prop_map = HashMap::new(); @@ -292,6 +302,14 @@ impl StatsView { view } + /// Live `rdf:reifies` links whose triple term has inner predicate `p`; + /// `None` when the stats carry no link counts. + pub fn link_count(&self, p: &Sid) -> Option { + self.links + .as_ref() + .map(|links| links.get(p).copied().unwrap_or(0)) + } + /// The per-class stats entry for `class_sid`, by binary search over the /// class table, which every producer sorts by `class_sid`. A miss reads as /// "no proof" to every caller, so an unsorted table declines rewrites @@ -615,6 +633,7 @@ mod tests { classes: None, graphs: None, historical_since_t: None, + links: None, } } @@ -660,6 +679,7 @@ mod tests { classes: None, graphs: None, historical_since_t: None, + links: None, }; let view = StatsView::from_db_stats(&stats); assert!(!view.has_property_stats()); @@ -684,6 +704,7 @@ mod tests { classes: None, graphs: None, historical_since_t: None, + links: None, }; let view = StatsView::from_db_stats(&stats); assert!(view.has_property_stats()); @@ -1027,6 +1048,7 @@ mod tests { }]), graphs: None, historical_since_t: None, + links: None, }; let view = StatsView::from_db_stats(&stats); assert!(view.has_class_stats()); diff --git a/fluree-db-core/src/stats_wire.rs b/fluree-db-core/src/stats_wire.rs index b5c8ac23df..5a53e400da 100644 --- a/fluree-db-core/src/stats_wire.rs +++ b/fluree-db-core/src/stats_wire.rs @@ -8,7 +8,7 @@ use crate::index_schema::{IndexSchema, SchemaPredicateInfo, SchemaPredicates}; use crate::index_stats::{ ClassPropertyUsage, ClassRefCount, ClassStatEntry, GraphPropertyStatEntry, GraphStatsEntry, - IndexStats, PropertyStatEntry, + IndexStats, LinkStatEntry, PropertyStatEntry, }; use crate::sid::Sid; use std::io; @@ -369,6 +369,56 @@ pub fn encode_class_tail(buf: &mut Vec, graphs: &[&GraphStatsEntry]) { } } +/// Wire tag identifying the reification-link tail. +const LINK_TAIL_TAG: u8 = 3; + +/// Append the live `rdf:reifies` link counts per inner predicate. Written +/// after the historical and class tails, which a reader that predates it +/// still parses; such a reader stops at this tag. +/// +/// ```text +/// [tag: u8 = 3] +/// [entry_count: varint] +/// per entry, by Sid: [ns_code: u16 LE][name_len: varint][name][count: varint] +/// ``` +pub fn encode_link_tail(buf: &mut Vec, links: &[LinkStatEntry]) { + use crate::commit::codec::varint::encode_varint; + let mut sorted: Vec<&LinkStatEntry> = links.iter().collect(); + sorted.sort_by(|a, b| a.sid.cmp(&b.sid)); + buf.push(LINK_TAIL_TAG); + encode_varint(sorted.len() as u64, buf); + for entry in sorted { + buf.extend_from_slice(&entry.sid.0.to_le_bytes()); + encode_varint(entry.sid.1.len() as u64, buf); + buf.extend_from_slice(entry.sid.1.as_bytes()); + encode_varint(entry.count, buf); + } +} + +/// Decode the link tail written by [`encode_link_tail`] (tag already consumed). +fn decode_link_tail(data: &[u8], pos: &mut usize) -> io::Result> { + let (n, cap) = read_count(data, pos)?; + let mut links = Vec::with_capacity(cap); + for _ in 0..n { + let ns_code = read_u16(data, pos)?; + let len = read_varint(data, pos)? as usize; + ensure_len(data, *pos, len, "link tail sid")?; + let name = std::str::from_utf8(&data[*pos..*pos + len]).map_err(|_| { + io::Error::new( + io::ErrorKind::InvalidData, + "link tail: sid name is not UTF-8", + ) + })?; + *pos += len; + let count = read_varint(data, pos)?; + links.push(LinkStatEntry { + sid: (ns_code, name.to_string()), + count, + }); + } + Ok(links) +} + /// Decode the class tail written by [`encode_class_tail`] (tag already /// consumed): each graph's class table. Decoded Sids share their names. fn decode_class_tail(data: &[u8], pos: &mut usize) -> io::Result)>> { @@ -632,9 +682,14 @@ pub fn decode_stats(data: &[u8]) -> io::Result<(IndexStats, usize)> { // Appended sections, each led by its tag; an unknown tag ends the parse // (the root length-prefixes the stats section, so the rest is skipped). let mut tail = None; + let mut links = None; while pos < data.len() { match data[pos] { HISTORICAL_TAIL_TAG => tail = decode_historical_tail(data, &mut pos)?, + LINK_TAIL_TAG => { + pos += 1; + links = Some(decode_link_tail(data, &mut pos)?); + } CLASS_TAIL_TAG => { pos += 1; for (g_id, classes) in decode_class_tail(data, &mut pos)? { @@ -671,6 +726,7 @@ pub fn decode_stats(data: &[u8]) -> io::Result<(IndexStats, usize)> { Some(graphs) }, historical_since_t: None, + links, }; apply_historical_tail(&mut stats, tail); diff --git a/fluree-db-indexer/src/build/incremental.rs b/fluree-db-indexer/src/build/incremental.rs index 7c0a1245b3..6b789fe0c3 100644 --- a/fluree-db-indexer/src/build/incremental.rs +++ b/fluree-db-indexer/src/build/incremental.rs @@ -2348,6 +2348,27 @@ pub async fn incremental_index( stats_hook.set_rdf_type_p_id(rdf_type_p_id); stats_hook.set_track_ref_targets(true); + // Live link counts carry forward from the base. A base that predates + // them leaves them unknown: a count of only this window's links would + // read as the ledger's. + let reifies_p_id = novelty.shared.predicates.get(fluree_vocab::rdf::REIFIES); + let base_links = base_root.stats.as_ref().and_then(|s| s.links.as_ref()); + if let (Some(reifies), Some(links)) = (reifies_p_id, base_links) { + for link in links { + let prefix = novelty + .shared + .ns_prefixes + .get(&link.sid.0) + .map(String::as_str) + .unwrap_or(""); + let iri = format!("{prefix}{}", link.sid.1); + if let Some(inner) = novelty.shared.predicates.get(&iri) { + stats_hook.seed_term_rows(reifies, inner, link.count); + } + } + } + let links_known = base_links.is_some(); + // Seed per-graph flake totals from base root stats. if let Some(ref base_stats) = base_root.stats { if let Some(ref graphs) = base_stats.graphs { @@ -3786,6 +3807,15 @@ pub async fn incremental_index( } let root_classes = fluree_db_core::index_stats::union_per_graph_classes(&final_graphs); + let links = links_known.then(|| { + crate::stats::link_stat_entries(&id_stats_result.term_rows, reifies_p_id, |p_id| { + let iri = novelty.shared.predicates.resolve(p_id).unwrap_or(""); + match trie.longest_match(iri) { + Some((code, prefix_len)) => (code, iri[prefix_len..].to_string()), + None => (0u16, iri.to_string()), + } + }) + }); is::IndexStats { flakes: id_stats_result.total_flakes, @@ -3794,6 +3824,7 @@ pub async fn incremental_index( classes: root_classes, graphs: Some(final_graphs), historical_since_t, + links, } }; diff --git a/fluree-db-indexer/src/build/rebuild.rs b/fluree-db-indexer/src/build/rebuild.rs index fb40230f54..ea6c80d234 100644 --- a/fluree-db-indexer/src/build/rebuild.rs +++ b/fluree-db-indexer/src/build/rebuild.rs @@ -1160,6 +1160,16 @@ where let root_classes = fluree_db_core::index_stats::union_per_graph_classes(&final_graphs); + let links = crate::stats::link_stat_entries( + &id_stats_result.term_rows, + shared.predicates.get(fluree_vocab::rdf::REIFIES), + |p_id| { + predicate_sids + .get(p_id as usize) + .cloned() + .unwrap_or_default() + }, + ); is::IndexStats { flakes: id_stats_result.total_flakes, @@ -1173,6 +1183,7 @@ where // hook, so the historical tag sets cover every `t` the // ledger has ever had. historical_since_t: Some(0), + links: Some(links), } }; diff --git a/fluree-db-indexer/src/run_index/build/build_from_commits.rs b/fluree-db-indexer/src/run_index/build/build_from_commits.rs index 4d5b173a8d..f4c463f282 100644 --- a/fluree-db-indexer/src/run_index/build/build_from_commits.rs +++ b/fluree-db-indexer/src/run_index/build/build_from_commits.rs @@ -1262,8 +1262,8 @@ pub fn build_indexes_from_commits( // commits, so cross-chunk duplicates were counted once per copy; // discount the copies the SPOT merge collapsed. HLL sketches are // duplicate-insensitive and need no correction. - for (&(p_id, o_type), &n) in &merge_duplicates { - target_hook.discount_import_duplicates(config.g_id, p_id, o_type, n); + for (&(p_id, o_type, inner), &n) in &merge_duplicates { + target_hook.discount_import_duplicates(config.g_id, p_id, o_type, inner, n); } tracing::info!( elapsed_ms = stats_merge_start.elapsed().as_millis(), @@ -1358,12 +1358,16 @@ pub fn build_indexes_from_commits( )) } +/// Cross-chunk duplicate copies the SPOT merge collapsed, keyed `(p_id, +/// o_type, inner predicate)`; the inner predicate is a triple-term row's and +/// 0 for every other row. +type DuplicateTally = FxHashMap<(u32, u16, u32), u64>; + struct SpotBuild { result: IndexBuildResult, class_stats: Option, - /// Cross-chunk duplicate copies the SPOT merge collapsed, keyed - /// `(p_id, o_type)` so the id-stats hook can discount them. - merge_duplicates: FxHashMap<(u32, u16), u64>, + /// Cross-chunk duplicate copies the SPOT merge collapsed. + merge_duplicates: DuplicateTally, } fn build_spot_index_from_commits( @@ -1418,7 +1422,7 @@ fn build_spot_index_from_commits( let mut class_stats_collector = rdf_type_p_id.map(|p_id| SpotClassStatsCollector::new(p_id, class_membership)); - let mut merge_duplicates: FxHashMap<(u32, u16), u64> = FxHashMap::default(); + let mut merge_duplicates: DuplicateTally = FxHashMap::default(); let total_rows = if commits.len() <= fd_plan.spot_fan_in { // Flat merge: one long-lived reader per chunk, all open at once. @@ -1593,7 +1597,7 @@ fn pump_spot_merge( g_id: u16, mut class_stats_collector: Option<&mut SpotClassStatsCollector>, progress: Option<&AtomicU64>, - duplicates: &mut FxHashMap<(u32, u16), u64>, + duplicates: &mut DuplicateTally, ) -> io::Result where T: MergeSource, @@ -1611,7 +1615,16 @@ where let pop_dropped = dropped - dropped_seen; dropped_seen = dropped; if pop_dropped > 0 { - *duplicates.entry((record.p_id, record.o_type)).or_insert(0) += pop_dropped; + let inner = if fluree_db_core::o_type::OType::from_u16(record.o_type) + == fluree_db_core::o_type::OType::TRIPLE_TERM + { + fluree_db_core::triple_term::term_handle_p_id(record.o_key) + } else { + 0 + }; + *duplicates + .entry((record.p_id, record.o_type, inner)) + .or_insert(0) += pop_dropped; } if op == 0 { continue; diff --git a/fluree-db-indexer/src/stats/id_hook.rs b/fluree-db-indexer/src/stats/id_hook.rs index f57cd6ff53..44bf15001f 100644 --- a/fluree-db-indexer/src/stats/id_hook.rs +++ b/fluree-db-indexer/src/stats/id_hook.rs @@ -157,7 +157,8 @@ pub struct StatsRecord { /// Construct a `StatsRecord` from a V2 `RunRecordV2` + op byte. /// /// Maps `OType` to the legacy fields needed by `IdStatsHook::on_record`: -/// - `o_kind`: 0x05 (REF_ID) for `OType::IRI_REF`, 0 otherwise (only REF detection matters) +/// - `o_kind`: 0x05 (REF_ID) for `OType::IRI_REF`, TRIPLE_TERM for a term +/// handle, 0 otherwise /// - `dt`: derived from `OType` category (approximate but consistent) /// - `o_hash`: uses `value_hash_v2(o_type, o_key)` (V2-compatible domain separation) /// - `lang_id`: extracted from `OType` if langString, 0 otherwise @@ -169,11 +170,14 @@ pub fn stats_record_from_v2( let ot = OType::from_u16(rec.o_type); - // Map OType to legacy o_kind (only REF_ID detection matters for class tracking). + // Map OType to legacy o_kind: only REF_ID (class tracking) and + // TRIPLE_TERM (link counts) are special-cased. let o_kind = if ot == OType::IRI_REF { 0x05 // ObjKind::REF_ID + } else if ot == OType::TRIPLE_TERM { + fluree_db_core::value_id::ObjKind::TRIPLE_TERM.as_u8() } else { - 0 // doesn't matter for stats — only REF_ID is special-cased + 0 }; // Map OType to approximate ValueTypeTag for datatype counting. @@ -269,6 +273,9 @@ pub struct IdStatsResult { pub graphs: Vec, /// Total flake count across all graphs. pub total_flakes: u64, + /// Live rows whose object is a triple term, ledger-wide, as `(p_id, the + /// term's inner predicate, count)`; zero counts dropped, sorted. + pub term_rows: Vec<(u32, u32, u64)>, } /// ID-based stats hook for import/index paths where GlobalDicts are available. @@ -338,6 +345,9 @@ pub struct IdStatsHook { /// Used at finalize-time to derive per-class language distributions by /// cross-referencing with subject_classes. subject_prop_langs: HashMap<(GraphId, u64), HashMap>>, + /// Rows whose object is a triple term: (p_id, the term's inner predicate) + /// → signed delta count. The inner predicate is the handle's high half. + term_rows: HashMap<(u32, u32), i64>, } impl IdStatsHook { @@ -393,6 +403,12 @@ impl IdStatsHook { self.track_ref_targets = enabled; } + /// Seed the triple-term row count for `(p_id, inner_p_id)` from a base + /// index, before records move it. + pub fn seed_term_rows(&mut self, p_id: u32, inner_p_id: u32, count: u64) { + *self.term_rows.entry((p_id, inner_p_id)).or_insert(0) += count as i64; + } + /// Process a single record with resolved IDs. /// /// Called per-op after the resolver maps Sids to numeric IDs. The signed @@ -457,6 +473,11 @@ impl IdStatsHook { // Historical tag set: every record's tag, regardless of op or delta. hll.note_historical_tag(rec.dt.as_u8()); + if rec.o_kind == fluree_db_core::value_id::ObjKind::TRIPLE_TERM.as_u8() { + let inner = fluree_db_core::triple_term::term_handle_p_id(rec.o_key); + *self.term_rows.entry((rec.p_id, inner)).or_insert(0) += delta; + } + // Track class membership and class→property attribution (graph-scoped). if let Some(rdf_type_pid) = self.rdf_type_p_id { if rec.p_id == rdf_type_pid && rec.o_kind == 0x05 { @@ -541,8 +562,23 @@ impl IdStatsHook { /// hooks never track class detail (rdf:type p_id unset), so class maps /// need no correction; the set-view class stats come from the /// `SpotClassStatsCollector` on the already-deduplicated merge output. - pub fn discount_import_duplicates(&mut self, g_id: GraphId, p_id: u32, o_type: u16, n: u64) { + /// + /// `inner_p_id` is a triple-term row's inner predicate (ignored for other + /// rows). + pub fn discount_import_duplicates( + &mut self, + g_id: GraphId, + p_id: u32, + o_type: u16, + inner_p_id: u32, + n: u64, + ) { let delta = n as i64; + if fluree_db_core::o_type::OType::from_u16(o_type) + == fluree_db_core::o_type::OType::TRIPLE_TERM + { + *self.term_rows.entry((p_id, inner_p_id)).or_insert(0) -= delta; + } self.flake_count = self.flake_count.saturating_sub(n as usize); *self.graph_flakes.entry(g_id).or_insert(0) -= delta; let hll = self @@ -581,6 +617,9 @@ impl IdStatsHook { for (key, delta) in other.class_counts { *self.class_counts.entry(key).or_insert(0) += delta; } + for (key, delta) in other.term_rows { + *self.term_rows.entry(key).or_insert(0) += delta; + } if !self.hll_only { for (key, class_map) in other.subject_class_deltas { let entry = self.subject_class_deltas.entry(key).or_default(); @@ -764,9 +803,18 @@ impl IdStatsHook { .map(|(_, &delta)| delta.max(0) as u64) .sum(); + let mut term_rows: Vec<(u32, u32, u64)> = self + .term_rows + .iter() + .filter(|(_, &n)| n > 0) + .map(|(&(p_id, inner), &n)| (p_id, inner, n as u64)) + .collect(); + term_rows.sort_unstable(); + IdStatsResult { graphs, total_flakes, + term_rows, } } @@ -1082,7 +1130,7 @@ mod tests { // The merge collapsed one copy. record() uses ValueTypeTag::INTEGER, // which OType::XSD_INTEGER maps back to. - hook.discount_import_duplicates(0, 1, OType::XSD_INTEGER.as_u16(), 1); + hook.discount_import_duplicates(0, 1, OType::XSD_INTEGER.as_u16(), 0, 1); assert_eq!(property_count(&hook), 1); assert_eq!(hook.graph_flakes_mut()[&0], 1); @@ -1104,4 +1152,61 @@ mod tests { hook.on_record(&record(false, 3)); assert_eq!(property_count(&hook), 1); } + + /// Link rows count per inner predicate (the handle's high half) with the + /// same signed deltas as property counts, through merge, the import + /// duplicate discount and a base seed. + #[test] + fn triple_term_rows_count_live_links_per_inner_predicate() { + use fluree_db_binary_index::format::run_record_v2::RunRecordV2; + use fluree_db_core::o_type::OType; + use fluree_db_core::subject_id::SubjectId; + use fluree_db_core::triple_term::term_handle; + + let link = |g_id: GraphId, inner: u32, seq: u32, op: u8| { + let rec = RunRecordV2 { + s_id: SubjectId(10 + u64::from(seq)), + o_key: term_handle(inner, seq), + p_id: 3, + t: 1, + o_i: u32::MAX, + o_type: OType::TRIPLE_TERM.as_u16(), + g_id, + }; + stats_record_from_v2(&rec, op) + }; + let rows = |hook: IdStatsHook| hook.finalize().term_rows; + + let mut hook = IdStatsHook::new(); + hook.on_record(&link(0, 7, 0, 1)); + hook.on_record(&link(2, 7, 1, 1)); + hook.on_record(&link(0, 9, 0, 1)); + hook.on_record(&link(0, 9, 0, 0)); + hook.on_record(&record(true, 1)); + let mut other = IdStatsHook::new(); + other.on_record(&link(0, 7, 2, 1)); + hook.merge_from(other); + hook.discount_import_duplicates(0, 3, OType::TRIPLE_TERM.as_u16(), 7, 1); + assert_eq!(rows(hook), vec![(3, 7, 2)]); + + let mut seeded = IdStatsHook::new(); + seeded.seed_term_rows(3, 9, 4); + seeded.on_record_with_base_presence(&link(0, 9, 0, 0), true); + seeded.on_record_with_base_presence(&link(0, 9, 5, 1), true); + assert_eq!(rows(seeded), vec![(3, 9, 3)]); + + let entries = crate::stats::link_stat_entries(&[(3, 7, 2), (4, 7, 5)], Some(3), |p| { + (100, format!("p{p}")) + }); + assert_eq!( + entries, + vec![fluree_db_core::LinkStatEntry { + sid: (100, "p7".to_string()), + count: 2 + }] + ); + assert!( + crate::stats::link_stat_entries(&[(3, 7, 2)], None, |_| (0, String::new())).is_empty() + ); + } } diff --git a/fluree-db-indexer/src/stats/mod.rs b/fluree-db-indexer/src/stats/mod.rs index 9e13a4c59f..569e187267 100644 --- a/fluree-db-indexer/src/stats/mod.rs +++ b/fluree-db-indexer/src/stats/mod.rs @@ -88,6 +88,30 @@ where }) } +/// The live `rdf:reifies` links per inner predicate, from the hook's +/// triple-term row counts, with each inner predicate's SID. Empty when the +/// ledger has no `rdf:reifies` predicate. +pub fn link_stat_entries( + term_rows: &[(u32, u32, u64)], + reifies_p_id: Option, + mut resolve_predicate_sid: F, +) -> Vec +where + F: FnMut(u32) -> (u16, String), +{ + let Some(reifies) = reifies_p_id else { + return Vec::new(); + }; + term_rows + .iter() + .filter(|&&(p_id, _, _)| p_id == reifies) + .map(|&(_, inner, count)| fluree_db_core::LinkStatEntry { + sid: resolve_predicate_sid(inner), + count, + }) + .collect() +} + /// The same roll-up for a caller that already holds each predicate's SID — the /// import builds a `predicate_sids` table as it goes, so it never needs the IRI /// round trip. diff --git a/fluree-db-novelty/src/runtime_stats.rs b/fluree-db-novelty/src/runtime_stats.rs index 7621244a4b..e6fe75b26e 100644 --- a/fluree-db-novelty/src/runtime_stats.rs +++ b/fluree-db-novelty/src/runtime_stats.rs @@ -6,8 +6,8 @@ use fluree_db_core::is_rdf_type; use fluree_db_core::range_provider::{RangeProvider, RangeQuery}; use fluree_db_core::{ ClassPropertyUsage, ClassRefCount, ClassStatEntry, GraphId, GraphPropertyStatEntry, - GraphStatsEntry, IndexStats, LedgerSnapshot, OverlayProvider, PropertyStatEntry, RangeMatch, - RangeOptions, RangeTest, RuntimePredicateId, RuntimeSmallDicts, Sid, ValueTypeTag, + GraphStatsEntry, IndexStats, LedgerSnapshot, LinkStatEntry, OverlayProvider, PropertyStatEntry, + RangeMatch, RangeOptions, RangeTest, RuntimePredicateId, RuntimeSmallDicts, Sid, ValueTypeTag, }; use fluree_db_core::{Flake, FlakeMeta, FlakeValue}; use fluree_vocab::namespaces::FLUREE_COMMIT; @@ -584,6 +584,7 @@ fn assemble_fast_stats_inner( .map(|(idx, entry)| (entry.g_id, idx)) .collect(); let mut flakes_delta: i64 = 0; + let mut link_deltas: HashMap = HashMap::new(); let mut graph_subject_classes: HashMap<(GraphId, Sid), HashSet> = HashMap::new(); // The subset of `graph_subject_classes` whose membership is new relative to // the base index — see [`RestatedAttribution`]. Only built when this pass @@ -645,6 +646,12 @@ fn assemble_fast_stats_inner( continue; } + if let FlakeValue::TripleTerm(term) = &flake.o { + if fluree_db_core::is_rdf_reifies(&flake.p) { + *link_deltas.entry(term.p.clone()).or_insert(0) += delta; + } + } + let datatype_tag = runtime_datatype_tag(flake); let sid_key = (flake.p.namespace_code, &*flake.p.name); if !indexed_ndv.contains(&sid_key) { @@ -729,6 +736,7 @@ fn assemble_fast_stats_inner( classes: None, graphs: None, historical_since_t: indexed.historical_since_t, + links: finalize_links(indexed, snapshot, link_deltas), }; if graphs.is_empty() { // The base's per-graph section is `None` or empty, so this copies nothing. @@ -1224,6 +1232,39 @@ fn graph_id_for_flake(snapshot: &LedgerSnapshot, flake: &Flake) -> GraphId { .unwrap_or(0) } +/// The base's live-link counts moved by novelty's links. A ledger never +/// indexed starts from none; a base whose counts are unknown stays unknown. +fn finalize_links( + indexed: &IndexStats, + snapshot: &LedgerSnapshot, + deltas: HashMap, +) -> Option> { + let base: &[LinkStatEntry] = match &indexed.links { + Some(links) => links, + None if snapshot.t == 0 => &[], + None => return None, + }; + let mut counts: HashMap<(u16, String), i64> = base + .iter() + .map(|l| (l.sid.clone(), l.count as i64)) + .collect(); + for (sid, delta) in deltas { + *counts + .entry((sid.namespace_code, sid.name.to_string())) + .or_insert(0) += delta; + } + let mut links: Vec = counts + .into_iter() + .filter(|(_, n)| *n > 0) + .map(|(sid, n)| LinkStatEntry { + sid, + count: n as u64, + }) + .collect(); + links.sort_by(|a, b| a.sid.cmp(&b.sid)); + Some(links) +} + fn indexed_t(indexed: &IndexStats, snapshot: &LedgerSnapshot) -> i64 { if indexed.graphs.is_some() || indexed.properties.is_some() || indexed.classes.is_some() { snapshot.t @@ -1801,6 +1842,7 @@ mod tests { classes: None, graphs: None, historical_since_t: None, + links: None, } } @@ -1884,6 +1926,7 @@ mod tests { classes: None, graphs: None, historical_since_t: None, + links: None, }; assert_eq!( fluree_db_core::StatsView::from_db_stats(&indexed).is_property_ref_only(&p), @@ -1997,6 +2040,7 @@ mod tests { classes: None, graphs: None, historical_since_t: None, + links: None, }; let alice = sid(10, "alice"); let bob = sid(10, "bob"); @@ -2091,6 +2135,7 @@ mod tests { classes: Some(vec![]), }]), historical_since_t: None, + links: None, }; let snapshot = LedgerSnapshot::genesis("test:main"); let mut novelty = Novelty::new(1); @@ -2150,6 +2195,7 @@ mod tests { }]), }]), historical_since_t: None, + links: None, }; let snapshot = LedgerSnapshot::genesis("test:main"); let mut novelty = Novelty::new(1); @@ -2254,6 +2300,7 @@ mod tests { properties: Some(Vec::new()), classes: None, graphs: None, + links: None, }; (snapshot, indexed, provider) } @@ -2628,6 +2675,7 @@ mod tests { }]), classes: None, graphs: None, + links: None, }; let mut novelty = Novelty::new(0); novelty @@ -2696,6 +2744,7 @@ mod tests { properties: vec![], classes, }]), + links: None, } } @@ -2998,6 +3047,7 @@ mod tests { classes: Some(base_classes), }]), historical_since_t: Some(0), + links: None, }; let mut novelty = Novelty::new(1); @@ -3143,4 +3193,79 @@ mod tests { let merged = assemble_planner_stats(&indexed, &genesis, &novelty, 3, Some(&lookup)); assert!(!Arc::ptr_eq(&merged, &indexed), "a real window must merge"); } + + /// Novelty's links move the base's live-link counts by inner predicate. A + /// ledger never indexed counts from none; a base without counts stays + /// unknown. + #[test] + fn novelty_links_move_the_live_link_counts() { + let knows = sid(10, "knows"); + let likes = sid(10, "likes"); + let link = |reifier: &str, p: &Sid, t: i64, op: bool| { + let term = fluree_db_core::TripleTermValue { + s: sid(10, "alice"), + p: p.clone(), + o: FlakeValue::Ref(sid(10, "bob")), + dt: fluree_db_core::edge::id_datatype_sid(), + lang: None, + }; + Flake::new( + sid(10, reifier), + fluree_db_core::rdf_reifies_sid().clone(), + FlakeValue::TripleTerm(Box::new(term)), + fluree_db_core::triple_term_datatype_sid().clone(), + t, + op, + None, + ) + }; + let mut novelty = Novelty::new(1); + novelty + .apply_commit( + vec![link("r1", &knows, 2, true), link("r2", &likes, 2, true)], + 2, + &HashMap::new(), + ) + .expect("links"); + novelty + .apply_commit(vec![link("r3", &knows, 3, false)], 3, &HashMap::new()) + .expect("retract"); + + let mut indexed = mixed_property_index(&sid(10, "p")); + indexed.links = Some(vec![LinkStatEntry { + sid: (10, "knows".to_string()), + count: 5, + }]); + let mut snapshot = LedgerSnapshot::genesis("test:main"); + snapshot.t = 1; + let merged = assemble_fast_stats(&indexed, &snapshot, &novelty, 3, None); + assert_eq!( + merged.links, + Some(vec![ + LinkStatEntry { + sid: (10, "knows".to_string()), + count: 5 + }, + LinkStatEntry { + sid: (10, "likes".to_string()), + count: 1 + }, + ]) + ); + + indexed.links = None; + let merged = assemble_fast_stats(&indexed, &snapshot, &novelty, 3, None); + assert_eq!(merged.links, None, "unknown base counts stay unknown"); + + let genesis = LedgerSnapshot::genesis("test:main"); + let merged = assemble_fast_stats(&IndexStats::default(), &genesis, &novelty, 3, None); + assert_eq!( + merged.links, + Some(vec![LinkStatEntry { + sid: (10, "likes".to_string()), + count: 1 + }]), + "a ledger never indexed counts its novelty's links" + ); + } } diff --git a/fluree-db-policy/src/index.rs b/fluree-db-policy/src/index.rs index aea4a738ec..860dcc8d42 100644 --- a/fluree-db-policy/src/index.rs +++ b/fluree-db-policy/src/index.rs @@ -326,6 +326,7 @@ mod tests { }]), graphs: None, historical_since_t: None, + links: None, } } @@ -545,6 +546,7 @@ mod tests { ]), graphs: None, historical_since_t: None, + links: None, }; let person_only: HashSet = [person.clone()].into_iter().collect(); diff --git a/fluree-db-query/src/explain.rs b/fluree-db-query/src/explain.rs index 6e6b5f16ce..b2822ccc98 100644 --- a/fluree-db-query/src/explain.rs +++ b/fluree-db-query/src/explain.rs @@ -453,7 +453,7 @@ impl fmt::Display for PatternDisplay { // Generalized explain for all pattern types // ============================================================================= -use crate::planner::{estimate_pattern, reorder_patterns, PatternEstimate}; +use crate::planner::{reorder_patterns, PatternEstimate}; /// Display information for any pattern type (generalized) #[derive(Debug, Clone)] @@ -490,15 +490,16 @@ pub fn explain_all_patterns(patterns: &[Pattern], stats: Option<&StatsView>) -> .map(|s| s.has_property_stats() || s.has_class_stats()) .unwrap_or(false); + let pins = crate::planner::link_pins(patterns); let original_patterns: Vec = patterns .iter() - .map(|p| build_general_pattern_display(p, stats)) + .map(|p| build_general_pattern_display(p, &pins, stats)) .collect(); let reordered = reorder_patterns(patterns, stats, &HashSet::new()); let optimized_patterns: Vec = reordered .iter() - .map(|p| build_general_pattern_display(p, stats)) + .map(|p| build_general_pattern_display(p, &pins, stats)) .collect(); let optimization = if patterns.len() <= 1 { @@ -526,9 +527,10 @@ pub fn explain_all_patterns(patterns: &[Pattern], stats: Option<&StatsView>) -> /// Build display info for any pattern type fn build_general_pattern_display( pattern: &Pattern, + pins: &std::collections::HashMap, stats: Option<&StatsView>, ) -> GeneralPatternDisplay { - let cardinality = estimate_pattern(pattern, &HashSet::new(), stats); + let cardinality = crate::planner::estimate_in_group(pattern, pins, &HashSet::new(), stats); let variables = pattern.produced_vars(); let triple_detail = if let Pattern::Triple(tp) = pattern { diff --git a/fluree-db-query/src/fast_count.rs b/fluree-db-query/src/fast_count.rs index 3b8a574736..bbddd92676 100644 --- a/fluree-db-query/src/fast_count.rs +++ b/fluree-db-query/src/fast_count.rs @@ -2305,6 +2305,7 @@ mod tests { classes: None, }]), historical_since_t: None, + links: None, } } @@ -2363,6 +2364,7 @@ mod tests { classes: None, graphs: None, historical_since_t: None, + links: None, }; assert_eq!(count_literal_rows_from_stats(&no_graphs, 0), None); } diff --git a/fluree-db-query/src/planner.rs b/fluree-db-query/src/planner.rs index 36000f1f28..3546ad60dd 100644 --- a/fluree-db-query/src/planner.rs +++ b/fluree-db-query/src/planner.rs @@ -10,7 +10,7 @@ use crate::ir::triple::{Ref, Term, TriplePattern}; use crate::ir::{CompareOp, Function, Grouping, Pattern, SubqueryPattern}; use crate::var_registry::VarId; -use fluree_db_core::{FlakeValue, PropertyStatData, StatsView}; +use fluree_db_core::{FlakeValue, PropertyStatData, Sid, StatsView}; use std::collections::{HashMap, HashSet}; // ============================================================================= @@ -1519,6 +1519,97 @@ struct RankedPattern { /// left-join ordering barrier (see [`left_join_order_barriers`]). Empty /// unless an OPTIONAL in the group shares a not-yet-certain variable with it. after_indices: Vec, + /// The inner predicate a sibling filter pins on this link triple's term + /// (see [`link_triple_pin`]). + link_pin: Option, +} + +impl RankedPattern { + fn estimate(&self, bound_vars: &HashSet, stats: Option<&StatsView>) -> PatternEstimate { + estimate_pinned(&self.pattern, self.link_pin.as_ref(), bound_vars, stats) + } +} + +/// [`estimate_pattern`], except that a link scan `?r rdf:reifies ?t` whose +/// term a filter pins to inner predicate `pin` reads only that predicate's +/// handle interval: its live links, not every link. +fn estimate_pinned( + pattern: &Pattern, + pin: Option<&Sid>, + bound_vars: &HashSet, + stats: Option<&StatsView>, +) -> PatternEstimate { + if let (Some(pin), Some(stats), Pattern::Triple(tp)) = (pin, stats, pattern) { + if classify_pattern(tp, bound_vars) == PatternType::PropertyScan { + if let Some(links) = stats.link_count(pin) { + return PatternEstimate::Source { + row_count: links as f64, + }; + } + } + } + estimate_pattern(pattern, bound_vars, stats) +} + +/// The inner predicates a group's filters pin on its term variables. +pub fn link_pins(patterns: &[Pattern]) -> HashMap { + patterns + .iter() + .filter_map(|p| match p { + Pattern::Filter(expr) => term_predicate_pin(expr), + _ => None, + }) + .collect() +} + +/// [`estimate_pattern`] for a pattern of a group whose filters pin `pins`. +pub fn estimate_in_group( + pattern: &Pattern, + pins: &HashMap, + bound_vars: &HashSet, + stats: Option<&StatsView>, +) -> PatternEstimate { + estimate_pinned( + pattern, + link_triple_pin(pattern, pins).as_ref(), + bound_vars, + stats, + ) +} + +/// The inner predicate `sameTerm(PREDICATE(?t),

)` pins on term variable +/// `?t`, as the link lowering writes it. +fn term_predicate_pin(filter: &Expression) -> Option<(VarId, Sid)> { + let Expression::Call { args, .. } = filter else { + return None; + }; + let var = args.iter().find_map(|arg| match arg { + Expression::Call { + func: Function::TriplePredicate, + args, + } => match args.as_slice() { + [Expression::Var(v)] => Some(*v), + _ => None, + }, + _ => None, + })?; + term_component_constraint(filter, var)? + .term_predicate + .map(|p| (var, p)) +} + +/// The pinned inner predicate of a link triple `?r rdf:reifies ?t`. +fn link_triple_pin(pattern: &Pattern, pins: &HashMap) -> Option { + let Pattern::Triple(tp) = pattern else { + return None; + }; + let Ref::Sid(p) = &tp.p else { + return None; + }; + if !fluree_db_core::is_rdf_reifies(p) { + return None; + } + pins.get(&tp.o.as_var()?).cloned() } /// A deferred pattern (FILTER/BIND) with pre-computed input variables. @@ -2023,6 +2114,7 @@ pub fn reorder_patterns_with_seed( } let mut bound_vars = seed.schema.clone(); + let link_pins = link_pins(patterns); // PIPELINE outputs of UNCORRELATED sibling subqueries (Cypher WITH-pipeline // producers). A pattern consuming one of these must be placed AFTER the @@ -2236,21 +2328,25 @@ pub fn reorder_patterns_with_seed( } } - match estimate_pattern(pattern, &bound_vars, stats) { + let link_pin = link_triple_pin(pattern, &link_pins); + match estimate_pinned(pattern, link_pin.as_ref(), &bound_vars, stats) { PatternEstimate::Source { .. } => sources.push(RankedPattern { orig_index: i, pattern: pattern.clone(), after_indices: barriers[i].clone(), + link_pin, }), PatternEstimate::Reducer { .. } => reducers.push(RankedPattern { orig_index: i, pattern: pattern.clone(), after_indices: barriers[i].clone(), + link_pin, }), PatternEstimate::Expander { .. } => expanders.push(RankedPattern { orig_index: i, pattern: pattern.clone(), after_indices: barriers[i].clone(), + link_pin, }), PatternEstimate::Deferred => { let mut required_vars: HashSet = @@ -2491,8 +2587,8 @@ fn try_place_reducer( .enumerate() .filter(|(_, rp)| pattern_shares_variables(&rp.pattern, bound_vars)) .min_by(|(_, a), (_, b)| { - let ca = estimate_pattern(&a.pattern, bound_vars, stats); - let cb = estimate_pattern(&b.pattern, bound_vars, stats); + let ca = a.estimate(bound_vars, stats); + let cb = b.estimate(bound_vars, stats); ca.multiplier() .partial_cmp(&cb.multiplier()) .unwrap_or(std::cmp::Ordering::Equal) @@ -2689,8 +2785,8 @@ fn rank_seed_candidates( | Pattern::S2Search(_) => 0_u8, _ => 1_u8, }; - let ci = estimate_pattern(&remaining[i].pattern, bound_vars, stats); - let cj = estimate_pattern(&remaining[j].pattern, bound_vars, stats); + let ci = remaining[i].estimate(bound_vars, stats); + let cj = remaining[j].estimate(bound_vars, stats); // 1. Search sources seed first — but only before anything is bound. let by_search = if has_bound { @@ -2934,8 +3030,8 @@ fn try_place_expander( .enumerate() .filter(|(_, rp)| pattern_shares_variables(&rp.pattern, bound_vars)) .min_by(|(_, a), (_, b)| { - let ca = estimate_pattern(&a.pattern, bound_vars, stats); - let cb = estimate_pattern(&b.pattern, bound_vars, stats); + let ca = a.estimate(bound_vars, stats); + let cb = b.estimate(bound_vars, stats); ca.multiplier() .partial_cmp(&cb.multiplier()) .unwrap_or(std::cmp::Ordering::Equal) @@ -3838,16 +3934,19 @@ mod tests { orig_index: 0, pattern: Pattern::Triple(make_pattern(product, "vendor", vendor)), after_indices: Vec::new(), + link_pin: None, }, RankedPattern { orig_index: 1, pattern: Pattern::Triple(make_pattern(product, "numeric", numeric)), after_indices: Vec::new(), + link_pin: None, }, RankedPattern { orig_index: 2, pattern: Pattern::Triple(make_pattern(VarId(3), "reviewer", vendor)), after_indices: Vec::new(), + link_pin: None, }, ]; let bound = HashSet::from([product]); @@ -7297,4 +7396,62 @@ mod tests { let sq = SubqueryPattern::new(vec![VarId(0)], unanchored_body()).with_distinct(); assert!(!subquery_output_estimate_is_bounded(&sq)); } + + /// S7's shape: a link scan whose term a filter pins to a rare inner + /// predicate drives the join once the stats count that predicate's live + /// links; ranked at every link, it loses to a smaller scan. + #[test] + fn a_pinned_link_scan_is_ranked_by_its_inner_predicates_links() { + let (source, ty, ann, term) = (VarId(0), VarId(1), VarId(2), VarId(3)); + let reifies = fluree_db_core::rdf_reifies_sid().clone(); + let part_of = Sid::new(100, "PART_OF"); + let patterns = vec![ + triple(source, "type", ty), + triple(ann, "derives_from", source), + Pattern::Triple(TriplePattern::new( + Ref::Var(ann), + Ref::Sid(reifies.clone()), + Term::Var(term), + )), + Pattern::Filter(Expression::Call { + func: Function::SameTerm, + args: vec![ + Expression::Call { + func: Function::TriplePredicate, + args: vec![Expression::Var(term)], + }, + Expression::Const(FlakeValue::Ref(part_of.clone())), + ], + }), + ]; + let mut stats = stats_with(&[("type", 600_000, 100), ("derives_from", 3_000_000, 1_000)]); + stats.properties.insert( + reifies.clone(), + PropertyStatData { + count: 2_000_000, + ndv_values: 2_000_000, + ndv_subjects: 2_000_000, + }, + ); + let first = + |stats: &StatsView| match &reorder_patterns(&patterns, Some(stats), &HashSet::new())[0] + { + Pattern::Triple(tp) => tp.p.clone(), + other => panic!("{other:?}"), + }; + assert_eq!(first(&stats), Ref::Sid(Sid::new(100, "type"))); + + stats.links = Some(HashMap::from([(part_of.clone(), 120_000)])); + assert_eq!(first(&stats), Ref::Sid(reifies)); + let pins = link_pins(&patterns); + assert_eq!( + estimate_in_group(&patterns[2], &pins, &HashSet::new(), Some(&stats)).row_count(), + 120_000.0 + ); + // Once the reifier is bound the scan is a probe, pinned or not. + assert!( + estimate_in_group(&patterns[2], &pins, &HashSet::from([ann]), Some(&stats)).row_count() + < 10.0 + ); + } } diff --git a/fluree-db-query/src/stats_cache.rs b/fluree-db-query/src/stats_cache.rs index a6411fb6c9..b9a144d9fc 100644 --- a/fluree-db-query/src/stats_cache.rs +++ b/fluree-db-query/src/stats_cache.rs @@ -423,6 +423,7 @@ mod tests { classes: None, graphs: None, historical_since_t: None, + links: None, })); let mut novelty = Novelty::new(1); @@ -615,6 +616,7 @@ mod tests { classes: None, graphs: None, historical_since_t: since, + links: None, })); snapshot }; From 2d56587b27b83f508e3cfe7bfdee241d39c0c8ec Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 21:32:30 -0400 Subject: [PATCH 34/92] fix(index): the first terms over an index without a term dictionary An incremental build that met a ledger's first annotations over an index with no term dictionary handed the reverse-tree update a sentinel branch CID that was never stored. The read failed and the indexer fell back to a full rebuild, so every ledger indexed before its first annotation paid one. That case now writes a fresh tree, as a full build does. --- fluree-db-api/tests/it_triple_term_links.rs | 48 ++++++++++++++ fluree-db-binary-index/src/dict/term_dict.rs | 4 +- fluree-db-indexer/src/build/incremental.rs | 68 ++++++++++---------- 3 files changed, 85 insertions(+), 35 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 3e7dc1863a..8f32f35c30 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -1624,3 +1624,51 @@ async fn link_counts_follow_every_build() { }) .await; } + +/// A ledger indexed before its first annotation: the incremental build that +/// meets it starts the term dictionary over a base that has none. +#[tokio::test] +async fn first_annotation_after_an_index_without_terms() { + use fluree_db_indexer::IndexerConfig; + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + + let fluree = FlureeBuilder::memory() + .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) + .build_memory(); + let ledger_id = "it/triple-term-links:first-annotation-incremental"; + let (local, handle) = + support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (store, _guard) = support::span_capture::init_test_tracing(); + local + .run_until(async { + let ledger = fluree + .insert_turtle( + support::genesis_ledger(&fluree, ledger_id), + "@prefix ex: .\nex:alice ex:knows ex:bob .\n", + ) + .await + .expect("plain data") + .ledger; + support::trigger_index_and_wait(&handle, ledger_id, ledger.t()).await; + support::wait_for_index_application(&fluree, ledger_id, ledger.t()).await; + let ledger = fluree.ledger(ledger_id).await.expect("reload"); + let ledger = fluree + .insert_turtle(ledger, CLAIMS) + .await + .expect("claims") + .ledger; + let t = ledger.t(); + support::trigger_index_and_wait(&handle, ledger_id, t).await; + support::wait_for_index_application(&fluree, ledger_id, t).await; + let ledger = fluree.ledger(ledger_id).await.expect("reload"); + assert_eq!(ledger.index_t(), t); + assert_eq!(links(&fluree, &ledger).await.len(), 4); + }) + .await; + let fallbacks = store.find_events("incremental indexing failed, falling back to full rebuild"); + assert!( + fallbacks.is_empty(), + "{:?}", + fallbacks.iter().map(|e| &e.fields).collect::>() + ); +} diff --git a/fluree-db-binary-index/src/dict/term_dict.rs b/fluree-db-binary-index/src/dict/term_dict.rs index 216128f6bb..0b49683d4c 100644 --- a/fluree-db-binary-index/src/dict/term_dict.rs +++ b/fluree-db-binary-index/src/dict/term_dict.rs @@ -381,8 +381,8 @@ async fn put(cs: &dyn ContentStore, kind: ContentKind, bytes: &[u8]) -> io::Resu cs.put(kind, bytes).await.map_err(io::Error::other) } -/// Build, upload and finalize a reverse tree from key-sorted entries. -async fn upload_reverse_tree( +/// Build, upload and finalize a term reverse tree from key-sorted entries. +pub async fn upload_reverse_tree( cs: &dyn ContentStore, entries: Vec, ) -> io::Result { diff --git a/fluree-db-indexer/src/build/incremental.rs b/fluree-db-indexer/src/build/incremental.rs index 6b789fe0c3..514bfe10ea 100644 --- a/fluree-db-indexer/src/build/incremental.rs +++ b/fluree-db-indexer/src/build/incremental.rs @@ -1309,39 +1309,41 @@ pub async fn incremental_index( root_builder.set_term_dict(Some(refs), consumed); } } else { - let empty_tree = fluree_db_binary_index::DictTreeRefs { - branch: fluree_db_core::ContentId::from_hex_digest( - fluree_db_core::content_kind::CODEC_FLUREE_DICT_BLOB, - &fluree_db_core::sha256_hex(b""), - ) - .expect("valid digest"), - leaves: Vec::new(), - }; - let (base_reverse, base_count) = match &base_terms { - Some(b) => (b.reverse.clone(), b.term_count), - None => (empty_tree, 0), - }; - let updated_tree = if base_terms.is_some() { - super::dicts::upload_incremental_reverse_tree_async_terms( - content_store.as_ref(), - &base_reverse, - &novelty.new_terms, - warm_cache.as_deref(), - ) - .await? - } else { - // No base dictionary: build the tree from scratch through the - // same core, against an empty existing tree. - super::dicts::upload_incremental_reverse_tree_async_terms( - content_store.as_ref(), - &fluree_db_binary_index::DictTreeRefs { - branch: base_reverse.branch.clone(), - leaves: Vec::new(), - }, - &novelty.new_terms, - warm_cache.as_deref(), - ) - .await? + let base_count = base_terms.as_ref().map_or(0, |b| b.term_count); + let updated_tree = match &base_terms { + Some(base) => { + super::dicts::upload_incremental_reverse_tree_async_terms( + content_store.as_ref(), + &base.reverse, + &novelty.new_terms, + warm_cache.as_deref(), + ) + .await? + } + // No base dictionary (the ledger's first terms meet an index + // that has none): a fresh tree, as a full build writes it. + None => { + let mut entries: Vec<_> = novelty + .new_terms + .iter() + .map(|(p_id, seq, key)| { + fluree_db_binary_index::dict::reverse_leaf::ReverseEntry { + key: key.clone(), + id: fluree_db_core::triple_term::term_handle(*p_id, *seq), + } + }) + .collect(); + entries.sort_by(|a, b| a.key.cmp(&b.key)); + super::types::UpdatedReverseTree { + tree_refs: fluree_db_binary_index::dict::term_dict::upload_reverse_tree( + content_store.as_ref(), + entries, + ) + .await + .map_err(|e| IndexerError::StorageWrite(e.to_string()))?, + replaced_cids: Vec::new(), + } + } }; let refs = fluree_db_binary_index::TermDictRefs { forward_packs, From bc919ac0b9a04d2fa5ab0658e5c7ddb56a9472e3 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 21:43:19 -0400 Subject: [PATCH 35/92] perf(query): read an object-bound triple term from an object-first tree The term dictionary's reverse tree is ordered subject-first, so a quoted pattern with a constant or bound object, `<< ?s :p :o >>`, walked the inner predicate's whole link interval (or every link, with a variable predicate) and checked each term's object. On the annotation benchmark dev slice that made S21 and S22 2.4 s against the bundle chain's ~100 ms, and P5, P8 and P13 three times slower. - The dictionary keeps a second reverse tree keyed object-first (o_type, o_key, p_id, s_id). Full builds write it, and incremental builds update it copy-on-write or write it fresh over a base with no dictionary. A dictionary written before it (term dictionary section version 1) keeps none until a reindex. - TermComponents anchors on a constant or bound object when the subject is not anchored: one range on the object, and the predicate when fixed. The lookup stamps `term-object`, falling back to the predicate scan without the tree. - The planner ranks an object-anchored TermComponents like a subject- anchored one, and unread-position elision keeps the pattern when its object is constant. Dev slice, link mode: S21 2.51 s -> 149 ms, S22 2.48 s -> 129 ms, P5, P8 and P13 ~145 -> ~50 ms; P10 19 -> 27 ms (one object-tree leaf read); answers unchanged. Full reindex +1.6 s. --- fluree-db-api/tests/it_triple_term_links.rs | 158 ++++++++++++++++ fluree-db-binary-index/src/dict/term_dict.rs | 176 +++++++++++++++--- .../src/format/index_root.rs | 48 ++++- .../src/format/wire_helpers.rs | 58 +++++- fluree-db-core/src/db.rs | 8 +- fluree-db-core/src/triple_term.rs | 37 ++++ fluree-db-indexer/src/build/dicts.rs | 44 +++-- fluree-db-indexer/src/build/incremental.rs | 80 ++++---- fluree-db-query/src/execute/where_plan.rs | 9 +- fluree-db-query/src/planner.rs | 13 +- fluree-db-query/src/term_components.rs | 169 ++++++++++++++--- 11 files changed, 687 insertions(+), 113 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 8f32f35c30..e8d1a672c5 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -1672,3 +1672,161 @@ async fn first_annotation_after_an_index_without_terms() { fallbacks.iter().map(|e| &e.fields).collect::>() ); } + +/// Object-bound reified triples answered by the object-first tree, with +/// what each object lookup stamped. +async fn object_bound_answers( + fluree: &fluree_db_api::Fluree, + ledger: &LedgerState, +) -> (Vec>>, Vec>) { + let queries = [ + "SELECT ?s ?src WHERE { << ?s ex:knows ex:bob >> ex:source ?src } ORDER BY ?s", + "SELECT (COUNT(*) AS ?n) WHERE { << ?s ?p ex:bob >> ex:source ?src }", + "SELECT ?src WHERE { << ?s ex:size 5 >> ex:source ?src }", + "SELECT ?o ?src WHERE { ex:alice ex:knows ?o . << ?s ex:knows ?o >> ex:source ?src } \ + ORDER BY ?o ?src", + ]; + // Register the stamp callsite before this thread's subscriber reads it. + run_link_query(fluree, ledger, queries[0].to_string()).await; + let (store, _guard) = support::span_capture::init_test_tracing(); + tracing::callsite::rebuild_interest_cache(); + let stamps = || -> Vec { + store + .find_events("fast-path outcome") + .iter() + .filter(|e| e.fields.get("site").map(String::as_str) == Some("term-object")) + .filter_map(|e| e.fields.get("outcome").cloned()) + .collect() + }; + let mut answers = Vec::new(); + let mut outcomes = Vec::new(); + for q in queries { + let before = stamps().len(); + answers.push(run_link_query(fluree, ledger, q.to_string()).await); + outcomes.push(stamps().split_off(before)); + } + (answers, outcomes) +} + +fn strings(rows: &[&[&str]]) -> Vec> { + rows.iter() + .map(|r| r.iter().map(std::string::ToString::to_string).collect()) + .collect() +} + +/// A constant or bound term object reads its terms from the object-first +/// reverse tree (a range on the object, and the predicate when fixed), after +/// import, beside novelty, after a reindex and after an incremental build. +#[tokio::test(flavor = "current_thread")] +async fn object_bound_terms_read_the_object_tree() { + use fluree_db_indexer::IndexerConfig; + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let expected = |extra: bool| { + let mut knows_bob = vec![vec!["ex:alice".to_string(), "ex:hr".to_string()]]; + if extra { + knows_bob.push(vec!["ex:erin".to_string(), "ex:web".to_string()]); + } + vec![ + knows_bob, + vec![vec![if extra { "2" } else { "1" }.to_string()]], + strings(&[&["ex:integer"]]), + if extra { + strings(&[ + &["ex:bob", "ex:hr"], + &["ex:bob", "ex:web"], + &["ex:carol", "ex:linkedin"], + ]) + } else { + strings(&[&["ex:bob", "ex:hr"], &["ex:carol", "ex:linkedin"]]) + }, + ] + }; + let erin = "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:erin ex:knows ex:bob {| ex:source ex:web |} .\n"; + let all_proceed = |outcomes: &[Vec]| { + for (i, per_query) in outcomes.iter().enumerate() { + assert!( + !per_query.is_empty(), + "query {i} never read the object tree" + ); + assert!( + per_query.iter().all(|o| o == "proceed"), + "query {i}: {per_query:?}" + ); + } + }; + + // Enough knows links elsewhere that the object range, not the + // predicate's link interval, drives. + let padding: String = std::iter::once("@prefix ex: .\n".to_string()) + .chain((0..50).map(|i| { + format!( + "ex:p{i} ex:knows ex:q{i} {{| ex:source ex:pad |}} .\n\ + ex:p{i} ex:size {} {{| ex:source ex:pad |}} .\n", + i + 100 + ) + })) + .collect(); + let alias = "it/triple-term-links:object-tree"; + let (fluree, ledger) = import( + &[ + ("claims.ttl", CLAIMS), + ("literals.ttl", LITERAL_CLAIMS), + ("padding.ttl", &padding), + ], + alias, + ) + .await; + let (answers, outcomes) = object_bound_answers(&fluree, &ledger).await; + assert_eq!(answers, expected(false)); + all_proceed(&outcomes); + + let ledger = fluree + .insert_turtle(ledger, erin) + .await + .expect("novelty") + .ledger; + let (answers, outcomes) = object_bound_answers(&fluree, &ledger).await; + assert_eq!(answers, expected(true), "novelty terms beside the tree"); + all_proceed(&outcomes); + + fluree + .reindex(alias, fluree_db_api::ReindexOptions::default()) + .await + .expect("reindex"); + let ledger = fluree.ledger(alias).await.expect("reload"); + let (answers, outcomes) = object_bound_answers(&fluree, &ledger).await; + assert_eq!(answers, expected(true), "after reindex"); + all_proceed(&outcomes); + + let fluree = FlureeBuilder::memory() + .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) + .build_memory(); + let ledger_id = "it/triple-term-links:object-tree-incremental"; + let (local, handle) = + support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + local + .run_until(async { + // The first build sees no annotation, so the next one writes the + // term dictionary over a base that has none. + let plain = "@prefix ex: .\nex:zed ex:knows ex:bob .\n"; + let mut ledger = support::genesis_ledger(&fluree, ledger_id); + for body in [plain, CLAIMS, LITERAL_CLAIMS, &padding, erin] { + ledger = fluree + .insert_turtle(ledger, body) + .await + .expect("insert") + .ledger; + let t = ledger.t(); + support::trigger_index_and_wait(&handle, ledger_id, t).await; + support::wait_for_index_application(&fluree, ledger_id, t).await; + ledger = fluree.ledger(ledger_id).await.expect("reload"); + assert_eq!(ledger.index_t(), t); + } + }) + .await; + let ledger = fluree.ledger(ledger_id).await.expect("reload"); + let (answers, outcomes) = object_bound_answers(&fluree, &ledger).await; + assert_eq!(answers, expected(true), "after incremental builds"); + all_proceed(&outcomes); +} diff --git a/fluree-db-binary-index/src/dict/term_dict.rs b/fluree-db-binary-index/src/dict/term_dict.rs index 0b49683d4c..43fe1b5816 100644 --- a/fluree-db-binary-index/src/dict/term_dict.rs +++ b/fluree-db-binary-index/src/dict/term_dict.rs @@ -50,6 +50,7 @@ fn next_subject(s_id: u64) -> [u8; TermKey::LEN] { pub struct TermDictReader { forward: BTreeMap, reverse: Option>, + object_reverse: Option>, watermarks: HashMap, term_count: u64, } @@ -87,9 +88,23 @@ impl TermDictReader { ) .await?, ); + let object_reverse = match &refs.object_reverse { + Some(tree) => Some( + DictTreeReader::from_refs_reusing( + &cs, + tree, + leaflet_cache, + Some(cache_dir), + prev.and_then(|p| p.object_reverse.as_ref()), + ) + .await?, + ), + None => None, + }; Ok(Self { forward, reverse, + object_reverse, watermarks: refs.watermarks.iter().copied().collect(), term_count: refs.term_count, }) @@ -187,6 +202,49 @@ impl TermDictReader { .collect() } + /// Every term whose base edge has object `(o_type, o_key)` (and predicate + /// `p_id`, when given), with its handle: one range over the object-first + /// tree. `None` when the dictionary predates that tree. + pub fn terms_with_object( + &self, + o_type: u16, + o_key: u64, + p_id: Option, + ) -> io::Result>> { + let Some(tree) = &self.object_reverse else { + return Ok(None); + }; + let mut start = [0u8; TermKey::LEN]; + start[..2].copy_from_slice(&o_type.to_be_bytes()); + start[2..10].copy_from_slice(&o_key.to_be_bytes()); + if let Some(p) = p_id { + start[10..14].copy_from_slice(&p.to_be_bytes()); + } + // Exclusive end: the key prefix plus one, at its last byte. + let prefix_len = if p_id.is_some() { 14 } else { 10 }; + let mut end = [0u8; TermKey::LEN]; + end[..prefix_len].copy_from_slice(&start[..prefix_len]); + let Some(last) = end[..prefix_len].iter().rposition(|b| *b != 0xFF) else { + return Ok(Some(Vec::new())); + }; + end[last] += 1; + end[last + 1..].fill(0); + tree.reverse_range_scan(&start, &end)? + .into_iter() + .map(|(bytes, handle)| { + TermKey::from_object_first_bytes(&bytes) + .map(|key| (key, handle)) + .ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + "term object key has the wrong width", + ) + }) + }) + .collect::>>() + .map(Some) + } + /// Every term of inner predicate `p_id`, with its handle, read through the /// forward packs in sequence order. pub fn terms_of_predicate(&self, p_id: u32) -> io::Result> { @@ -324,6 +382,7 @@ impl TermDictBuilder { // Group by predicate, ordered by seq, so each stream is contiguous from 0. let mut by_pred: BTreeMap> = BTreeMap::new(); let mut reverse_entries: Vec = Vec::with_capacity(self.map.len()); + let mut object_entries: Vec = Vec::with_capacity(self.map.len()); for (key, handle) in &self.map { let bytes = key.to_be_bytes(); by_pred @@ -334,6 +393,10 @@ impl TermDictBuilder { key: bytes.to_vec(), id: *handle, }); + object_entries.push(ReverseEntry { + key: key.to_object_first_bytes().to_vec(), + id: *handle, + }); } let mut forward_packs = Vec::with_capacity(by_pred.len()); @@ -367,12 +430,15 @@ impl TermDictBuilder { reverse_entries.sort_unstable_by(|a, b| a.key.cmp(&b.key)); let reverse = upload_reverse_tree(cs, reverse_entries).await?; + object_entries.sort_unstable_by(|a, b| a.key.cmp(&b.key)); + let object_reverse = upload_reverse_tree(cs, object_entries).await?; Ok(TermDictRefs { forward_packs, reverse, watermarks, term_count, + object_reverse: Some(object_reverse), }) } } @@ -484,6 +550,7 @@ mod tests { let reader = TermDictReader { forward: BTreeMap::from([(5, a), (9, b)]), reverse: None, + object_reverse: None, watermarks: HashMap::new(), term_count: 110, }; @@ -496,8 +563,6 @@ mod tests { /// terms, across leaves, at the edges of the key space too. #[test] fn terms_with_subject_reads_the_prefix_range() { - use crate::dict::builder::build_reverse_tree; - use crate::dict::reader::DictTreeReader; let keys = [ key(4, 9, 1), key(5, 2, 7), @@ -507,29 +572,7 @@ mod tests { key(6, 0, 0), key(u64::MAX, 1, 1), ]; - let mut entries: Vec = keys - .iter() - .enumerate() - .map(|(i, k)| ReverseEntry { - key: k.to_be_bytes().to_vec(), - id: i as u64 + 100, - }) - .collect(); - entries.sort_by(|a, b| a.key.cmp(&b.key)); - let built = build_reverse_tree(entries, 64).unwrap(); - let leaves = built - .leaves - .iter() - .zip(&built.branch.leaves) - .map(|(leaf, entry)| (entry.address.clone(), leaf.bytes.clone())) - .collect(); - assert!(built.branch.leaves.len() > 1, "the range must cross leaves"); - let reader = TermDictReader { - forward: BTreeMap::new(), - reverse: Some(Arc::new(DictTreeReader::from_memory(built.branch, leaves))), - watermarks: HashMap::new(), - term_count: keys.len() as u64, - }; + let reader = reader_over(&keys); let handles = |s: u64, p: Option| -> Vec { reader .terms_with_subject(s, p) @@ -549,6 +592,89 @@ mod tests { assert_eq!(handles(7, None), Vec::::new()); } + /// Both reverse trees over `keys`, with handle `100 + i`, small leaves so + /// ranges cross them. + fn reader_over(keys: &[TermKey]) -> TermDictReader { + use crate::dict::builder::build_reverse_tree; + use crate::dict::reader::DictTreeReader; + let tree = |encode: fn(&TermKey) -> [u8; TermKey::LEN]| { + let mut entries: Vec = keys + .iter() + .enumerate() + .map(|(i, k)| ReverseEntry { + key: encode(k).to_vec(), + id: i as u64 + 100, + }) + .collect(); + entries.sort_by(|a, b| a.key.cmp(&b.key)); + let built = build_reverse_tree(entries, 64).unwrap(); + assert!(built.branch.leaves.len() > 1, "the range must cross leaves"); + let leaves = built + .leaves + .iter() + .zip(&built.branch.leaves) + .map(|(leaf, entry)| (entry.address.clone(), leaf.bytes.clone())) + .collect(); + Some(Arc::new(DictTreeReader::from_memory(built.branch, leaves))) + }; + TermDictReader { + forward: BTreeMap::new(), + reverse: tree(TermKey::to_be_bytes), + object_reverse: tree(TermKey::to_object_first_bytes), + watermarks: HashMap::new(), + term_count: keys.len() as u64, + } + } + + /// An object prefix (and object + predicate prefix) selects exactly its + /// terms, at the top of the key space too. + #[test] + fn terms_with_object_reads_the_prefix_range() { + let keys = [ + key(4, 9, 1), + key(5, 2, 7), + key(5, 9, 1), + key(6, 9, 1), + key(3, 1, 1), + key(5, u32::MAX, u64::MAX), + key(9, u32::MAX, u64::MAX), + key(1, 0, 2), + ]; + let reader = reader_over(&keys); + let handles = |o: u64, p: Option| -> Vec { + let mut found: Vec = reader + .terms_with_object(OType::IRI_REF.as_u16(), o, p) + .unwrap() + .expect("object tree") + .into_iter() + .map(|(k, h)| { + assert_eq!(k, keys[(h - 100) as usize]); + h + }) + .collect(); + found.sort_unstable(); + found + }; + assert_eq!(handles(1, None), vec![100, 102, 103, 104]); + assert_eq!(handles(1, Some(9)), vec![100, 102, 103]); + assert_eq!(handles(7, None), vec![101]); + assert_eq!(handles(u64::MAX, None), vec![105, 106]); + assert_eq!(handles(u64::MAX, Some(u32::MAX)), vec![105, 106]); + assert_eq!(handles(3, None), Vec::::new()); + let other_type = reader + .terms_with_object(OType::XSD_STRING.as_u16(), 1, None) + .unwrap(); + assert_eq!(other_type, Some(Vec::new())); + let no_tree = TermDictReader { + object_reverse: None, + ..reader_over(&keys) + }; + assert!(no_tree + .terms_with_object(OType::IRI_REF.as_u16(), 1, None) + .unwrap() + .is_none()); + } + #[test] fn builder_continues_above_watermarks() { let mut b = TermDictBuilder::above_watermarks(&[(5, 41)]); diff --git a/fluree-db-binary-index/src/format/index_root.rs b/fluree-db-binary-index/src/format/index_root.rs index cd631a3a9d..62d6d21295 100644 --- a/fluree-db-binary-index/src/format/index_root.rs +++ b/fluree-db-binary-index/src/format/index_root.rs @@ -1319,13 +1319,15 @@ impl IndexRoot { ids.push(ann.reverse_branch_cid.clone()); } - // Triple-term dictionary: forward packs + reverse branch and leaves. + // Triple-term dictionary: forward packs + both reverse trees. if let Some(ref td) = self.term_dict { for (_, packs) in &td.forward_packs { ids.extend(packs.iter().map(|e| e.pack_cid.clone())); } - ids.push(td.reverse.branch.clone()); - ids.extend(td.reverse.leaves.iter().cloned()); + for tree in std::iter::once(&td.reverse).chain(&td.object_reverse) { + ids.push(tree.branch.clone()); + ids.extend(tree.leaves.iter().cloned()); + } } ids.sort(); @@ -1564,6 +1566,46 @@ mod tests { assert!(decoded.annotation_index.is_none()); } + /// The term dictionary section round-trips with both reverse trees, the + /// core metadata decoder walks past it, and GC sees every tree CID. + #[test] + fn fir6_round_trip_with_term_dict_trees() { + let cid = |tag: &[u8]| ContentId::new(fluree_db_core::ContentKind::Commit, tag); + let tree = |tag: &str| crate::format::wire_helpers::DictTreeRefs { + branch: cid(format!("{tag}-branch").as_bytes()), + leaves: vec![cid(format!("{tag}-leaf").as_bytes())], + }; + let mut root = minimal_root_v6(); + root.term_dict = Some(crate::format::wire_helpers::TermDictRefs { + forward_packs: vec![( + 7, + vec![crate::format::wire_helpers::PackBranchEntry { + first_id: 0, + last_id: 3, + pack_cid: cid(b"pack"), + }], + )], + reverse: tree("subject"), + watermarks: vec![(7, 3)], + term_count: 4, + object_reverse: Some(tree("object")), + }); + let bytes = root.encode(); + let decoded = IndexRoot::decode(&bytes).unwrap(); + assert_eq!(decoded.term_dict, root.term_dict); + fluree_db_core::LedgerSnapshot::from_root_bytes(&bytes).expect("core metadata decode"); + + let ids = root.all_cas_ids(); + for c in [ + "object-branch", + "object-leaf", + "subject-branch", + "subject-leaf", + ] { + assert!(ids.contains(&cid(c.as_bytes())), "{c} unreachable"); + } + } + #[test] fn fir6_round_trip_with_annotation_index() { let mut root = minimal_root_v6(); diff --git a/fluree-db-binary-index/src/format/wire_helpers.rs b/fluree-db-binary-index/src/format/wire_helpers.rs index 202393ecba..f3f12c4b86 100644 --- a/fluree-db-binary-index/src/format/wire_helpers.rs +++ b/fluree-db-binary-index/src/format/wire_helpers.rs @@ -138,6 +138,9 @@ pub struct TermDictRefs { pub watermarks: Vec<(u32, u32)>, /// Distinct terms in the dictionary. pub term_count: u64, + /// Reverse tree in object-first order (`TermKey::to_object_first_bytes`) + /// → handle. `None` for a dictionary written before it existed. + pub object_reverse: Option, } /// Per-graph specialty arena refs (numbig, vectors, spatial). @@ -408,20 +411,22 @@ pub(crate) fn read_dict_pack_refs(data: &[u8], pos: &mut usize) -> io::Result, refs: &TermDictRefs) { buf.push(TERM_DICT_REFS_VERSION); @@ -446,12 +451,19 @@ pub(crate) fn write_term_dict_refs(buf: &mut Vec, refs: &TermDictRefs) { buf.extend_from_slice(&wm.to_le_bytes()); } buf.extend_from_slice(&refs.term_count.to_le_bytes()); + match &refs.object_reverse { + Some(tree) => { + buf.push(1); + write_dict_tree_refs(buf, tree); + } + None => buf.push(0), + } } /// Read triple-term dictionary refs written by [`write_term_dict_refs`]. pub(crate) fn read_term_dict_refs(data: &[u8], pos: &mut usize) -> io::Result { let version = read_u8_at(data, pos)?; - if version != TERM_DICT_REFS_VERSION { + if !(1..=TERM_DICT_REFS_VERSION).contains(&version) { return Err(io::Error::new( io::ErrorKind::InvalidData, format!("term dict refs: unsupported version {version}"), @@ -484,11 +496,17 @@ pub(crate) fn read_term_dict_refs(data: &[u8], pos: &mut usize) -> io::Result= 2 && read_u8_at(data, pos)? != 0 { + Some(read_dict_tree_refs(data, pos)?) + } else { + None + }; Ok(TermDictRefs { forward_packs, reverse, watermarks, term_count, + object_reverse, }) } @@ -512,6 +530,38 @@ pub(crate) fn read_dict_tree_refs(data: &[u8], pos: &mut usize) -> io::Result std::io::Result /// Skip the triple-term dictionary section. Matches /// `write_term_dict_refs` in binary-index: version, per-predicate forward - /// packs, reverse tree refs, per-predicate watermarks, term count. + /// packs, reverse tree refs, per-predicate watermarks, term count, and + /// from version 2 the optional object reverse tree. fn skip_term_dict_refs(bytes: &[u8], pos: &mut usize) -> std::io::Result<()> { let version = read_u8(bytes, pos)?; - if version != 1 { + if !(1..=2).contains(&version) { return Err(std::io::Error::new( std::io::ErrorKind::InvalidData, format!("FIR6: unsupported term dict section version {version}"), @@ -871,6 +872,9 @@ fn decode_fir6_metadata(bytes: &[u8]) -> std::io::Result let _wm = read_u32(bytes, pos)?; } let _term_count = read_u64(bytes, pos)?; + if version >= 2 && read_u8(bytes, pos)? != 0 { + skip_dict_tree_refs(bytes, pos)?; + } Ok(()) } diff --git a/fluree-db-core/src/triple_term.rs b/fluree-db-core/src/triple_term.rs index 9a35d7a43e..c6c2d95d53 100644 --- a/fluree-db-core/src/triple_term.rs +++ b/fluree-db-core/src/triple_term.rs @@ -194,6 +194,33 @@ impl TermKey { pub fn subject_prefix(s_id: u64) -> [u8; 8] { s_id.to_be_bytes() } + + /// Big-endian object-first key bytes: `o_type, o_key, p_id, s_id`, the + /// object reverse tree's order, so an object-bound reified-triple + /// pattern is one key range. + #[inline] + pub fn to_object_first_bytes(&self) -> [u8; Self::LEN] { + let mut b = [0u8; Self::LEN]; + b[0..2].copy_from_slice(&self.o_type.as_u16().to_be_bytes()); + b[2..10].copy_from_slice(&self.o_key.to_be_bytes()); + b[10..14].copy_from_slice(&self.p_id.to_be_bytes()); + b[14..22].copy_from_slice(&self.s_id.to_be_bytes()); + b + } + + /// Decode a key written by [`Self::to_object_first_bytes`]. + #[inline] + pub fn from_object_first_bytes(b: &[u8]) -> Option { + if b.len() != Self::LEN { + return None; + } + Some(Self { + o_type: OType::from_u16(u16::from_be_bytes(b[0..2].try_into().ok()?)), + o_key: u64::from_be_bytes(b[2..10].try_into().ok()?), + p_id: u32::from_be_bytes(b[10..14].try_into().ok()?), + s_id: u64::from_be_bytes(b[14..22].try_into().ok()?), + }) + } } #[cfg(test)] @@ -281,5 +308,15 @@ mod tests { }; assert!(k2.to_be_bytes() > b); assert!(TermKey::from_be_bytes(&b[..10]).is_none()); + + let o = k.to_object_first_bytes(); + assert_eq!(TermKey::from_object_first_bytes(&o), Some(k)); + let other_object = TermKey { + o_key: 78, + s_id: 0, + ..k + }; + assert!(other_object.to_object_first_bytes() > o, "object first"); + assert!(TermKey::from_object_first_bytes(&o[..10]).is_none()); } } diff --git a/fluree-db-indexer/src/build/dicts.rs b/fluree-db-indexer/src/build/dicts.rs index 4306c3cb3c..65b32cffa6 100644 --- a/fluree-db-indexer/src/build/dicts.rs +++ b/fluree-db-indexer/src/build/dicts.rs @@ -61,25 +61,45 @@ pub(crate) async fn upload_incremental_reverse_tree_async_strings( /// Core async reverse tree upload: pre-fetch affected leaves, spawn_blocking /// for CoW update, async-upload new artifacts. -/// Triple-term reverse tree append: entries are `(encoded TermKey, handle)`. -pub(crate) async fn upload_incremental_reverse_tree_async_terms( - content_store: &dyn ContentStore, - existing_refs: &DictTreeRefs, +/// Reverse-tree entries for this window's new terms, `(p_id, seq, +/// subject-first TermKey bytes)`, keyed subject-first or object-first, sorted. +pub(crate) fn term_reverse_entries( new_terms: &[(u32, u32, Vec)], - warm_cache: Option<&LeafletCache>, -) -> Result { + object_first: bool, +) -> Result> { use fluree_db_binary_index::dict::reverse_leaf::ReverseEntry; - use fluree_db_core::triple_term::term_handle; + use fluree_db_core::triple_term::{term_handle, TermKey}; - let mut entries: Vec = new_terms + let mut entries = new_terms .iter() - .map(|(p_id, seq, key)| ReverseEntry { - key: key.clone(), - id: term_handle(*p_id, *seq), + .map(|(p_id, seq, key)| { + let key = if object_first { + TermKey::from_be_bytes(key) + .ok_or_else(|| IndexerError::StorageWrite("malformed term key".into()))? + .to_object_first_bytes() + .to_vec() + } else { + key.clone() + }; + Ok(ReverseEntry { + key, + id: term_handle(*p_id, *seq), + }) }) - .collect(); + .collect::>>()?; entries.sort_by(|a, b| a.key.cmp(&b.key)); + Ok(entries) +} +/// Triple-term reverse tree append, in either key order. +pub(crate) async fn upload_incremental_reverse_tree_async_terms( + content_store: &dyn ContentStore, + existing_refs: &DictTreeRefs, + new_terms: &[(u32, u32, Vec)], + object_first: bool, + warm_cache: Option<&LeafletCache>, +) -> Result { + let entries = term_reverse_entries(new_terms, object_first)?; upload_incremental_reverse_tree_core( content_store, fluree_db_core::DictKind::TermReverse, diff --git a/fluree-db-indexer/src/build/incremental.rs b/fluree-db-indexer/src/build/incremental.rs index 514bfe10ea..792bf881a0 100644 --- a/fluree-db-indexer/src/build/incremental.rs +++ b/fluree-db-indexer/src/build/incremental.rs @@ -1310,53 +1310,66 @@ pub async fn incremental_index( } } else { let base_count = base_terms.as_ref().map_or(0, |b| b.term_count); - let updated_tree = match &base_terms { - Some(base) => { - super::dicts::upload_incremental_reverse_tree_async_terms( - content_store.as_ref(), - &base.reverse, - &novelty.new_terms, - warm_cache.as_deref(), - ) - .await? - } - // No base dictionary (the ledger's first terms meet an index - // that has none): a fresh tree, as a full build writes it. - None => { - let mut entries: Vec<_> = novelty - .new_terms - .iter() - .map(|(p_id, seq, key)| { - fluree_db_binary_index::dict::reverse_leaf::ReverseEntry { - key: key.clone(), - id: fluree_db_core::triple_term::term_handle(*p_id, *seq), - } - }) - .collect(); - entries.sort_by(|a, b| a.key.cmp(&b.key)); - super::types::UpdatedReverseTree { - tree_refs: fluree_db_binary_index::dict::term_dict::upload_reverse_tree( - content_store.as_ref(), - entries, - ) - .await - .map_err(|e| IndexerError::StorageWrite(e.to_string()))?, - replaced_cids: Vec::new(), + // Each reverse tree is updated copy-on-write over the base's, or + // written fresh when there is no base dictionary (the ledger's + // first terms meet an index that has none). A base dictionary + // that predates the object tree keeps none until a rebuild. + let update_tree = |base: Option, + object_first: bool| { + let content_store = content_store.as_ref(); + let new_terms = &novelty.new_terms; + let warm_cache = warm_cache.as_deref(); + async move { + match base { + Some(base) => { + super::dicts::upload_incremental_reverse_tree_async_terms( + content_store, + &base, + new_terms, + object_first, + warm_cache, + ) + .await + } + None => Ok(super::types::UpdatedReverseTree { + tree_refs: + fluree_db_binary_index::dict::term_dict::upload_reverse_tree( + content_store, + super::dicts::term_reverse_entries(new_terms, object_first)?, + ) + .await + .map_err(|e| IndexerError::StorageWrite(e.to_string()))?, + replaced_cids: Vec::new(), + }), } } }; + let updated_tree = + update_tree(base_terms.as_ref().map(|b| b.reverse.clone()), false).await?; + let object_tree = match &base_terms { + Some(base) => match &base.object_reverse { + Some(tree) => Some(update_tree(Some(tree.clone()), true).await?), + None => None, + }, + None => Some(update_tree(None, true).await?), + }; + let mut replaced = updated_tree.replaced_cids; + let object_reverse = object_tree.map(|tree| { + replaced.extend(tree.replaced_cids); + tree.tree_refs + }); let refs = fluree_db_binary_index::TermDictRefs { forward_packs, reverse: updated_tree.tree_refs, watermarks: novelty.term_watermarks.clone(), term_count: base_count + novelty.new_terms.len() as u64, + object_reverse, }; tracing::debug!( new_terms = novelty.new_terms.len(), term_count = refs.term_count, "V6 Phase 3: triple-term dictionary updated" ); - let mut replaced = updated_tree.replaced_cids; replaced.extend(consumed); root_builder.set_term_dict(Some(refs), replaced); } @@ -5154,6 +5167,7 @@ mod compaction_tests { }, watermarks: vec![(7, 26), (9, 2)], term_count: 30, + object_reverse: None, }; let mut cycle = CompactionBudget::new(); diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index f6597fb35e..c83ce8f3f9 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -2380,9 +2380,14 @@ pub(crate) fn elide_unread_term_binds( } } // Constant components are also the link scan's filters; the - // pattern stays for a variable to bind or a subject to anchor. + // pattern stays for a variable to bind or an endpoint to + // anchor. let keep = tc.components().into_iter().any(|c| c.var().is_some()) - || matches!(tc.subject, crate::ir::Component::Node(_)); + || matches!(tc.subject, crate::ir::Component::Node(_)) + || matches!( + tc.object, + crate::ir::Component::Node(_) | crate::ir::Component::Literal(..) + ); if !keep { changed = true; } diff --git a/fluree-db-query/src/planner.rs b/fluree-db-query/src/planner.rs index 3546ad60dd..734321036b 100644 --- a/fluree-db-query/src/planner.rs +++ b/fluree-db-query/src/planner.rs @@ -1160,15 +1160,16 @@ pub fn estimate_pattern( row_count: estimate_branch_cardinality(patterns, stats), }, - // A bound term is one dictionary decode; a bound subject is a reverse- - // tree prefix range. Otherwise the link scan should bind the term - // first, so this ranks after any real scan. + // A bound term is one dictionary decode; a bound subject or object is + // a reverse-tree prefix range. Otherwise the link scan should bind the + // term first, so this ranks after any real scan. Pattern::TermComponents(tc) => { - let anchored = match &tc.subject { + let anchors = |c: &crate::ir::Component| match c { crate::ir::Component::Var(v) => bound_vars.contains(v), - crate::ir::Component::Node(_) => true, - _ => false, + crate::ir::Component::Node(_) | crate::ir::Component::Literal(..) => true, + crate::ir::Component::Any => false, }; + let anchored = anchors(&tc.subject) || anchors(&tc.object); let row_count = if bound_vars.contains(&tc.term) { HIGHLY_SELECTIVE } else if anchored { diff --git a/fluree-db-query/src/term_components.rs b/fluree-db-query/src/term_components.rs index 39616c1caf..04d0bd14d8 100644 --- a/fluree-db-query/src/term_components.rs +++ b/fluree-db-query/src/term_components.rs @@ -21,6 +21,7 @@ use crate::binding::{Batch, Binding}; use crate::context::ExecutionContext; use crate::error::{QueryError, Result}; +use crate::fast_path_outcome::{stamp_fast_path, FastPathFallback, FastPathOutcome}; use crate::ir::{Component, TermComponentsPattern}; use crate::operator::{BoxedOperator, Operator, OperatorState}; use crate::var_registry::VarId; @@ -252,6 +253,27 @@ impl TermComponentsOperator { } } + /// The object's `(o_type, o_key)` for an anchored lookup, from a constant + /// or a bound node or literal; `Err(())` when the bound value names no + /// interned object. + fn anchor_object( + &self, + row: &[Binding], + store: &BinaryIndexStore, + ctx: &ExecutionContext<'_>, + ) -> Result>> { + match &self.pattern.object { + Component::Node(_) | Component::Literal(..) => { + Ok(self.constants.and_then(|c| c.o).map(Ok)) + } + Component::Var(v) => match self.value(row, *v) { + Some(binding) => object_key(binding, store, ctx), + None => Ok(None), + }, + Component::Any => Ok(None), + } + } + /// The predicate's `p_id` when it is fixed for this row. fn fixed_predicate(&self, row: &[Binding], store: &BinaryIndexStore) -> Option { match &self.pattern.predicate { @@ -366,6 +388,10 @@ impl TermComponentsOperator { } } +/// Routing stamp for the object-first term lookup; a dictionary without the +/// object tree falls back to the predicate scan. +const OBJECT_SITE: &str = "term-object"; + /// A component no dictionary can name is a miss, not an error. fn missing(r: std::io::Result) -> std::io::Result> { match r { @@ -392,6 +418,47 @@ fn materialized_matches(component: &Component, position: usize, term: &TripleTer } } +/// A bound object's term-key encoding: `None` for a binding no term key can +/// be read from here (the lookup falls back to a scan), `Err(())` for one no +/// indexed term can carry. +fn object_key( + binding: &Binding, + store: &BinaryIndexStore, + ctx: &ExecutionContext<'_>, +) -> Result>> { + let iri_ref = OType::IRI_REF.as_u16(); + match binding { + Binding::EncodedSid { s_id, .. } => Ok(Some(Ok((iri_ref, *s_id)))), + Binding::Sid { .. } | Binding::IriMatch { .. } | Binding::Iri(_) => Ok(Some( + subject_id(binding, store, ctx)? + .map(|id| (iri_ref, id)) + .ok_or(()), + )), + Binding::Lit { val, dtc, .. } => { + let (dt, lang) = match dtc { + DatatypeConstraint::Explicit(dt) => (dt.clone(), None), + DatatypeConstraint::LangTag(tag) => ( + Sid::new( + fluree_vocab::namespaces::RDF, + fluree_vocab::rdf_names::LANG_STRING, + ), + Some(tag.as_ref()), + ), + }; + let key = missing(crate::binary_scan::term_object_key( + val, + &dt, + lang, + store, + ctx.dict_novelty.as_ref(), + )) + .map_err(|e| QueryError::from_io("term components: object", e))?; + Ok(Some(key.map(|(ot, k)| (ot.as_u16(), k)).ok_or(()))) + } + _ => Ok(None), + } +} + /// The interned `s_id` a subject binding names, or `None` when it names none /// (a literal, or an IRI no dictionary holds). fn subject_id( @@ -429,14 +496,17 @@ impl Operator for TermComponentsOperator { fn plan_details(&self) -> serde_json::Map { let bound = self.child.schema(); + let anchors = |c: &Component| match c { + Component::Node(_) | Component::Literal(..) => true, + Component::Var(v) => bound.contains(v), + Component::Any => false, + }; let access = if bound.contains(&self.pattern.term) { "term" - } else if match &self.pattern.subject { - Component::Node(_) => true, - Component::Var(v) => bound.contains(v), - _ => false, - } { + } else if anchors(&self.pattern.subject) { "subject" + } else if anchors(&self.pattern.object) { + "object" } else { "scan" }; @@ -483,6 +553,8 @@ impl Operator for TermComponentsOperator { let mut by_prefix: HashMap<(u64, Option), Arc>> = HashMap::new(); let mut by_predicate: HashMap, Arc>> = HashMap::new(); + let mut by_object: HashMap<(u16, u64, Option), Arc>> = + HashMap::new(); for row_idx in 0..input.len() { let mut row: Vec = (0..self.schema.len()) .map(|col| { @@ -558,29 +630,74 @@ impl Operator for TermComponentsOperator { }, // No interned subject: only novelty can match. Some(Err(())) => None, - None => match by_predicate.get(&p_id) { - Some(found) => Some(Arc::clone(found)), - None => { - let predicates: Vec = match p_id { - Some(p) => vec![p], - None => terms.predicates().collect(), - }; - let mut all = Vec::new(); - for p in predicates { - all.extend(terms.terms_of_predicate(p).map_err( - |e| { - QueryError::from_io( - "term components: scan", - e, - ) - }, - )?); + None => { + let by_obj = match self.anchor_object(&row, &store, ctx)? { + Some(Err(())) => Some(Arc::new(Vec::new())), + Some(Ok((o_type, o_key))) => { + match by_object.get(&(o_type, o_key, p_id)) { + Some(found) => Some(Arc::clone(found)), + None => { + let found = terms + .terms_with_object(o_type, o_key, p_id) + .map_err(|e| { + QueryError::from_io( + "term components: object", + e, + ) + })?; + stamp_fast_path( + OBJECT_SITE, + match found { + Some(_) => FastPathOutcome::Proceed, + None => FastPathOutcome::Fallback( + FastPathFallback::GateDeclined, + ), + }, + ); + found.map(|found| { + let found = Arc::new(found); + by_object.insert( + (o_type, o_key, p_id), + Arc::clone(&found), + ); + found + }) + } + } } - let found = Arc::new(all); - by_predicate.insert(p_id, Arc::clone(&found)); - Some(found) + None => None, + }; + match by_obj { + Some(found) => Some(found), + // No object anchor, or a dictionary + // without the object tree: scan. + None => match by_predicate.get(&p_id) { + Some(found) => Some(Arc::clone(found)), + None => { + let predicates: Vec = match p_id { + Some(p) => vec![p], + None => terms.predicates().collect(), + }; + let mut all = Vec::new(); + for p in predicates { + all.extend( + terms.terms_of_predicate(p).map_err( + |e| { + QueryError::from_io( + "term components: scan", + e, + ) + }, + )?, + ); + } + let found = Arc::new(all); + by_predicate.insert(p_id, Arc::clone(&found)); + Some(found) + } + }, } - }, + } }; if let Some(found) = found { candidates.extend(found.iter().map(|(key, handle)| { From 7f2a2b48eeb236a5712d83bc8f857c1db95b6757 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 22:14:36 -0400 Subject: [PATCH 36/92] feat(query): SPARQL 1.2 TRIPLE() and triple-term expressions TRIPLE(s, p, o), and `<<( s p o )>>` in expression position, failed at lowering. They now build the triple term a link holds. - The subject must be an IRI or blank node, the predicate an IRI, and the object any term, a triple term included; anything else leaves the result unbound. A literal argument keeps its datatype or language tag, which a constant in expression position drops: `TRIPLE(:s, :p, "5"^^xsd:int)` is not the term with `5`. - Components are read as bindings, encoded ones decoded first, and BIND binds the term as a value, so isTRIPLE, SUBJECT, PREDICATE and OBJECT apply to it. - encoded_equivalent composes a term's handle, so a constructed term joins and groups with the handle a link scan binds; before it, BIND(TRIPLE(..) AS ?t) followed by `?r rdf:reifies ?t` matched nothing. --- fluree-db-api/tests/it_triple_term_links.rs | 84 +++++++++++++++ fluree-db-query/src/eval.rs | 8 ++ fluree-db-query/src/eval/dispatch.rs | 1 + fluree-db-query/src/eval/rdf.rs | 109 ++++++++++++++++++++ fluree-db-query/src/ir/expression.rs | 2 + fluree-db-query/src/object_binding.rs | 6 ++ fluree-db-sparql/src/lower/expression.rs | 45 +++++--- fluree-db-sparql/src/lower/term.rs | 2 +- 8 files changed, 243 insertions(+), 14 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index e8d1a672c5..d18d0fef00 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -1830,3 +1830,87 @@ async fn object_bound_terms_read_the_object_tree() { assert_eq!(answers, expected(true), "after incremental builds"); all_proceed(&outcomes); } + +/// `TRIPLE(s, p, o)` builds the term a link holds: it joins, filters and +/// groups with links whether the index or novelty holds them, a literal +/// component keeps its datatype or tag, and a component of the wrong kind +/// leaves the result unbound. +#[tokio::test] +async fn triple_constructs_the_terms_links_hold() { + let (fluree, ledger) = import( + &[("claims.ttl", CLAIMS), ("literals.ttl", LITERAL_CLAIMS)], + "it/triple-term-links:triple-fn", + ) + .await; + let check = |ledger: LedgerState| { + let fluree = &fluree; + async move { + let run = |body: &str| run_link_query(fluree, &ledger, body.to_string()); + for body in [ + "SELECT ?r WHERE { BIND(TRIPLE(ex:alice, ex:knows, ex:bob) AS ?t) ?r rdf:reifies ?t }", + "SELECT ?r WHERE { ?r rdf:reifies ?t FILTER(?t = TRIPLE(ex:alice, ex:knows, ex:bob)) }", + "SELECT ?r WHERE { BIND(<<( ex:alice ex:knows ex:bob )>> AS ?t) ?r rdf:reifies ?t }", + ] { + assert_eq!(run(body).await, strings(&[&["ex:claim1"]]), "{body}"); + } + for (object, source) in [ + ("\"5\"^^xsd:int", "ex:int"), + ("5", "ex:integer"), + ("\"chat\"@fr", "ex:fr"), + ] { + let body = format!( + "SELECT ?src WHERE {{ BIND(TRIPLE(ex:doc, ?p, {object}) AS ?t) \ + ?r rdf:reifies ?t ; ex:source ?src \ + VALUES ?p {{ ex:size ex:title }} }}" + ); + assert_eq!(run(&body).await, strings(&[&[source]]), "{object}"); + } + // The constructed term and the link's are one group. + let got = run( + "SELECT (COUNT(*) AS ?n) WHERE { { ?r rdf:reifies ?t } UNION \ + { BIND(TRIPLE(ex:alice, ex:knows, ex:bob) AS ?t) } } \ + GROUP BY ?t HAVING (COUNT(*) > 1)", + ) + .await; + assert_eq!(got, strings(&[&["2"]]), "{got:?}"); + } + }; + check(ledger.clone()).await; + let ledger = fluree + .insert_turtle( + ledger, + "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:erin ex:knows ex:frank ~ ex:claim9 {| ex:source ex:web |} .\n", + ) + .await + .expect("novelty") + .ledger; + check(ledger.clone()).await; + let got = run_link_query( + &fluree, + &ledger, + "SELECT ?r WHERE { BIND(TRIPLE(ex:erin, ex:knows, ex:frank) AS ?t) ?r rdf:reifies ?t }" + .to_string(), + ) + .await; + assert_eq!(got, strings(&[&["ex:claim9"]]), "a term only novelty holds"); + + let got = run_link_query( + &fluree, + &ledger, + "SELECT (isTRIPLE(?t) AS ?is) (SUBJECT(?t) AS ?s) (DATATYPE(OBJECT(?t)) AS ?dt) \ + (isTRIPLE(OBJECT(?n)) AS ?nested) (BOUND(?bad1) AS ?b1) (BOUND(?bad2) AS ?b2) \ + (BOUND(?bad3) AS ?b3) WHERE { \ + BIND(TRIPLE(ex:a, ex:b, \"5\"^^xsd:int) AS ?t) \ + BIND(TRIPLE(ex:r, ex:says, ?t) AS ?n) \ + BIND(TRIPLE(\"lit\", ex:b, ex:c) AS ?bad1) \ + BIND(TRIPLE(ex:a, \"x\", ex:c) AS ?bad2) \ + BIND(TRIPLE(ex:a, BNODE(), ex:c) AS ?bad3) }" + .to_string(), + ) + .await; + assert_eq!( + got, + strings(&[&["true", "ex:a", "xsd:int", "true", "false", "false", "false"]]) + ); +} diff --git a/fluree-db-query/src/eval.rs b/fluree-db-query/src/eval.rs index 1c6e105f16..aa9853a327 100644 --- a/fluree-db-query/src/eval.rs +++ b/fluree-db-query/src/eval.rs @@ -292,6 +292,14 @@ impl Expression { return Ok((**b).clone()); } + if let Expression::Call { + func: Function::Triple, + args, + } = self + { + return Ok(rdf::triple_binding(args, row, ctx)?.unwrap_or(Binding::Unbound)); + } + // A term accessor on a late-materialized handle binds the component // encoded, without materializing the term. if let Expression::Call { diff --git a/fluree-db-query/src/eval/dispatch.rs b/fluree-db-query/src/eval/dispatch.rs index c6164dc312..bd4ad2c6be 100644 --- a/fluree-db-query/src/eval/dispatch.rs +++ b/fluree-db-query/src/eval/dispatch.rs @@ -117,6 +117,7 @@ impl Function { Function::TriplePredicate => rdf::eval_triple_predicate(args, row, ctx), Function::TripleObject => rdf::eval_triple_object(args, row, ctx), Function::IsTriple => rdf::eval_is_triple(args, row, ctx), + Function::Triple => rdf::eval_triple(args, row, ctx), Function::Iri => rdf::eval_iri(args, row, ctx), Function::Bnode => rdf::eval_bnode(args, row, ctx), diff --git a/fluree-db-query/src/eval/rdf.rs b/fluree-db-query/src/eval/rdf.rs index dd9efcf585..43ead7bc63 100644 --- a/fluree-db-query/src/eval/rdf.rs +++ b/fluree-db-query/src/eval/rdf.rs @@ -681,6 +681,115 @@ pub fn eval_triple_object( eval_term_accessor(Function::TripleObject, args, row, ctx, "OBJECT") } +/// SPARQL 1.2 `TRIPLE(s, p, o)` as a binding: the triple term, or `None` +/// when a component is unbound or of the wrong kind (a subject that is not an +/// IRI or blank node, a predicate that is not an IRI). Each component keeps +/// its binding's identity, so an object keeps its datatype or language tag. +pub(crate) fn triple_binding( + args: &[Expression], + row: &R, + ctx: Option<&ExecutionContext<'_>>, +) -> Result> { + check_arity(args, 3, "TRIPLE")?; + let mut parts = Vec::with_capacity(3); + for arg in args { + let binding = match arg { + Expression::Var(v) => row.get(*v).cloned().unwrap_or(Binding::Unbound), + other => other.try_eval_to_binding(row, ctx)?, + }; + parts.push(materialized_component(binding, ctx)?); + } + let node = |b: &Binding| -> Option { + match b { + Binding::Sid { sid, .. } => Some(sid.clone()), + Binding::IriMatch { primary_sid, .. } => Some(primary_sid.clone()), + Binding::Iri(iri) => ctx.and_then(|c| c.encode_iri(iri)), + _ => None, + } + }; + let Some(s) = node(&parts[0]) else { + return Ok(None); + }; + let Some(p) = + node(&parts[1]).filter(|p| p.namespace_code != fluree_vocab::namespaces::BLANK_NODE) + else { + return Ok(None); + }; + let (o, dt, lang) = match &parts[2] { + Binding::Lit { val, dtc, .. } => match dtc { + fluree_db_core::DatatypeConstraint::Explicit(dt) => (val.clone(), dt.clone(), None), + fluree_db_core::DatatypeConstraint::LangTag(tag) => ( + val.clone(), + Sid::new( + fluree_vocab::namespaces::RDF, + fluree_vocab::rdf_names::LANG_STRING, + ), + Some(tag.to_string()), + ), + }, + other => match node(other) { + Some(sid) => ( + fluree_db_core::FlakeValue::Ref(sid), + fluree_db_core::edge::id_datatype_sid(), + None, + ), + None => return Ok(None), + }, + }; + let term = fluree_db_core::TripleTermValue { s, p, o, dt, lang }; + Ok(Some(Binding::Lit { + val: fluree_db_core::FlakeValue::TripleTerm(Box::new(term)), + dtc: fluree_db_core::DatatypeConstraint::Explicit( + fluree_db_core::triple_term_datatype_sid().clone(), + ), + t: None, + op: None, + p_id: None, + })) +} + +/// A component binding in decoded form: an encoded term through the +/// dictionaries, any other encoded value through the graph view. +fn materialized_component(binding: Binding, ctx: Option<&ExecutionContext<'_>>) -> Result { + match &binding { + Binding::EncodedLit { o_kind, .. } + if *o_kind == fluree_db_core::value_id::ObjKind::TRIPLE_TERM.as_u8() => + { + Ok(match super::binding_to_comparable(Some(&binding), ctx)? { + Some(ComparableValue::TypedLiteral { val, .. }) => Binding::Lit { + val, + dtc: fluree_db_core::DatatypeConstraint::Explicit( + fluree_db_core::triple_term_datatype_sid().clone(), + ), + t: None, + op: None, + p_id: None, + }, + _ => Binding::Unbound, + }) + } + Binding::EncodedLit { .. } | Binding::EncodedSid { .. } | Binding::EncodedPid { .. } => { + let gv = ctx.and_then(ExecutionContext::graph_view); + Ok(crate::group_aggregate::materialize_encoded( + &binding, + gv.as_ref(), + )) + } + _ => Ok(binding), + } +} + +pub fn eval_triple( + args: &[Expression], + row: &R, + ctx: Option<&ExecutionContext<'_>>, +) -> Result> { + match triple_binding(args, row, ctx)? { + Some(binding) => super::binding_to_comparable(Some(&binding), ctx), + None => Ok(None), + } +} + pub fn eval_is_triple( args: &[Expression], row: &R, diff --git a/fluree-db-query/src/ir/expression.rs b/fluree-db-query/src/ir/expression.rs index a714fac995..4e6c7d61d0 100644 --- a/fluree-db-query/src/ir/expression.rs +++ b/fluree-db-query/src/ir/expression.rs @@ -983,6 +983,8 @@ pub enum Function { TripleObject, /// SPARQL 1.2 `isTRIPLE(term)`. IsTriple, + /// SPARQL 1.2 `TRIPLE(s, p, o)`: the triple term with those components. + Triple, // ========================================================================= // Fluree-specific functions diff --git a/fluree-db-query/src/object_binding.rs b/fluree-db-query/src/object_binding.rs index 2c179b53df..3e63fc7cf5 100644 --- a/fluree-db-query/src/object_binding.rs +++ b/fluree-db-query/src/object_binding.rs @@ -343,6 +343,12 @@ pub(crate) fn encoded_equivalent(binding: &Binding, store: &BinaryIndexStore) -> 0, ) } + // A link scan binds a term as its handle. + (FlakeValue::TripleTerm(term), _) => { + let (_, handle) = + crate::binary_scan::compose_term_handle(term, store, None).ok()?; + (ObjKind::TRIPLE_TERM.as_u8(), handle, 0, 0) + } _ => return None, }; Some(Binding::EncodedLit { diff --git a/fluree-db-sparql/src/lower/expression.rs b/fluree-db-sparql/src/lower/expression.rs index 9471020cd3..33e2abda01 100644 --- a/fluree-db-sparql/src/lower/expression.rs +++ b/fluree-db-sparql/src/lower/expression.rs @@ -227,17 +227,38 @@ impl LoweringContext<'_, E> { &mut self, name: &FunctionName, args: &[AstExpression], - span: SourceSpan, + _span: SourceSpan, ) -> Result { - // SPARQL 1.2 triple-term functions are accepted and arity-validated at - // parse time, but have no evaluable implementation yet: defer per - // burn-down decision D-1 (accept-then-defer). A query that reaches here - // fails at lower time with a clean `not_implemented`, not a parse error. + // A literal component keeps its datatype or language tag, which a + // constant in expression position would drop: `TRIPLE(:s, :p, + // "5"^^xsd:int)` names a different term from the one with `5`. if matches!(name, FunctionName::Triple) { - return Err(LowerError::not_implemented( - "SPARQL 1.2 TRIPLE(s, p, o) construction", - span, - )); + let args = args + .iter() + .map(|a| match a { + AstExpression::Literal(lit) => { + match self.lower_literal_with_constraint(lit)? { + (fluree_db_query::ir::Term::Value(val), Some(dtc)) => { + Ok(Expression::Resolved(Box::new( + fluree_db_query::binding::Binding::Lit { + val, + dtc, + t: None, + op: None, + p_id: None, + }, + ))) + } + _ => self.lower_expression(a), + } + } + other => self.lower_expression(other), + }) + .collect::>>()?; + return Ok(Expression::Call { + func: Function::Triple, + args, + }); } let func = match name { @@ -325,10 +346,8 @@ impl LoweringContext<'_, E> { FunctionName::Predicate => Function::TriplePredicate, FunctionName::Object => Function::TripleObject, FunctionName::IsTriple => Function::IsTriple, - // Handled by the `not_implemented` early return above. - FunctionName::Triple => { - unreachable!("TRIPLE() defers via the early return") - } + // Handled by the early return above. + FunctionName::Triple => unreachable!("TRIPLE() lowers above"), // Extension functions FunctionName::Extension(iri) => { diff --git a/fluree-db-sparql/src/lower/term.rs b/fluree-db-sparql/src/lower/term.rs index ecd31bb931..66d7ba3714 100644 --- a/fluree-db-sparql/src/lower/term.rs +++ b/fluree-db-sparql/src/lower/term.rs @@ -229,7 +229,7 @@ impl LoweringContext<'_, E> { /// Like [`Self::lower_literal`] but also returns the /// datatype/language constraint that pins the lexical value to a /// specific RDF datatype or language tag. - fn lower_literal_with_constraint( + pub(super) fn lower_literal_with_constraint( &mut self, lit: &Literal, ) -> Result<(Term, Option)> { From 003d7d80714431ca91830d5542bdd5d53c33e243 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 22:28:17 -0400 Subject: [PATCH 37/92] fix(ledger): a staged view reads the links of its own annotations Novelty derives links when a commit is applied, but the staged view that SHACL validation and post-state policy read (StagedLedger) layers the staged flakes over novelty without applying them, so a quoted pattern under the link lowering missed the transaction's own annotations: a sh:sparql constraint over `<< $this :knows :mallory >>` let the violating annotation commit. The bundle chain caught it. StagedLedger now asks the base novelty for the links its staged attachment ops derive (Novelty::links_for, the commit-time derivation without the commit) and overlays them beside the staged flakes. They are read-only: staged_flakes() and into_parts() return only the staged flakes, so a commit never carries a link. Previews (GraphDb::from_staged) already applied the staged flakes to a novelty clone and derived their links; a test now pins that too. --- fluree-db-api/tests/it_triple_term_links.rs | 85 ++++++++++++++ fluree-db-ledger/src/staged.rs | 121 ++++++++++++++++---- fluree-db-novelty/src/links.rs | 17 +++ 3 files changed, 199 insertions(+), 24 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index d18d0fef00..71d5349689 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -1914,3 +1914,88 @@ async fn triple_constructs_the_terms_links_hold() { strings(&[&["true", "ex:a", "xsd:int", "true", "false", "false", "false"]]) ); } + +/// A SHACL SPARQL constraint over a quoted pattern validates the post-state +/// with the transaction's own annotations in it, as a committed read would. +#[cfg(feature = "shacl")] +#[tokio::test] +async fn shacl_sparql_constraints_see_the_transactions_links() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let fluree = FlureeBuilder::memory().build_memory(); + let ledger = support::genesis_ledger(&fluree, "it/triple-term-links:shacl-staged"); + let ledger = fluree + .insert_turtle( + ledger, + r#"@prefix sh: . +@prefix ex: . +ex:AliceShape a sh:NodeShape ; + sh:targetNode ex:alice ; + sh:sparql ex:noMallory . +ex:noMallory sh:message "no claim may say alice knows mallory" ; + sh:select "SELECT $this WHERE { << $this >> ?src }" . +"#, + ) + .await + .expect("shape") + .ledger; + let claim = |o: &str| { + format!( + "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:alice ex:knows ex:{o} {{| ex:source ex:web |}} .\n" + ) + }; + let ledger = fluree + .insert_turtle(ledger, &claim("bob")) + .await + .expect("a conforming claim") + .ledger; + let rejected = fluree.insert_turtle(ledger, &claim("mallory")).await; + assert!( + rejected.is_err(), + "the staged annotation must violate the shape" + ); +} + +/// A preview of a staged transaction reads the links of its own +/// annotations, over an index and over a ledger never indexed. +#[tokio::test] +async fn previews_read_the_links_of_their_own_annotations() { + std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); + let (fluree, indexed) = import(&[("claims.ttl", CLAIMS)], "it/triple-term-links:preview").await; + let memory = FlureeBuilder::memory().build_memory(); + let never_indexed = memory + .insert_turtle( + support::genesis_ledger(&memory, "it/triple-term-links:preview-memory"), + CLAIMS, + ) + .await + .expect("claims") + .ledger; + for (fluree, ledger) in [(&fluree, indexed), (&memory, never_indexed)] { + let staged = fluree + .stage_owned(ledger) + .upsert_turtle( + "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:erin ex:knows ex:frank ~ ex:claim9 {| ex:source ex:web |} .\n", + ) + .stage() + .await + .expect("stage"); + let preview = fluree_db_api::GraphDb::from_staged(&staged).expect("preview"); + let result = fluree + .query( + &preview, + "PREFIX ex: \n\ + SELECT ?s WHERE { << ?s ex:knows ex:frank >> ex:source ex:web }", + ) + .await + .expect("preview query"); + let got = rows( + &result + .to_jsonld_async(preview.as_graph_db_ref()) + .await + .expect("format"), + ); + assert_eq!(got, strings(&[&["ex:erin"]])); + } +} diff --git a/fluree-db-ledger/src/staged.rs b/fluree-db-ledger/src/staged.rs index e9900eee28..b3a39952a7 100644 --- a/fluree-db-ledger/src/staged.rs +++ b/fluree-db-ledger/src/staged.rs @@ -39,16 +39,17 @@ impl StagedStore { fn len(&self) -> usize { self.flakes.len() } - - fn is_empty(&self) -> bool { - self.flakes.is_empty() - } } /// Staged overlay - maintains sorted vectors like Novelty, with per-flake graph IDs /// for efficient graph filtering. struct StagedOverlay { + /// The staged flakes, then the links they derive (see + /// [`StagedOverlay::with_links`]). store: StagedStore, + /// How many of `store.flakes` are staged; the rest are derived links, + /// which reads see and the commit never carries. + staged_len: usize, /// Pre-computed GraphId per flake (parallel to store.flakes, same indices) flake_graph_ids: Vec, spot: Vec, @@ -65,17 +66,6 @@ impl StagedOverlay { staged_t: i64, reverse_graph: &HashMap, ) -> Result { - if flakes.is_empty() { - return Ok(Self { - store: StagedStore::new(vec![]), - flake_graph_ids: vec![], - spot: vec![], - psot: vec![], - post: vec![], - opst: vec![], - }); - } - // Single pass: stamp `t` and pre-compute graph IDs for all flakes — // strict, no silent fallback. Unknown graph Sids are a programming // error (reverse_graph is built from build_reverse_graph() which is @@ -94,6 +84,29 @@ impl StagedOverlay { flake_graph_ids.push(g_id); } + Ok(Self::indexed(flakes, flake_graph_ids)) + } + + /// Add the links the staged attachment ops derive, each with its graph. + fn with_links(self, links: Vec<(GraphId, Flake)>) -> Self { + if links.is_empty() { + return self; + } + let staged_len = self.staged_len; + let mut flakes = self.store.flakes; + let mut flake_graph_ids = self.flake_graph_ids; + for (g_id, link) in links { + flake_graph_ids.push(g_id); + flakes.push(link); + } + Self { + staged_len, + ..Self::indexed(flakes, flake_graph_ids) + } + } + + fn indexed(flakes: Vec, flake_graph_ids: Vec) -> Self { + let staged_len = flakes.len(); let store = StagedStore::new(flakes); let ids: Vec = (0..store.len() as FlakeId).collect(); @@ -111,14 +124,15 @@ impl StagedOverlay { let mut opst = ids; opst.sort_by(|&a, &b| IndexType::Opst.compare(store.get(a), store.get(b))); - Ok(Self { + Self { store, + staged_len, flake_graph_ids, spot, psot, post, opst, - }) + } } fn get_index(&self, index: IndexType) -> &[FlakeId] { @@ -222,8 +236,16 @@ impl StagedLedger { ) -> Result { let staged_epoch = base.novelty.epoch + 1; let staged_t = base.t() + 1; + let staged = StagedOverlay::from_flakes(flakes, staged_t, reverse_graph)?; + let links = base.novelty.links_for( + staged + .flake_graph_ids + .iter() + .copied() + .zip(&staged.store.flakes), + )?; Ok(Self { - staged: StagedOverlay::from_flakes(flakes, staged_t, reverse_graph)?, + staged: staged.with_links(links), staged_epoch, content_version: fluree_db_core::overlay::next_overlay_content_version(), dicts_cover_staged: false, @@ -270,17 +292,17 @@ impl StagedLedger { /// Get the number of staged flakes pub fn staged_len(&self) -> usize { - self.staged.store.len() + self.staged.staged_len } /// Check if there are staged flakes pub fn has_staged(&self) -> bool { - !self.staged.store.is_empty() + self.staged.staged_len > 0 } /// Get a reference to the staged flakes pub fn staged_flakes(&self) -> &[Flake] { - &self.staged.store.flakes + &self.staged.store.flakes[..self.staged.staged_len] } /// Each staged flake with the ledger graph id staging routed it to. @@ -289,7 +311,7 @@ impl StagedLedger { .flake_graph_ids .iter() .copied() - .zip(&self.staged.store.flakes) + .zip(self.staged_flakes()) } /// Get a reference to the underlying database @@ -299,7 +321,9 @@ impl StagedLedger { /// Consume the view and return the base state and staged flakes pub fn into_parts(self) -> (LedgerState, Vec) { - (self.base, self.staged.store.flakes) + let mut flakes = self.staged.store.flakes; + flakes.truncate(self.staged.staged_len); + (self.base, flakes) } /// The effective as-of time for this staged view. @@ -455,10 +479,59 @@ mod tests { assert!(!a.dicts_cover_staged()); } + /// The links a staged attachment derives are read through the view, and + /// the flakes a commit takes from it are only the staged ones. + #[test] + fn staged_attachments_overlay_links_the_commit_never_carries() { + use fluree_db_core::namespaces::{ + reifies_object_sid, reifies_predicate_sid, reifies_subject_sid, + }; + use fluree_db_core::{is_rdf_reifies, LedgerSnapshot}; + + let state = LedgerState::new(LedgerSnapshot::genesis("test:main"), Novelty::new(0)); + let reifier = Sid::new(100, "r"); + let slot = |p: &Sid, o: FlakeValue, dt: Sid| { + Flake::new(reifier.clone(), p.clone(), o, dt, 1, true, None) + }; + let bundle = vec![ + slot( + reifies_subject_sid(), + FlakeValue::Ref(Sid::new(100, "a")), + fluree_db_core::edge::id_datatype_sid(), + ), + slot( + reifies_predicate_sid(), + FlakeValue::Ref(Sid::new(100, "p")), + fluree_db_core::edge::id_datatype_sid(), + ), + slot( + reifies_object_sid(), + FlakeValue::String("x".into()), + fluree_db_core::edge::xsd_string_datatype_sid(), + ), + ]; + let view = StagedLedger::new(state, bundle, &HashMap::new()).unwrap(); + assert_eq!(view.staged_len(), 3); + assert_eq!(view.staged_flakes().len(), 3); + assert_eq!(view.staged_flakes_by_graph().count(), 3); + let mut links = Vec::new(); + view.for_each_overlay_flake(0, IndexType::Psot, None, None, true, i64::MAX, &mut |f| { + if is_rdf_reifies(&f.p) { + links.push(f.clone()); + } + }); + assert_eq!(links.len(), 1, "{links:?}"); + assert_eq!(links[0].s, reifier); + assert_eq!(links[0].t, 1); + let (_, flakes) = view.into_parts(); + assert_eq!(flakes.len(), 3); + assert!(!flakes.iter().any(|f| is_rdf_reifies(&f.p))); + } + #[test] fn test_staged_overlay_empty() { let staged = StagedOverlay::from_flakes(vec![], 1, &HashMap::new()).unwrap(); - assert!(staged.store.is_empty()); + assert!(staged.store.flakes.is_empty()); } #[test] diff --git a/fluree-db-novelty/src/links.rs b/fluree-db-novelty/src/links.rs index 367843d3af..bdceb7aa7e 100644 --- a/fluree-db-novelty/src/links.rs +++ b/fluree-db-novelty/src/links.rs @@ -113,6 +113,23 @@ impl Novelty { Ok(derived.into_iter().map(|(_, f)| f).collect()) } + /// The links a batch of routed flakes would derive if committed next, + /// without applying it, each with its graph. Empty while this novelty's + /// links wait for an index. + pub fn links_for<'a>( + &self, + batch: impl IntoIterator, + ) -> Result> { + let Some(base) = self.current_link_base() else { + return Ok(Vec::new()); + }; + let touched = touched_reifiers(batch); + if touched.is_empty() { + return Ok(Vec::new()); + } + self.derive_links(base, &touched, self.t) + } + /// The base links are derived from, when every earlier commit has its /// links; `None` leaves the next commit's links pending. pub(crate) fn current_link_base(&self) -> Option<&LinkBase> { From e978ccf4a04568d6292e6ee40e24bbbf28527a62 Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 23:52:11 -0400 Subject: [PATCH 38/92] feat(api): render triple terms in every result format A triple term (a link's object, or a TRIPLE() result) rendered as its debug form, `<<( [13:alice] [13:knows] ref:[13:bob] )>>` typed f:tripleTerm, in every format. Each format now writes it as a term: - SPARQL JSON: the SPARQL 1.2 `triple` term, {"type": "triple", "value": {"subject", "predicate", "object"}}, from the DOM and streaming writers. - SPARQL XML: . - TSV/CSV: <<( s p o )>>, components written as the cell writer writes them, a string object quoted. - JSON-LD, typed JSON and agent JSON: JSON-LD 1.1 has no triple terms, so a JSON-LD-star embedded node, {"@id": {"@id": s, p: o}}, its object rendered as that format renders any value. - CONSTRUCT: `?r rdf:reifies ?t`, with a triple term bound to ?t, writes what `?r rdf:reifies <<( s p o )>>` writes: the term's triple and the reification. The graph IR holds triple terms only as reifications, so under any other predicate the term is a literal holding its N-Triples text. Crawl output keeps its literal arms: neither the wildcard nor an explicit selection surfaces rdf:reifies on a reifier, so no crawl reaches one. --- docs/query/construct.md | 4 +- docs/query/output-formats.md | 23 +++ fluree-db-api/src/format/construct.rs | 148 +++++++++++++-- fluree-db-api/src/format/delimited.rs | 21 ++- fluree-db-api/src/format/jsonld.rs | 13 +- fluree-db-api/src/format/mod.rs | 50 +++++ fluree-db-api/src/format/sparql.rs | 40 +++- fluree-db-api/src/format/sparql_xml.rs | 19 ++ fluree-db-api/src/format/typed.rs | 18 +- fluree-db-api/tests/it_triple_term_links.rs | 198 ++++++++++++++++++++ 10 files changed, 504 insertions(+), 30 deletions(-) diff --git a/docs/query/construct.md b/docs/query/construct.md index 8c68dc08f5..3dc05802bf 100644 --- a/docs/query/construct.md +++ b/docs/query/construct.md @@ -365,7 +365,9 @@ details of each. - `?r rdf:reifies <<( s p o )>>` in a template also writes `s p o`, the same as the annotation tail `s p o ~ ?r`. Fluree reifies asserted edges only, as every write form does (see [Edge annotations](../concepts/edge-annotations.md)), so a result never carries a reifier - without its triple. + without its triple. `?r rdf:reifies ?t`, with `?t` bound to a triple term, writes the same. + Under any other predicate, a bound triple term is written as a literal holding its N-Triples + text: a result graph holds triple terms only as reifications. - A SPARQL datalog rule whose head (the template) annotates an edge or writes into a named graph is rejected: rules infer default-graph triples only. diff --git a/docs/query/output-formats.md b/docs/query/output-formats.md index 650293fb9a..492fe13f06 100644 --- a/docs/query/output-formats.md +++ b/docs/query/output-formats.md @@ -351,6 +351,29 @@ All formats use the same representation: {"type": "uri", "value": "http://example.org/ns/alice"} ``` +### Triple Terms + +A triple term (an `rdf:reifies` object, or the result of `TRIPLE(...)`) is written in SPARQL +JSON as a `triple` term +([SPARQL 1.2 Query Results JSON Format](https://www.w3.org/TR/sparql12-results-json/)): + +```json +{"type": "triple", "value": { + "subject": {"type": "uri", "value": "http://example.org/ns/alice"}, + "predicate": {"type": "uri", "value": "http://example.org/ns/knows"}, + "object": {"type": "uri", "value": "http://example.org/ns/bob"}}} +``` + +SPARQL XML uses the matching `` element, with ``, `` and `` +children. JSON-LD 1.1 has no triple terms. JSON-LD and Typed JSON write one as an embedded +node, the form of the JSON-LD-star community group report: + +```json +{"@id": {"@id": "ex:alice", "ex:knows": {"@id": "ex:bob"}}} +``` + +TSV and CSV write `<<( s p o )>>`. + ## Rust API Use `FormatterConfig` to control output format via the query builder API: diff --git a/fluree-db-api/src/format/construct.rs b/fluree-db-api/src/format/construct.rs index 52ec5b5f6a..0d1446972d 100644 --- a/fluree-db-api/src/format/construct.rs +++ b/fluree-db-api/src/format/construct.rs @@ -138,25 +138,58 @@ pub(super) fn instantiate_construct_graph( for batch in &result.batches { for row in 0..batch.len() { row_bnodes.clear(); - let mut resolve = |slot: &Slot, position| -> Result> { - Ok(match slot { - Slot::Const(term) => term.clone(), - Slot::Blank(v) => Some(IrTerm::BlankNode(row_blank( - *v, - &mut row_bnodes, - &mut bnode_counter, - ))), - Slot::Var(v) => match batch.get(row, *v) { - Some(binding) => terms.binding(binding, position)?, - None => None, - }, - }) - }; + let mut resolve = + |terms: &mut TermResolver<'_>, slot: &Slot, position| -> Result> { + Ok(match slot { + Slot::Const(term) => term.clone(), + Slot::Blank(v) => Some(IrTerm::BlankNode(row_blank( + *v, + &mut row_bnodes, + &mut bnode_counter, + ))), + Slot::Var(v) => match batch.get(row, *v) { + Some(binding) => terms.binding(binding, position)?, + None => None, + }, + }) + }; 'pattern: for (slots, graph_slot, reifier_slots) in &patterns { + // `?r rdf:reifies ?t` with a triple term bound to ?t writes + // what `?r rdf:reifies <<( s p o )>>` does: the term's triple + // and ?r's reification of it. + if let Slot::Var(v) = &slots[2] { + if let Some(components) = batch + .get(row, *v) + .map(|b| terms.triple_term(b)) + .transpose()? + .flatten() + { + let (Some(reifier), Some(p)) = ( + resolve(&mut terms, &slots[0], Position::Subject)?, + resolve(&mut terms, &slots[1], Position::Predicate)?, + ) else { + continue 'pattern; + }; + if matches!(&p, IrTerm::Iri(iri) if &**iri == rdf::REIFIES) { + let graph = match graph_slot { + Some(slot) => match resolve(&mut terms, slot, Position::Graph)? { + Some(name) => Some(name), + None => continue 'pattern, + }, + None => None, + }; + let [ts, tp, to] = components; + let g = dataset.graph_mut(graph.as_ref()); + g.add_reification(ts.clone(), tp.clone(), to.clone(), reifier); + g.add(Triple::new(ts, tp, to)); + continue 'pattern; + } + } + } let mut triple: [Option; 3] = [None, None, None]; for (i, (slot, position)) in slots.iter().zip(POSITIONS).enumerate() { // Skip if any term is unbound (incomplete triple) - let Some(term) = resolve(slot, position)? else { + let Some(term) = resolve(&mut terms, slot, position)? else { continue 'pattern; }; triple[i] = Some(term); @@ -169,7 +202,7 @@ pub(super) fn instantiate_construct_graph( continue; } let graph = match graph_slot { - Some(slot) => match resolve(slot, Position::Graph)? { + Some(slot) => match resolve(&mut terms, slot, Position::Graph)? { Some(name) => Some(name), // An unbound graph name writes nothing. None => continue 'pattern, @@ -179,7 +212,7 @@ pub(super) fn instantiate_construct_graph( // Reifiers bound on this row; an unbound one attaches nothing. reifiers.clear(); for slot in reifier_slots { - if let Some(r) = resolve(slot, Position::Subject)? { + if let Some(r) = resolve(&mut terms, slot, Position::Subject)? { reifiers.push(r); } } @@ -350,7 +383,88 @@ impl TermResolver<'_> { } } + /// A binding's triple term as the IR terms of its subject, predicate and + /// object, when it holds one whose object the graph IR can carry (any + /// term but a nested triple term). + fn triple_term(&mut self, binding: &Binding) -> Result> { + if binding.is_encoded() { + let materialized = super::materialize::materialize_binding(self.result, binding)?; + return self.triple_term(&materialized); + } + let Binding::Lit { + val: FlakeValue::TripleTerm(term), + .. + } = binding + else { + return Ok(None); + }; + if matches!(term.o, FlakeValue::TripleTerm(_)) { + return Ok(None); + } + let [s, p, o] = super::triple_term_components(term); + let s = self.binding(&s, Position::Subject)?; + let p = self.binding(&p, Position::Predicate)?; + let o = self.binding(&o, Position::Object)?; + Ok(match (s, p, o) { + (Some(s), Some(p), Some(o)) => Some([s, p, o]), + _ => None, + }) + } + + /// A triple term in N-Triples syntax with full IRIs: the lexical form of a + /// triple term the graph IR cannot carry as a term. + fn triple_term_text(&mut self, term: &fluree_db_core::TripleTermValue) -> Result { + let mut out = String::from("<<( "); + for component in super::triple_term_components(term) { + match &component { + Binding::Sid { sid, .. } => { + let iri = self.sid_iri(sid)?; + if iri.starts_with("_:") { + out.push_str(&iri); + } else { + out.push('<'); + out.push_str(&iri); + out.push('>'); + } + } + Binding::Lit { + val: FlakeValue::TripleTerm(inner), + .. + } => out.push_str(&self.triple_term_text(inner)?), + Binding::Lit { val, dtc, .. } => { + let lexical = match val { + FlakeValue::String(s) | FlakeValue::Json(s) => s.clone(), + other => other.to_string(), + }; + out.push_str(&serde_json::to_string(&lexical).unwrap_or_default()); + match dtc.lang_tag() { + Some(tag) => { + out.push('@'); + out.push_str(tag); + } + None => { + out.push_str("^^<"); + out.push_str(&self.sid_iri(dtc.datatype())?); + out.push('>'); + } + } + } + _ => {} + } + out.push(' '); + } + out.push_str(")>>"); + Ok(out) + } + fn literal(&mut self, val: &FlakeValue, dtc: &DatatypeConstraint) -> Result> { + if let FlakeValue::TripleTerm(term) = val { + return Ok(Some(IrTerm::Literal { + value: LiteralValue::String(Arc::from(self.triple_term_text(term)?)), + datatype: self.datatype(dtc.datatype())?, + language: None, + })); + } let (datatype, language) = match (val, dtc.lang_tag()) { (FlakeValue::String(_), Some(tag)) => { (Datatype::rdf_lang_string(), Some(Arc::from(tag))) diff --git a/fluree-db-api/src/format/delimited.rs b/fluree-db-api/src/format/delimited.rs index bda9e5400a..3fe156499d 100644 --- a/fluree-db-api/src/format/delimited.rs +++ b/fluree-db-api/src/format/delimited.rs @@ -636,7 +636,26 @@ fn write_flake_value(cell: &mut Vec, val: &FlakeValue, compactor: &IriCompac } FlakeValue::Json(json_str) => cell.extend_from_slice(json_str.as_bytes()), FlakeValue::GeoPoint(v) => cell.extend_from_slice(v.to_string().as_bytes()), - FlakeValue::TripleTerm(_) => cell.extend_from_slice(val.to_string().as_bytes()), + FlakeValue::TripleTerm(term) => { + cell.extend_from_slice(b"<<( "); + write_flake_value(cell, &FlakeValue::Ref(term.s.clone()), compactor); + cell.push(b' '); + write_flake_value(cell, &FlakeValue::Ref(term.p.clone()), compactor); + cell.push(b' '); + match &term.o { + // Quoted, so a string's spaces cannot run into the term's end. + FlakeValue::String(s) => { + let quoted = serde_json::to_string(s).unwrap_or_default(); + cell.extend_from_slice(quoted.as_bytes()); + if let Some(lang) = &term.lang { + cell.push(b'@'); + cell.extend_from_slice(lang.as_bytes()); + } + } + other => write_flake_value(cell, other, compactor), + } + cell.extend_from_slice(b" )>>"); + } } } diff --git a/fluree-db-api/src/format/jsonld.rs b/fluree-db-api/src/format/jsonld.rs index 7f3fcdee57..8c6ec0750d 100644 --- a/fluree-db-api/src/format/jsonld.rs +++ b/fluree-db-api/src/format/jsonld.rs @@ -287,6 +287,10 @@ fn write_lit( dtc: &fluree_db_core::DatatypeConstraint, compactor: &IriCompactor, ) -> Result<()> { + if let FlakeValue::TripleTerm(term) = val { + let node = super::triple_term_node(term, compactor, |o| format_binding(o, compactor))?; + return push_value(out, &node); + } let dt = dtc.datatype(); let dt_full = compactor.decode_sid(dt)?; let dt_compact = compactor.compact_sid(dt)?; @@ -409,6 +413,11 @@ pub(crate) fn format_binding(binding: &Binding, compactor: &IriCompactor) -> Res Binding::Iri(iri) => Ok(JsonValue::String(iri.to_string())), // Literal value - never contains Ref (enforced by Binding::from_object) + Binding::Lit { + val: FlakeValue::TripleTerm(term), + .. + } => super::triple_term_node(term, compactor, |o| format_binding(o, compactor)), + Binding::Lit { val, dtc, .. } => { let dt = dtc.datatype(); // Full datatype IRI string (e.g., "http://www.w3.org/2001/XMLSchema#string" or "@json") @@ -496,7 +505,7 @@ pub(crate) fn format_binding(binding: &Binding, compactor: &IriCompactor) -> Res FlakeValue::DayTimeDuration(v) => Ok(JsonValue::String(v.to_string())), FlakeValue::Duration(v) => Ok(JsonValue::String(v.to_string())), FlakeValue::GeoPoint(v) => Ok(JsonValue::String(v.to_string())), - FlakeValue::TripleTerm(_) => Ok(JsonValue::String(val.to_string())), + FlakeValue::TripleTerm(_) => unreachable!("rendered as an embedded node"), }; } @@ -546,7 +555,7 @@ pub(crate) fn format_binding(binding: &Binding, compactor: &IriCompactor) -> Res FlakeValue::DayTimeDuration(v) => JsonValue::String(v.to_string()), FlakeValue::Duration(v) => JsonValue::String(v.to_string()), FlakeValue::GeoPoint(v) => JsonValue::String(v.to_string()), - FlakeValue::TripleTerm(_) => JsonValue::String(val.to_string()), + FlakeValue::TripleTerm(_) => unreachable!("rendered as an embedded node"), }; Ok(json!({ diff --git a/fluree-db-api/src/format/mod.rs b/fluree-db-api/src/format/mod.rs index e28ae8e473..f96da05175 100644 --- a/fluree-db-api/src/format/mod.rs +++ b/fluree-db-api/src/format/mod.rs @@ -59,6 +59,56 @@ pub(crate) mod sparql; mod sparql_xml; mod typed; +/// A triple term's subject, predicate and object as the bindings a formatter +/// renders: nodes, or an object literal with its datatype or language tag (a +/// nested term among them, which the formatter renders recursively). +pub(crate) fn triple_term_components( + term: &fluree_db_core::TripleTermValue, +) -> [fluree_db_query::binding::Binding; 3] { + use fluree_db_core::{DatatypeConstraint, FlakeValue}; + use fluree_db_query::binding::Binding; + let object = match &term.o { + FlakeValue::Ref(sid) => Binding::sid(sid.clone()), + other => Binding::Lit { + val: other.clone(), + dtc: match &term.lang { + Some(lang) => DatatypeConstraint::LangTag(std::sync::Arc::from(lang.as_str())), + None => DatatypeConstraint::Explicit(term.dt.clone()), + }, + t: None, + op: None, + p_id: None, + }, + }; + [ + Binding::sid(term.s.clone()), + Binding::sid(term.p.clone()), + object, + ] +} + +/// A triple term as a JSON-LD-star embedded node, `{"@id": {"@id": s, p: o}}` +/// (JSON-LD 1.1 has no triple terms). `object` renders a literal object, or a +/// nested term, as the calling formatter renders that value anywhere else. +pub(crate) fn triple_term_node( + term: &fluree_db_core::TripleTermValue, + compactor: &IriCompactor, + object: impl FnOnce(&fluree_db_query::binding::Binding) -> Result, +) -> Result { + let [_, _, o] = triple_term_components(term); + let o = match &term.o { + fluree_db_core::FlakeValue::Ref(sid) => json!({ "@id": compactor.compact_id_sid(sid)? }), + _ => object(&o)?, + }; + let mut node = serde_json::Map::with_capacity(2); + node.insert( + "@id".to_string(), + JsonValue::String(compactor.compact_id_sid(&term.s)?), + ); + node.insert(compactor.compact_sid(&term.p)?, o); + Ok(json!({ "@id": node })) +} + /// Registry-name predicate: does this variable name belong to a /// non-projected internal / non-distinguished variable that should be /// hidden from `SELECT *` wildcard output? diff --git a/fluree-db-api/src/format/sparql.rs b/fluree-db-api/src/format/sparql.rs index 4635d1f431..2204cc98c7 100644 --- a/fluree-db-api/src/format/sparql.rs +++ b/fluree-db-api/src/format/sparql.rs @@ -340,6 +340,25 @@ fn write_term(out: &mut String, binding: &Binding, compactor: &IriCompactor) -> write_node(out, iri.as_ref()); } } + Binding::Lit { + val: FlakeValue::TripleTerm(term), + .. + } => { + out.push_str(r#"{"type":"triple","value":{"#); + for (i, (name, component)) in ["subject", "predicate", "object"] + .into_iter() + .zip(super::triple_term_components(term)) + .enumerate() + { + if i > 0 { + out.push(','); + } + push_json_string(out, name); + out.push(':'); + write_term(out, &component, compactor)?; + } + out.push_str("}}"); + } Binding::Lit { val, dtc, .. } => { let dt_iri = compactor.decode_sid(dtc.datatype())?; write_literal( @@ -674,11 +693,22 @@ fn format_binding( "value": v.to_string(), "datatype": dt_iri }))), - FlakeValue::TripleTerm(_) => Ok(Some(json!({ - "type": "literal", - "value": val.to_string(), - "datatype": dt_iri - }))), + FlakeValue::TripleTerm(term) => { + let mut value = serde_json::Map::new(); + for (name, component) in ["subject", "predicate", "object"] + .into_iter() + .zip(super::triple_term_components(term)) + { + let rendered = + format_binding(result, &component, compactor)?.ok_or_else(|| { + FormatError::InvalidBinding(format!( + "triple term {name} has no value" + )) + })?; + value.insert(name.to_string(), rendered); + } + Ok(Some(json!({ "type": "triple", "value": value }))) + } } } diff --git a/fluree-db-api/src/format/sparql_xml.rs b/fluree-db-api/src/format/sparql_xml.rs index f103447cd5..1171b81e7f 100644 --- a/fluree-db-api/src/format/sparql_xml.rs +++ b/fluree-db-api/src/format/sparql_xml.rs @@ -210,6 +210,25 @@ fn write_term( Binding::Sid { sid, .. } => write_sid_ref(out, compactor, sid)?, Binding::IriMatch { iri, .. } => write_iri_ref(out, iri.as_ref()), Binding::Iri(iri) => write_iri_ref(out, iri.as_ref()), + Binding::Lit { + val: FlakeValue::TripleTerm(term), + .. + } => { + out.push_str(""); + for (name, component) in ["subject", "predicate", "object"] + .into_iter() + .zip(super::triple_term_components(term)) + { + out.push('<'); + out.push_str(name); + out.push('>'); + write_term(out, result, &component, compactor, gv)?; + out.push_str("'); + } + out.push_str(""); + } Binding::Lit { val, dtc, .. } => write_literal(out, compactor, val, dtc)?, // Encoded subject/predicate refs resolve directly to their full IRI — diff --git a/fluree-db-api/src/format/typed.rs b/fluree-db-api/src/format/typed.rs index 3841be5863..400da5a96e 100644 --- a/fluree-db-api/src/format/typed.rs +++ b/fluree-db-api/src/format/typed.rs @@ -191,6 +191,14 @@ fn write_value( push_json_string(out, iri.as_ref()); out.push('}'); } + Binding::Lit { + val: FlakeValue::TripleTerm(term), + .. + } => { + let node = + super::triple_term_node(term, compactor, |o| format_binding(result, o, compactor))?; + push_value(out, &node)?; + } Binding::Lit { val, dtc, .. } => write_lit(out, val, dtc, compactor)?, Binding::Grouped(values) => { out.push('['); @@ -388,6 +396,11 @@ pub(crate) fn format_binding( // Raw IRI string (from graph source, not in namespace table) Binding::Iri(iri) => Ok(json!({"@id": iri.as_ref()})), + Binding::Lit { + val: FlakeValue::TripleTerm(term), + .. + } => super::triple_term_node(term, compactor, |o| format_binding(result, o, compactor)), + // Literal value - always include @type (except language-tagged) Binding::Lit { val, dtc, .. } => { let dt_iri = compactor.compact_sid(dtc.datatype())?; @@ -516,10 +529,7 @@ pub(crate) fn format_binding( "@value": v.to_string(), "@type": dt_iri })), - FlakeValue::TripleTerm(_) => Ok(json!({ - "@value": val.to_string(), - "@type": dt_iri - })), + FlakeValue::TripleTerm(_) => unreachable!("rendered as an embedded node"), } } diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 71d5349689..edaa534a72 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -1999,3 +1999,201 @@ async fn previews_read_the_links_of_their_own_annotations() { assert_eq!(got, strings(&[&["ex:erin"]])); } } + +/// A triple term renders as SPARQL 1.2's `triple` term in the SPARQL result +/// formats, as `<<( s p o )>>` in delimited text, and as a JSON-LD-star +/// embedded node in the JSON-LD formats, whichever writer (DOM or streaming) +/// produces it. +#[tokio::test] +async fn triple_terms_render_in_every_result_format() { + use fluree_db_api::format::{format_results_string, FormatterConfig}; + let (fluree, ledger) = import( + &[("claims.ttl", CLAIMS), ("literals.ttl", LITERAL_CLAIMS)], + "it/triple-term-links:formats", + ) + .await; + let select = "PREFIX ex: \n\ + PREFIX rdf: \n\ + SELECT ?src ?t WHERE { ?r rdf:reifies ?t ; ex:source ?src \ + VALUES ?src { ex:hr ex:fr ex:int } } ORDER BY ?src"; + let result = support::query_sparql(&fluree, &ledger, select) + .await + .expect("link query"); + let string = |config: FormatterConfig| { + format_results_string(&result, &result.context, &ledger.snapshot, &config).expect("format") + }; + let parsed = |config: FormatterConfig| -> JsonValue { + serde_json::from_str(&string(config)).expect("JSON output") + }; + + let iri = |l: &str| json!({"type": "uri", "value": format!("http://example.org/{l}")}); + let triple = |s: &str, p: &str, o: JsonValue| json!({"type": "triple", "value": {"subject": iri(s), "predicate": iri(p), "object": o}}); + let sparql_json = result + .to_sparql_json(&ledger.snapshot) + .expect("SPARQL JSON"); + let terms: Vec<&JsonValue> = sparql_json["results"]["bindings"] + .as_array() + .expect("bindings") + .iter() + .map(|b| &b["t"]) + .collect(); + assert_eq!( + terms, + [ + &triple( + "doc", + "title", + json!({"type": "literal", "value": "chat", "xml:lang": "fr"}) + ), + &triple("alice", "knows", iri("bob")), + &triple( + "doc", + "size", + json!({"type": "literal", "value": "5", + "datatype": "http://www.w3.org/2001/XMLSchema#int"}) + ), + ] + ); + assert_eq!(parsed(FormatterConfig::sparql_json()), sparql_json); + + let xml = string(FormatterConfig::sparql_xml()); + for term in [ + "http://example.org/doc\ + http://example.org/title\ + chat", + "http://example.org/alice\ + http://example.org/knows\ + http://example.org/bob", + "5", + ] { + assert!(xml.contains(term), "{term} in {xml}"); + } + + let tsv = result.to_tsv(&ledger.snapshot).expect("TSV"); + assert!( + tsv.contains( + "\t<<( http://example.org/alice http://example.org/knows http://example.org/bob )>>\n" + ), + "{tsv}" + ); + assert!( + tsv.contains("\t<<( http://example.org/doc http://example.org/title \"chat\"@fr )>>\n"), + "{tsv}" + ); + let csv = result.to_csv(&ledger.snapshot).expect("CSV"); + assert!( + csv.contains( + ",\"<<( http://example.org/doc http://example.org/title \"\"chat\"\"@fr )>>\"" + ), + "{csv}" + ); + + let node = |s: &str, p: &str, o: JsonValue| json!({"@id": {"@id": s, p: o}}); + let int = json!({"@value": 5, "@type": "http://www.w3.org/2001/XMLSchema#int"}); + let jsonld = result.to_jsonld(&ledger.snapshot).expect("JSON-LD"); + assert_eq!( + jsonld, + json!([ + [ + "ex:fr", + node( + "ex:doc", + "ex:title", + json!({"@value": "chat", "@language": "fr"}) + ) + ], + [ + "ex:hr", + node("ex:alice", "ex:knows", json!({"@id": "ex:bob"})) + ], + ["ex:int", node("ex:doc", "ex:size", int.clone())], + ]) + ); + assert_eq!(parsed(FormatterConfig::jsonld()), jsonld); + let typed = result.to_typed_json(&ledger.snapshot).expect("typed JSON"); + assert_eq!( + typed[1]["?t"], + node("ex:alice", "ex:knows", json!({"@id": "ex:bob"})) + ); + assert_eq!(typed[2]["?t"], node("ex:doc", "ex:size", int)); + assert_eq!(parsed(FormatterConfig::typed_json()), typed); +} + +/// `?r rdf:reifies ?t` in a CONSTRUCT template, with a triple term bound to +/// ?t, writes what the explicit `<<( s p o )>>` template writes. Under any +/// other predicate the term (which the graph model cannot hold as an object) +/// is written as its N-Triples text. +#[tokio::test] +async fn construct_writes_triple_terms_as_reifications() { + use fluree_db_api::format::{format_results_string, FormatterConfig}; + let (fluree, ledger) = import( + &[("claims.ttl", CLAIMS), ("literals.ttl", LITERAL_CLAIMS)], + "it/triple-term-links:construct", + ) + .await; + let render = |result: &fluree_db_api::QueryResult, config: FormatterConfig| { + format_results_string(result, &result.context, &ledger.snapshot, &config).expect("format") + }; + let construct = |template: &'static str, filter: &'static str| { + let fluree = &fluree; + let ledger = &ledger; + async move { + let sparql = format!( + "PREFIX ex: \n\ + PREFIX rdf: \n\ + CONSTRUCT {{ {template} }} WHERE {{ {filter} }}" + ); + support::query_sparql(fluree, ledger, &sparql) + .await + .expect("construct") + } + }; + let sorted = |result: &fluree_db_api::QueryResult| { + let mut lines: Vec = render(result, FormatterConfig::ntriples()) + .lines() + .map(str::to_string) + .collect(); + lines.sort(); + lines + }; + + for (term_form, template_form) in [ + ( + "?r rdf:reifies ?t ; ex:source ex:hr", + "?r rdf:reifies <<( ?s ?p ?o )>> ; ex:source ex:hr", + ), + ( + "?r rdf:reifies ?t ; ex:source ex:fr", + "?r rdf:reifies <<( ?s ?p ?o )>> ; ex:source ex:fr", + ), + ] { + let by_term = construct("?r rdf:reifies ?t", term_form).await; + let by_template = construct("?r rdf:reifies <<( ?s ?p ?o )>>", template_form).await; + let lines = sorted(&by_term); + assert_eq!( + lines.len(), + 2, + "the base triple and its reification: {lines:#?}" + ); + assert_eq!(lines, sorted(&by_template), "{term_form}"); + assert_eq!( + by_term.to_construct(&ledger.snapshot).expect("JSON-LD"), + by_template.to_construct(&ledger.snapshot).expect("JSON-LD"), + "{term_form}" + ); + } + let hr = construct("?r rdf:reifies ?t", "?r rdf:reifies ?t ; ex:source ex:hr").await; + assert_eq!( + hr.to_construct(&ledger.snapshot).expect("JSON-LD")["@graph"], + json!([{"@id": "ex:alice", "ex:knows": [{"@id": "ex:bob", "@annotation": {"@id": "ex:claim1"}}]}]) + ); + + let about = construct("?r ex:about ?t", "?r rdf:reifies ?t ; ex:source ex:fr").await; + let nt = render(&about, FormatterConfig::ntriples()); + assert!( + nt.contains( + r#" "<<( \"chat\"@fr )>>"^^"# + ), + "{nt}" + ); +} From 1069d4d9f338e1bfc939e5bfae058c3cdaac614d Mon Sep 17 00:00:00 2001 From: bplatz Date: Thu, 1 Oct 2026 23:54:53 -0400 Subject: [PATCH 39/92] fix(api): keep rdf:reifies out of the Cypher catalog procedures db.propertyKeys() listed rdf:reifies on any ledger with an annotated edge, which a Cypher relationship CREATE writes: stats count the link flakes as a property, and the procedures hid only Fluree's own vocabulary, where the f:reifies* bundle predicates lived. display_name now hides the link too, as wildcard scans do, for every procedure that renders a predicate. --- fluree-db-api/src/cypher_procedures.rs | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/fluree-db-api/src/cypher_procedures.rs b/fluree-db-api/src/cypher_procedures.rs index d7bf7f16c7..451bac1be2 100644 --- a/fluree-db-api/src/cypher_procedures.rs +++ b/fluree-db-api/src/cypher_procedures.rs @@ -515,13 +515,17 @@ struct SchemaNames { /// Render a class/predicate SID back to the identifier a Cypher user would /// write: term overrides reversed, `@vocab` prefix stripped, otherwise the /// full IRI. `None` for Fluree system vocabulary (commit metadata, -/// edge-annotation reifiers, …) — not part of the user's property graph. +/// edge-annotation reifiers, …) and for `rdf:reifies`, the link an +/// annotation's reifier holds — not part of the user's property graph. fn display_name( sid: &Sid, snapshot: &LedgerSnapshot, vocab: Option<&str>, overrides: &HashMap, ) -> Option { + if fluree_db_core::is_rdf_reifies(sid) { + return None; + } let prefix = snapshot.namespaces().get(&sid.namespace_code)?; let iri = format!("{}{}", prefix, sid.name); if iri.starts_with("https://ns.flur.ee/") { From 891c398b1656d1b2f76039c44345d5f5378bf2f7 Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 07:04:01 -0400 Subject: [PATCH 40/92] fix(api): export no longer writes the index's rdf:reifies links On an indexed annotated ledger, export wrote each derived link as an ordinary triple, `r rdf:reifies "<<( [13:alice] ... )>>"^^f:tripleTerm`, next to the annotation syntax it already emits, in every format, so `create --from` an export failed. Links are rebuilt from the f:reifies* bundles and never enter commits; the row writers now skip them (by o_type) in every mode. Novelty links never reach the export cursor. --- fluree-db-api/src/export.rs | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/fluree-db-api/src/export.rs b/fluree-db-api/src/export.rs index ce569ec730..efd127a93a 100644 --- a/fluree-db-api/src/export.rs +++ b/fluree-db-api/src/export.rs @@ -428,6 +428,14 @@ impl<'a> AnnotationContext<'a> { } } +/// An `rdf:reifies` link row. Links are derived from the `f:reifies*` bundles +/// (which export writes as annotation syntax) and never enter commits, so +/// export drops them in every mode, `--raw-reifies` included. +#[inline] +fn is_link_row(o_type: u16) -> bool { + o_type == OType::TRIPLE_TERM.as_u16() +} + /// Live reifiers for every row of `batch`, row-aligned. /// /// Returns an empty vec when the export is not emitting annotation syntax; @@ -456,6 +464,9 @@ async fn batch_reifiers( continue; // the bundle itself is never an annotated edge } let o_type = batch.o_type.get_or(row, 0); + if is_link_row(o_type) { + continue; + } let o_key = batch.o_key.get(row); let Some(p) = resolver.resolve_predicate_sid(p_id) else { continue; @@ -682,6 +693,9 @@ fn write_turtle_batch( let p_id = batch.p_id.get_or(row, 0); let o_type = batch.o_type.get_or(row, 0); let o_key = batch.o_key.get(row); + if is_link_row(o_type) { + continue; + } // The `f:reifies*` bundle is the on-disk encoding of an annotation, // not a triple the ledger was asked to hold. It is replaced by the @@ -877,6 +891,9 @@ pub async fn export_graph_jsonld( let p_id = batch.p_id.get_or(row, 0); let o_type = batch.o_type.get_or(row, 0); let o_key = batch.o_key.get(row); + if is_link_row(o_type) { + continue; + } if let Some(ann) = ann.as_ref() { if ann.is_reifies_row(p_id) { @@ -1560,6 +1577,9 @@ fn write_batch( let p_id = batch.p_id.get_or(row, 0); let o_type = batch.o_type.get_or(row, 0); let o_key = batch.o_key.get(row); + if is_link_row(o_type) { + continue; + } if let Some(ann) = ann { if ann.is_reifies_row(p_id) { From f3af076bd187cd89c79286829462200189cf9864 Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 09:03:56 -0400 Subject: [PATCH 41/92] fix(query): a link is visible only when the triple it names is `r rdf:reifies <<( s p o )>>` carries s, p and o in its triple term, so under a policy hiding `p` a plain `?r rdf:reifies ?t` scan handed back the hidden edge's endpoints through ?t. The policy enforcer now admits a flake whose object is a triple term only when the triple that term names would be visible, through nested terms too, caching the term subjects' classes for f:onClass rules. Every path that filters flakes by policy applies it. --- fluree-db-api/tests/it_edge_annotations.rs | 72 +++++++++++++++++++ fluree-db-query/src/policy/enforcer.rs | 80 +++++++++++++++++++++- 2 files changed, 151 insertions(+), 1 deletion(-) diff --git a/fluree-db-api/tests/it_edge_annotations.rs b/fluree-db-api/tests/it_edge_annotations.rs index 45a4d3798a..1dfc3c846b 100644 --- a/fluree-db-api/tests/it_edge_annotations.rs +++ b/fluree-db-api/tests/it_edge_annotations.rs @@ -5378,6 +5378,78 @@ async fn policy_hiding_base_edge_blocks_annotation_rooted_query() { ); } +/// A link `r rdf:reifies <<( s p o )>>` names its triple, so a policy hiding +/// the triple hides the link on every route to it, not only `@reifies`: +/// SPARQL's quoted pattern and a plain `rdf:reifies` scan. +#[tokio::test] +async fn policy_hiding_base_edge_hides_its_link() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/edge-annotations:policy-hides-link"; + let ledger0 = genesis_ledger(&fluree, ledger_id); + let insert = json!({ + "@context": ctx(), + "@id": "ex:alice", + "ex:worksFor": { + "@id": "ex:acme", + "@annotation": { "@id": "ex:emp-A", "ex:role": "Engineer" } + } + }); + fluree.insert(ledger0, &insert).await.expect("insert"); + let ledger = fluree.ledger(ledger_id).await.expect("reload"); + + let opts = fluree_db_api::GovernanceOptions { + policy: Some(json!([{ + "@id": "ex:hide-worksFor", + "f:required": true, + "f:onProperty": [{"@id": "http://example.org/worksFor"}], + "f:action": "f:view", + "f:query": serde_json::to_string(&json!({ + "where": {"@id": "?$identity", "@type": "http://example.org/NeverMatches"} + })).unwrap() + }])), + default_allow: Some(true), + ..Default::default() + }; + let policy = fluree_db_api::policy_builder::build_policy_context_from_opts( + &ledger.snapshot, + ledger.novelty.as_ref(), + Some(ledger.novelty.as_ref()), + ledger.t(), + &opts, + &[0], + ) + .await + .expect("policy context"); + + let reifies = ""; + for sparql in [ + format!( + "SELECT ?person ?org WHERE {{ ?r {reifies} \ + <<( ?person ?org )>> }}" + ), + format!("SELECT ?r ?t WHERE {{ ?r {reifies} ?t }}"), + ] { + let count = |db: fluree_db_api::GraphDb| { + let fluree = &fluree; + let ledger = &ledger; + let sparql = sparql.clone(); + async move { + let result = fluree.query(&db, sparql.as_str()).await.expect("query"); + let rows = result.to_jsonld(&ledger.snapshot).expect("to_jsonld"); + rows.as_array().map_or(0, Vec::len) + } + }; + assert_eq!( + count(support::graphdb_from_ledger(&ledger)).await, + 1, + "{sparql}" + ); + let policed = + support::graphdb_from_ledger(&ledger).with_policy(std::sync::Arc::new(policy.clone())); + assert_eq!(count(policed).await, 0, "the hidden edge's link: {sparql}"); + } +} + // ===================================================================== // #1467 — reification-aware COPY/MOVE/ADD re-homing (named-source cases) // ===================================================================== diff --git a/fluree-db-query/src/policy/enforcer.rs b/fluree-db-query/src/policy/enforcer.rs index 67a82806b7..5d099409ff 100644 --- a/fluree-db-query/src/policy/enforcer.rs +++ b/fluree-db-query/src/policy/enforcer.rs @@ -4,7 +4,9 @@ use super::QueryPolicyExecutor; use crate::error::Result; -use fluree_db_core::{Flake, GraphId, LedgerSnapshot, OverlayProvider, Sid, Tracker}; +use fluree_db_core::{ + Flake, FlakeValue, GraphId, LedgerSnapshot, OverlayProvider, Sid, Tracker, TripleTermValue, +}; use fluree_db_policy::{is_schema_flake, PolicyContext}; use std::sync::Arc; @@ -117,6 +119,8 @@ impl QueryPolicyEnforcer { // Create executor using the GRAPH's snapshot/overlay/to_t (not ctx-level!) let executor = QueryPolicyExecutor::with_overlay(snapshot, overlay, to_t); + self.cache_term_subject_classes(snapshot, g_id, overlay, to_t, &flakes) + .await?; let mut result = Vec::with_capacity(flakes.len()); @@ -126,6 +130,12 @@ impl QueryPolicyEnforcer { result.push(flake); continue; } + if !self + .term_visible(g_id, to_t, &flake.o, &executor, tracker) + .await? + { + continue; + } // Get subject classes from cache let subject_classes = self @@ -184,6 +194,14 @@ impl QueryPolicyEnforcer { // Create executor using the GRAPH's snapshot/overlay/to_t let executor = QueryPolicyExecutor::with_overlay(snapshot, overlay, to_t); + self.cache_term_subject_classes(snapshot, g_id, overlay, to_t, std::slice::from_ref(flake)) + .await?; + if !self + .term_visible(g_id, to_t, &flake.o, &executor, tracker) + .await? + { + return Ok(false); + } // Get subject classes from cache let subject_classes = self @@ -205,6 +223,66 @@ impl QueryPolicyEnforcer { .map_err(|e| crate::error::QueryError::Policy(e.to_string())) } + /// A triple-term object shows the triple it names, so a flake holding one + /// (an `rdf:reifies` link) is visible only when that triple would be, and + /// so on through a nested term. Other objects pass. + async fn term_visible( + &self, + g_id: GraphId, + to_t: i64, + object: &FlakeValue, + executor: &QueryPolicyExecutor<'_>, + tracker: &Tracker, + ) -> Result { + let mut object = object; + while let FlakeValue::TripleTerm(term) = object { + let classes = self + .policy + .get_cached_subject_classes(g_id, to_t, &term.s) + .unwrap_or_default(); + let allowed = self + .policy + .allow_view_flake_async(&term.s, &term.p, &term.o, &classes, executor, tracker) + .await + .map_err(|e| crate::error::QueryError::Policy(e.to_string()))?; + if !allowed { + return Ok(false); + } + object = &term.o; + } + Ok(true) + } + + /// Cache the classes of the subjects of every triple term among `flakes`' + /// objects, which `term_visible` checks like any flake subject. + async fn cache_term_subject_classes( + &self, + snapshot: &LedgerSnapshot, + g_id: GraphId, + overlay: &dyn OverlayProvider, + to_t: i64, + flakes: &[Flake], + ) -> Result<()> { + let mut subjects: Vec = Vec::new(); + for flake in flakes { + let mut object = &flake.o; + while let FlakeValue::TripleTerm(term) = object { + let TripleTermValue { s, o, .. } = &**term; + subjects.push(s.clone()); + object = o; + } + } + if subjects.is_empty() { + return Ok(()); + } + subjects.sort(); + subjects.dedup(); + let db = fluree_db_core::GraphDbRef::new(snapshot, g_id, overlay, to_t); + self.populate_class_cache_for_graph(db, &subjects) + .await + .map_err(|e| crate::error::QueryError::Policy(e.to_string())) + } + /// Populate the class cache for subjects using a graph database reference. /// /// Call this before filtering to ensure class lookups are cached. From 992d45337ce36df738b809cbe2ca8fdcd911c6d7 Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 09:04:01 -0400 Subject: [PATCH 42/92] feat(query): reified-triple patterns always read the rdf:reifies link The link lowering ran only under FLUREE_ANNOTATION_TERMS=1, and only for SPARQL: JSON-LD `@reifies` always walked the f:reifies* bundle chain. - `lower_reified_link` moves to fluree-db-query, and both surfaces call it for every reified-triple pattern (`<< s p o ~ ?r >>`, `?r rdf:reifies <<( s p o )>>`, `@reifies`). Pattern::AnnotationTarget has no producer left and is removed. Annotation syntax (`~ ?r`, `{| |}`, `@annotation`, Cypher relationship properties) keeps EdgeAnnotation and its chain. The flag is gone. - An index built before links holds an annotated ledger's annotations only as bundles, so link reads over it answered without them. Such an index (has_annotations, no term dictionary) now refuses link reads with an error asking for `fluree reindex`, and an incremental build over it leaves link synthesis off, so it never gains a term dictionary covering its window alone. A full rebuild links every annotation. - CONSTRUCT writes a constant triple-term object (the link of a fully constant quoted triple in a CONSTRUCT WHERE template) as a reification, as it does a bound one. W3C construct-2 caught it. - The explain test that pinned the bundle-chain expansion of `@reifies` now pins it for `@annotation`, the form that still expands. An index this version builds cannot be read by older binaries, and an annotated ledger needs a reindex after upgrading. --- docs/concepts/edge-annotations.md | 6 +- docs/design/edge-annotations.md | 12 +- fluree-db-api/src/explain.rs | 9 +- fluree-db-api/src/format/construct.rs | 76 +++--- fluree-db-api/src/format/hydration.rs | 4 +- fluree-db-api/tests/grp_triple_terms.rs | 1 - .../tests/it_edge_annotations_indexed.rs | 20 +- fluree-db-api/tests/it_triple_term_links.rs | 157 ++++++++++-- .../run_index/build/incremental_resolve.rs | 10 +- fluree-db-query/src/binary_scan.rs | 3 + fluree-db-query/src/datalog_rules/parse.rs | 3 +- fluree-db-query/src/datalog_rules/validate.rs | 6 +- fluree-db-query/src/execute/where_plan.rs | 54 ++--- fluree-db-query/src/explain.rs | 7 - fluree-db-query/src/ir.rs | 2 +- fluree-db-query/src/ir/pattern.rs | 52 +--- fluree-db-query/src/ir/term_components.rs | 162 ++++++++++++- fluree-db-query/src/parse/lower.rs | 16 +- fluree-db-query/src/planner.rs | 15 +- fluree-db-query/src/r2rml/rewrite.rs | 10 +- fluree-db-query/src/rewrite.rs | 1 - fluree-db-query/src/rewrite_owl_ql.rs | 1 - fluree-db-query/src/term_components.rs | 21 ++ fluree-db-sparql/src/lower/annotation.rs | 223 ++---------------- fluree-db-sparql/src/lower/construct.rs | 5 - fluree-db-sparql/src/lower/mod.rs | 72 +++--- fluree-db-sparql/src/lower/pattern.rs | 4 +- fluree-db-sparql/src/lower/rdf_star.rs | 7 +- 28 files changed, 497 insertions(+), 462 deletions(-) diff --git a/docs/concepts/edge-annotations.md b/docs/concepts/edge-annotations.md index 1db70fb81d..e38c92e14f 100644 --- a/docs/concepts/edge-annotations.md +++ b/docs/concepts/edge-annotations.md @@ -185,7 +185,7 @@ When you start from the metadata — "find every employment with `role = Enginee } ``` -`@reifies` is the same idea as `rdf:reifies` in RDF 1.2 — given an annotation subject, walk to the edge it reifies. Fluree resolves it through the reverse attachment index, so it's cheap regardless of how many annotations exist in the ledger. +`@reifies` is the same idea as `rdf:reifies` in RDF 1.2 — given an annotation subject, walk to the edge it reifies. Fluree reads it, as it reads SPARQL's `?r rdf:reifies <<( s p o )>>` and `<< s p o ~ ?r >>`, from the reifier's `rdf:reifies` link, whose object is a triple term the index resolves through its term dictionary, so it's cheap regardless of how many annotations exist in the ledger. A policy that hides an edge hides its link too: the link is visible only when the triple it names would be. ### Subject expansion @@ -566,8 +566,7 @@ Today's surface covers the common LPG / RDF-star use cases. The following are no - **Annotations on list-occurrence triples.** `@list` membership is in scope as a future extension; the on-disk format already reserves space for it. Today, annotating a list element is rejected at parse time. - **Reifiers for unasserted triples.** `@reifies` must point at an asserted edge. Pure-proposition reification (claims about triples that are not in the graph) is deferred. - **Reifiers for multiple triples.** One annotation subject corresponds to one edge. Reifying several unrelated triples from a single annotation isn't allowed. -- **Triple terms as values.** `ex:doc ex:mentions <<( ex:s ex:p ex:o )>>` is not a representable value on any surface (JSON-LD, SPARQL, Turtle): the only accepted position for `<<( ... )>>` is the object of `rdf:reifies`, where it names an edge rather than storing a triple. Binding a whole triple to a variable (`BIND(<<( ?s ?p ?o )>> AS ?t)`, `VALUES ?t { <<( ... )>> }`) and returning one in a result set are the same gap. Use a separate annotation subject. -- **SPARQL 1.2 triple-term functions and constructor.** `TRIPLE`, `SUBJECT`, `PREDICATE`, `OBJECT`, `isTRIPLE`, and the `BIND(<<( ?s ?p ?o )>> AS ?t)` triple-term constructor parse (with arity checks) but fail at lowering with a `not_implemented` error — they presuppose triple terms as first-class values, which v1's LPG model does not represent. +- **Triple terms as stored values.** `ex:doc ex:mentions <<( ex:s ex:p ex:o )>>` is not a representable value on any write surface (JSON-LD, SPARQL, Turtle): the only accepted position for `<<( ... )>>` is the object of `rdf:reifies`, where it names an edge rather than storing a triple. Use a separate annotation subject. A SPARQL query can still build and return one: `TRIPLE`, `SUBJECT`, `PREDICATE`, `OBJECT`, `isTRIPLE` and `BIND(<<( ?s ?p ?o )>> AS ?t)` work on terms, and a bound term renders in every result format (see [Output formats](../query/output-formats.md#triple-terms)). JSON-LD queries have no syntax for these functions yet. - **SPARQL UPDATE annotations inside named graphs.** Annotation tails under `GRAPH { }` / `WITH ` in SPARQL UPDATE are rejected; write named-graph annotations with JSON-LD `@annotation` or TriG-star. - **Unasserted reified triples.** RDF 1.2's `<< s p o >>` and `r rdf:reifies <<( s p o )>>` do not assert `s p o`; Fluree's do (the reifier is lifecycle-coupled to a live edge). A W3C test that depends on a reifier existing for a triple that is *not* in the graph therefore diverges. @@ -580,6 +579,7 @@ The mandated SPARQL 1.2 `VERSION "1.2"` prologue declaration is **accepted** (le - Annotation properties are stored as ordinary RDF facts. Time travel, policy, history, export, and reasoning all work on them without special cases. - Retraction cascade has a fast path: when both the index root and current novelty know the ledger has no annotations, base-edge retracts skip the attachment lookup entirely. - Branch fork, pack/sync, and ledger drop all walk annotation arena artifacts as part of the index reachability set, so annotated ledgers round-trip cleanly across these operations. +- Reified-triple patterns read the `rdf:reifies` links an index build derives from each annotation. An index built by an earlier Fluree version has none, so on an annotated ledger those queries fail with an error asking for a rebuild until it is reindexed (`fluree reindex `); an incremental build does not add them. For the index format, sidecar layout, sort orders, and garbage-collection treatment, see the [Edge annotations design doc](../design/edge-annotations.md). diff --git a/docs/design/edge-annotations.md b/docs/design/edge-annotations.md index 4f90a81a81..f8375a4504 100644 --- a/docs/design/edge-annotations.md +++ b/docs/design/edge-annotations.md @@ -164,9 +164,17 @@ Annotation forward and reverse branch CIDs are returned by `IndexRoot::all_cas_i - Old annotation CIDs become garbage only when no retained root references them. - Leaf-level CIDs behind a branch are walked during the GC-diff pass; `drop.rs` and `gc/collector.rs` both call into the expanded-CAS expansion helpers so a strict GC pass never deletes a still-reachable leaf. +## Reified-triple patterns read the link + +Reified-triple patterns (`<< s p o ~ ?r >>`, `?r rdf:reifies <<( s p o )>>`, JSON-LD `@reifies`) do not use the chain below. `lower_reified_link` (`fluree-db-query/src/ir/term_components.rs`), shared by the SPARQL and JSON-LD lowerings, emits the reifier's `rdf:reifies` link, `?r rdf:reifies ?t`, whose object is a triple-term handle the index's term dictionary resolves, plus `TermComponents(?t, s, p, o)` relating the term to its components and a `sameTerm` filter per constant component, which the planner turns into the scan's handle interval. A fully constant edge composes to a constant term instead. + +The link names its triple without joining the base edge, so visibility is checked on the term: `QueryPolicyEnforcer` lets a flake whose object is a triple term through only when the triple that term names would be visible, recursively for nested terms. A policy hiding `ex:worksFor` therefore hides the links to `ex:worksFor` edges on every route, not only on annotation-syntax patterns. + +Links exist only in indexes built since the link form. An annotated ledger (`has_annotations`) whose root has no term dictionary predates them, and its annotations live only in the `f:reifies*` bundles: link reads there (`BinaryScanOperator` on `rdf:reifies`, `TermComponentsOperator`) fail asking for a rebuild rather than answering without those annotations, and an incremental build over such a root leaves link synthesis off, so it never acquires a term dictionary that covers only its window. A full rebuild interns a term for every attachment in history and links them all. + ## Query IR expansion -`Pattern::EdgeAnnotation` and `Pattern::AnnotationTarget` are not executed as dedicated operators. `expand_edge_annotation_patterns` (`fluree-db-query/src/execute/where_plan.rs`) flattens them into a base-edge `Pattern::Triple` plus three required `f:reifies*` triples (`f:reifiesSubject` / `f:reifiesPredicate` / `f:reifiesObject`) plus body patterns before the join planner runs. The standard scan/join/dedup machinery handles the rest, and the planner's `reorder_patterns` picks the cheaper direction (edge-first vs annotation-first) based on the merged `AnnotationStats` selectivity. +Annotation syntax (`s p o ~ ?r`, `{| |}`, JSON-LD `@annotation`, Cypher relationship properties) lowers to `Pattern::EdgeAnnotation`, which is not executed as a dedicated operator. `expand_edge_annotation_patterns` (`fluree-db-query/src/execute/where_plan.rs`) flattens it into a base-edge `Pattern::Triple` plus three required `f:reifies*` triples (`f:reifiesSubject` / `f:reifiesPredicate` / `f:reifiesObject`) plus body patterns before the join planner runs. The standard scan/join/dedup machinery handles the rest, and the planner's `reorder_patterns` picks the cheaper direction (edge-first vs annotation-first) based on the merged `AnnotationStats` selectivity. The expansion does **not** emit an `f:reifiesGraph` constraint triple. Graph correlation is structural instead: each chain is scoped to one source at a time (the `Pattern::DefaultGraphSource` wrapper below, or an enclosing `Pattern::Graph`), so every `f:reifies*` lookup resolves against the same flake-level `g` per iteration — matching how `EdgeKey` derives graph identity. **Consequence:** the strict bundle-wellformedness rules under [`f:reifies*` durable encoding](#freifies-durable-encoding) above (required-slot, duplicate-slot, and `f:reifiesGraph`-vs-`g` `GraphMismatch` rejection) are guaranteed only on the **decode paths** — warmup / arena-build / hydration / cascade. Query-side annotation matching keys off the subject/predicate/object slots within the scoped graph; a malformed bundle introduced by bulk import (e.g. a named-graph bundle missing `f:reifiesGraph`) that decode would skip can still satisfy an `@annotation` / `rdf:reifies` query. Well-formed Fluree writes never produce such bundles — the JSON-LD writer emits `f:reifiesGraph` for named-graph edges and the SPARQL UPDATE surface is default-graph only — so this boundary matters only for hand-authored or externally-imported `f:reifies*` data. @@ -193,7 +201,7 @@ Multi-source default-graph datasets (`from: [g1, g2]`) wrap each expanded chain `elide_redundant_chain` rewrites the generic chain before it is planned. The write invariants — every reifier carries exactly one `f:reifiesSubject` / `f:reifiesPredicate` / `f:reifiesObject`, written into the edge's own graph, with the base triple asserted (`@reifies` without it is rejected; retracting the base cascades) — mean the base edge never removes a row once a reifies lookup has bound the reifier, and a lookup whose position is a variable nothing reads (neither a later sibling, nor the post-WHERE pipeline, nor the wrapper's own body) is a cardinality-one no-op. So the base edge is dropped, each unread variable position drops its lookup, constant positions keep theirs, and at least one lookup always remains (a reifier bound by the body still has to be a reifier: P3 has three `:derives_from` rows on plain subjects that P2 must not count). A read *variable predicate* blocks the whole rewrite — the base scan binds it as a predicate id, the reifies lookup as a plain ref, and those are not interchangeable downstream. The rewrite is gated on root policy, a single ledger and a non-history query; a past-`t` snapshot qualifies, because a reifier's bundle and its base edge are written and cascaded in the same commit. -Two things count as a read that are easy to miss. A variable in two positions, or naming the reifier, keeps its lookups: once the base edge is gone they are the only thing equating those positions (`<< ?s :p ?s >>` must not match `:a :p :b`). And under `SELECT *` there is no projection-pushdown set, so the post-WHERE read set is every variable `collect_var_stats` finds in the un-expanded WHERE clause. That walk has to look inside `EdgeAnnotation` / `AnnotationTarget`, or the quoted triple's own variables are dropped from every row (W3C `sparql12/eval-triple-terms` basic-4, basic-5, pattern-7, pattern-8). +Two things count as a read that are easy to miss. A variable in two positions, or naming the reifier, keeps its lookups: once the base edge is gone they are the only thing equating those positions (`<< ?s :p ?s >>` must not match `:a :p :b`). And under `SELECT *` there is no projection-pushdown set, so the post-WHERE read set is every variable `collect_var_stats` finds in the un-expanded WHERE clause. That walk has to look inside `EdgeAnnotation`, or the annotated triple's own variables are dropped from every row (W3C `sparql12/eval-triple-terms` basic-4, basic-5, pattern-7, pattern-8). Measured on the full StarBench ledger the predicate-only chain went from one SPOT point scan per reifier to one POST range plus the body's batched probe: P11 4.9 s → 0.42 s, C11 8.4 s → 0.61 s, S5 6.9 s → 1.8 s, and P2 on the elided chain (a single `f:reifiesSubject` range over all 21.4M reifiers) 10.6 s against 65 s on the enumeration — which is why the elided chain is costed at ~2 rows per reifier and the enumeration at ~12 per pair. diff --git a/fluree-db-api/src/explain.rs b/fluree-db-api/src/explain.rs index f32c557678..2bdb625e6b 100644 --- a/fluree-db-api/src/explain.rs +++ b/fluree-db-api/src/explain.rs @@ -345,10 +345,6 @@ fn logical_node( node.insert("kind".into(), json!("edge-annotation")); node.insert("patterns".into(), children(body)); } - Pattern::AnnotationTarget { body, .. } => { - node.insert("kind".into(), json!("annotation-target")); - node.insert("patterns".into(), children(body)); - } } JsonValue::Object(node) } @@ -537,9 +533,8 @@ fn explain_from_parsed( } // Expand edge-annotation IR into the same triple chain the - // executor uses (`Pattern::EdgeAnnotation` / - // `Pattern::AnnotationTarget` → base edge + 3 `f:reifies*` lookups - // + body). Without this, edge-annotation queries appear as empty + // executor uses (`Pattern::EdgeAnnotation` → base edge + 3 + // `f:reifies*` lookups + body). Without this, edge-annotation queries appear as empty // in `/explain` output because `collect_triples_in_order` doesn't // descend into those container patterns. let expanded_patterns = expand_edge_annotation_patterns(&parsed.patterns); diff --git a/fluree-db-api/src/format/construct.rs b/fluree-db-api/src/format/construct.rs index 0d1446972d..c9937ca200 100644 --- a/fluree-db-api/src/format/construct.rs +++ b/fluree-db-api/src/format/construct.rs @@ -100,7 +100,8 @@ pub(super) fn instantiate_construct_graph( Ref::Var(v) => Ok(terms_slot(template, *v)), constant => terms.constant_ref(constant, position).map(Slot::Const), }; - // Each pattern's slots, graph, and the reifiers attached to it. + // Each pattern's slots, graph, the reifiers attached to it, and the + // components of a constant triple-term object. let mut patterns = Vec::with_capacity(template.patterns().len()); for (i, pattern) in template.patterns().iter().enumerate() { let s = slot_of(&mut terms, &pattern.s, Position::Subject)?; @@ -109,11 +110,15 @@ pub(super) fn instantiate_construct_graph( Term::Var(v) => terms_slot(template, *v), constant => Slot::Const(terms.constant_object(constant, pattern.dtc.as_ref())?), }; + let const_term = match &pattern.o { + Term::Value(FlakeValue::TripleTerm(term)) => terms.term_components(term)?, + _ => None, + }; let graph = match template.graph(i) { Some(g) => Some(slot_of(&mut terms, g, Position::Graph)?), None => None, }; - patterns.push(([s, p, o], graph, Vec::new())); + patterns.push(([s, p, o], graph, Vec::new(), const_term)); } for r in template.reifications() { let reifier = slot_of(&mut terms, &r.reifier, Position::Subject)?; @@ -153,37 +158,38 @@ pub(super) fn instantiate_construct_graph( }, }) }; - 'pattern: for (slots, graph_slot, reifier_slots) in &patterns { - // `?r rdf:reifies ?t` with a triple term bound to ?t writes - // what `?r rdf:reifies <<( s p o )>>` does: the term's triple - // and ?r's reification of it. - if let Slot::Var(v) = &slots[2] { - if let Some(components) = batch + 'pattern: for (slots, graph_slot, reifier_slots, const_term) in &patterns { + // `?r rdf:reifies ?t` with a triple term bound to ?t (or a + // constant one) writes what `?r rdf:reifies <<( s p o )>>` + // does: the term's triple and ?r's reification of it. + let components = match &slots[2] { + Slot::Var(v) => batch .get(row, *v) .map(|b| terms.triple_term(b)) .transpose()? - .flatten() - { - let (Some(reifier), Some(p)) = ( - resolve(&mut terms, &slots[0], Position::Subject)?, - resolve(&mut terms, &slots[1], Position::Predicate)?, - ) else { - continue 'pattern; + .flatten(), + _ => const_term.clone(), + }; + if let Some(components) = components { + let (Some(reifier), Some(p)) = ( + resolve(&mut terms, &slots[0], Position::Subject)?, + resolve(&mut terms, &slots[1], Position::Predicate)?, + ) else { + continue 'pattern; + }; + if matches!(&p, IrTerm::Iri(iri) if &**iri == rdf::REIFIES) { + let graph = match graph_slot { + Some(slot) => match resolve(&mut terms, slot, Position::Graph)? { + Some(name) => Some(name), + None => continue 'pattern, + }, + None => None, }; - if matches!(&p, IrTerm::Iri(iri) if &**iri == rdf::REIFIES) { - let graph = match graph_slot { - Some(slot) => match resolve(&mut terms, slot, Position::Graph)? { - Some(name) => Some(name), - None => continue 'pattern, - }, - None => None, - }; - let [ts, tp, to] = components; - let g = dataset.graph_mut(graph.as_ref()); - g.add_reification(ts.clone(), tp.clone(), to.clone(), reifier); - g.add(Triple::new(ts, tp, to)); - continue 'pattern; - } + let [ts, tp, to] = components; + let g = dataset.graph_mut(graph.as_ref()); + g.add_reification(ts.clone(), tp.clone(), to.clone(), reifier); + g.add(Triple::new(ts, tp, to)); + continue 'pattern; } } let mut triple: [Option; 3] = [None, None, None]; @@ -384,8 +390,7 @@ impl TermResolver<'_> { } /// A binding's triple term as the IR terms of its subject, predicate and - /// object, when it holds one whose object the graph IR can carry (any - /// term but a nested triple term). + /// object (see [`Self::term_components`]). fn triple_term(&mut self, binding: &Binding) -> Result> { if binding.is_encoded() { let materialized = super::materialize::materialize_binding(self.result, binding)?; @@ -398,6 +403,15 @@ impl TermResolver<'_> { else { return Ok(None); }; + self.term_components(term) + } + + /// A triple term as the IR terms of its subject, predicate and object, + /// when the graph IR can carry its object (any term but a nested one). + fn term_components( + &mut self, + term: &fluree_db_core::TripleTermValue, + ) -> Result> { if matches!(term.o, FlakeValue::TripleTerm(_)) { return Ok(None); } diff --git a/fluree-db-api/src/format/hydration.rs b/fluree-db-api/src/format/hydration.rs index 947ae8397a..088f1c00b2 100644 --- a/fluree-db-api/src/format/hydration.rs +++ b/fluree-db-api/src/format/hydration.rs @@ -1369,8 +1369,8 @@ impl<'a> HydrationFormatter<'a> { // // Explicitly-listed levels can still reach these via // a `Pattern::Triple` lookup at the planner layer - // (which is what the `Pattern::EdgeAnnotation` / - // `AnnotationTarget` IR expansion does), but those + // (which is what the `Pattern::EdgeAnnotation` IR + // expansion and the `rdf:reifies` link lowering do), but those // patterns don't go through hydration. if fluree_db_core::is_scan_hidden_predicate(&pred) { continue; diff --git a/fluree-db-api/tests/grp_triple_terms.rs b/fluree-db-api/tests/grp_triple_terms.rs index df82b0003a..302b63c194 100644 --- a/fluree-db-api/tests/grp_triple_terms.rs +++ b/fluree-db-api/tests/grp_triple_terms.rs @@ -1,6 +1,5 @@ #[path = "support/mod.rs"] mod support; -// Own binary: the link-lowering tests set a process-wide environment flag. #[path = "it_triple_term_links.rs"] mod it_triple_term_links; diff --git a/fluree-db-api/tests/it_edge_annotations_indexed.rs b/fluree-db-api/tests/it_edge_annotations_indexed.rs index fc115ee366..8ac6dc92d3 100644 --- a/fluree-db-api/tests/it_edge_annotations_indexed.rs +++ b/fluree-db-api/tests/it_edge_annotations_indexed.rs @@ -840,8 +840,8 @@ async fn non_annotation_ledger_skips_inject_annotations() { #[tokio::test] async fn explain_tags_annotation_role_and_uses_arena_stats() { - // M3.2: `/explain` must (a) expand `@annotation` / `@reifies` - // patterns the same way the executor does, (b) tag the resulting + // M3.2: `/explain` must (a) expand `@annotation` patterns the same + // way the executor does, (b) tag the resulting // `f:reifies*` triples with their slot name so the chosen ordering // is observable, and (c) report stats as available when the // annotation arena is sealed even if no other property stats @@ -876,18 +876,18 @@ async fn explain_tags_annotation_role_and_uses_arena_stats() { "arena must be sealed for the explain test to exercise M3.1 stats" ); - // `@reifies`-rooted query: filter by annotation metadata, - // ask for the edge it reifies. Lowering produces a - // `Pattern::AnnotationTarget` which `/explain` should now - // expand into a base edge triple + 3 `f:reifies*` lookups. + // Edge-rooted query filtered by annotation metadata. Lowering + // produces a `Pattern::EdgeAnnotation`, which `/explain` + // expands into a base edge triple + 3 `f:reifies*` lookups. + // (`@reifies` reads the `rdf:reifies` link instead.) let query = json!({ "@context": ctx(), "select": ["?person", "?org"], "where": { - "ex:role": "Engineer", - "@reifies": { - "@id": "?person", - "ex:worksFor": { "@id": "?org" } + "@id": "?person", + "ex:worksFor": { + "@id": "?org", + "@annotation": { "ex:role": "Engineer" } } } }); diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index edaa534a72..3bf6b0668d 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -255,13 +255,12 @@ async fn incremental_index_appends_new_terms_and_reuses_existing_handles() { .await; } -/// The link-based lowering (`FLUREE_ANNOTATION_TERMS=1`): reified-triple -/// patterns scan `rdf:reifies` and decompose or constrain the term instead -/// of walking the bundle chain. Every shape the design doc's access-path -/// table names, on the same imported claims. +/// The link lowering: reified-triple patterns scan `rdf:reifies` and +/// decompose or constrain the term instead of walking the bundle chain. +/// Every shape the design doc's access-path table names, on the same +/// imported claims. #[tokio::test] async fn link_lowering_answers_reified_triple_shapes() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let (fluree, ledger) = import(&[("claims.ttl", CLAIMS)], "it/triple-term-links:lowering").await; let q = |body: &str| { format!( @@ -376,7 +375,6 @@ async fn run_link_query( /// keeps the tag and datatype too. #[tokio::test] async fn link_lowering_matches_literal_objects_by_term() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let (fluree, ledger) = import( &[("literals.ttl", LITERAL_CLAIMS)], "it/triple-term-links:literals", @@ -426,7 +424,6 @@ async fn link_lowering_matches_literal_objects_by_term() { /// already holds may be in any representation. #[tokio::test] async fn link_lowering_joins_component_variables_in_every_scope() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let (fluree, ledger) = import(&[("claims.ttl", CLAIMS)], "it/triple-term-links:scopes").await; let run = |body: &str| run_link_query(&fluree, &ledger, body.to_string()); @@ -478,7 +475,6 @@ async fn link_lowering_joins_component_variables_in_every_scope() { /// predicates admit no handle: the second filter stays in the plan. #[tokio::test] async fn link_lowering_keeps_contradictory_predicate_filters() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let (fluree, ledger) = import( &[("claims.ttl", CLAIMS)], "it/triple-term-links:contradiction", @@ -500,7 +496,6 @@ async fn link_lowering_keeps_contradictory_predicate_filters() { /// still the count of matching links. #[tokio::test] async fn link_lowering_counts_without_decomposing_unread_positions() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let (fluree, ledger) = import(&[("claims.ttl", CLAIMS)], "it/triple-term-links:count").await; let got = run_link_query( &fluree, @@ -672,7 +667,6 @@ async fn assert_object_types_survive(fluree: &fluree_db_api::Fluree, ledger: &Le /// path once an unrelated unindexed commit turns late materialization off. #[tokio::test] async fn link_lowering_keeps_object_types_through_accessors_and_novelty() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let (fluree, ledger) = import( &[("literals.ttl", LITERAL_CLAIMS)], "it/triple-term-links:accessor-types", @@ -889,7 +883,6 @@ ex:d ex:q ex:e ~ ex:r3 {| ex:src ex:z |} . /// carry the first one's column to it although only a count reads it. #[tokio::test] async fn link_lowering_joins_chained_edges_when_only_counted() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let (fluree, ledger) = import( &[("chained.ttl", CHAINED_CLAIMS)], "it/triple-term-links:chained", @@ -916,7 +909,6 @@ async fn link_lowering_joins_chained_edges_when_only_counted() { /// links the first edge's scan is cheapest.) #[tokio::test] async fn link_lowering_drives_chained_edges_through_their_subjects() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let alias = "it/triple-term-links:chained-plan"; let (fluree, _ledger) = import(&[("chained.ttl", CHAINED_CLAIMS)], alias).await; let view = fluree.db(alias).await.expect("view"); @@ -1092,7 +1084,6 @@ async fn change_claims_without_indexing( /// where the links wait for the index store before they can be derived. #[tokio::test] async fn novelty_carries_links_until_the_index_covers_them() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let alias = "it/triple-term-links:novelty"; let db_dir = tempfile::tempdir().expect("db tmpdir"); let data_dir = tempfile::tempdir().expect("data tmpdir"); @@ -1166,7 +1157,6 @@ async fn novelty_carries_links_until_the_index_covers_them() { /// term dictionary at all. #[tokio::test] async fn novelty_carries_links_on_a_ledger_never_indexed() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let fluree = FlureeBuilder::memory().build_memory(); let ledger = support::genesis_ledger(&fluree, "it/triple-term-links:never-indexed"); let ledger = fluree @@ -1187,7 +1177,6 @@ async fn novelty_carries_links_on_a_ledger_never_indexed() { #[tokio::test] async fn novelty_links_follow_the_index_they_were_published_over() { use fluree_db_indexer::IndexerConfig; - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let fluree = FlureeBuilder::memory() .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) @@ -1259,7 +1248,6 @@ async fn novelty_links_follow_the_index_they_were_published_over() { /// overlay instead of the raw-flake lane the plan declines on. #[tokio::test(flavor = "current_thread")] async fn novelty_terms_keep_link_counts_on_the_count_plan() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let (fluree, ledger) = import( &[("claims.ttl", CLAIMS)], "it/triple-term-links:novelty-count", @@ -1367,7 +1355,6 @@ async fn assert_arena_links(fluree: &fluree_db_api::Fluree, ledger: &LedgerState /// string ids are not the global ones. #[tokio::test] async fn arena_kind_objects_link_through_import_and_reindex() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let alias = "it/triple-term-links:arena"; let pad = "@prefix ex: .\nex:pad ex:label \"!pad\" , \"0pad\" .\n"; let (fluree, ledger) = import(&[("a.ttl", pad), ("b.ttl", ARENA_CLAIMS)], alias).await; @@ -1408,7 +1395,6 @@ async fn arena_kind_objects_link_through_import_and_reindex() { /// A ledger never indexed holds these links in novelty. #[tokio::test] async fn arena_kind_objects_link_in_novelty() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let fluree = FlureeBuilder::memory().build_memory(); let ledger = support::genesis_ledger(&fluree, "it/triple-term-links:arena-novelty"); let ledger = fluree @@ -1427,7 +1413,6 @@ async fn arena_kind_objects_link_in_novelty() { #[tokio::test] async fn arena_kind_links_follow_incremental_repoints_in_every_graph() { use fluree_db_indexer::IndexerConfig; - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let fluree = FlureeBuilder::memory() .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) @@ -1575,7 +1560,6 @@ async fn link_scan_estimate(fluree: &fluree_db_api::Fluree, ledger: &LedgerState #[tokio::test] async fn link_counts_follow_every_build() { use fluree_db_indexer::IndexerConfig; - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let expected = |knows: u64| Some(vec![("age".to_string(), 1), ("knows".to_string(), knows)]); let alias = "it/triple-term-links:link-counts"; @@ -1630,7 +1614,6 @@ async fn link_counts_follow_every_build() { #[tokio::test] async fn first_annotation_after_an_index_without_terms() { use fluree_db_indexer::IndexerConfig; - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let fluree = FlureeBuilder::memory() .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) @@ -1720,7 +1703,6 @@ fn strings(rows: &[&[&str]]) -> Vec> { #[tokio::test(flavor = "current_thread")] async fn object_bound_terms_read_the_object_tree() { use fluree_db_indexer::IndexerConfig; - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let expected = |extra: bool| { let mut knows_bob = vec![vec!["ex:alice".to_string(), "ex:hr".to_string()]]; if extra { @@ -1920,7 +1902,6 @@ async fn triple_constructs_the_terms_links_hold() { #[cfg(feature = "shacl")] #[tokio::test] async fn shacl_sparql_constraints_see_the_transactions_links() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let fluree = FlureeBuilder::memory().build_memory(); let ledger = support::genesis_ledger(&fluree, "it/triple-term-links:shacl-staged"); let ledger = fluree @@ -1960,7 +1941,6 @@ ex:noMallory sh:message "no claim may say alice knows mallory" ; /// annotations, over an index and over a ledger never indexed. #[tokio::test] async fn previews_read_the_links_of_their_own_annotations() { - std::env::set_var("FLUREE_ANNOTATION_TERMS", "1"); let (fluree, indexed) = import(&[("claims.ttl", CLAIMS)], "it/triple-term-links:preview").await; let memory = FlureeBuilder::memory().build_memory(); let never_indexed = memory @@ -2125,6 +2105,7 @@ async fn triple_terms_render_in_every_result_format() { /// is written as its N-Triples text. #[tokio::test] async fn construct_writes_triple_terms_as_reifications() { + const REIFIES: &str = ""; use fluree_db_api::format::{format_results_string, FormatterConfig}; let (fluree, ledger) = import( &[("claims.ttl", CLAIMS), ("literals.ttl", LITERAL_CLAIMS)], @@ -2188,6 +2169,28 @@ async fn construct_writes_triple_terms_as_reifications() { json!([{"@id": "ex:alice", "ex:knows": [{"@id": "ex:bob", "@annotation": {"@id": "ex:claim1"}}]}]) ); + // The shorthand's template is the WHERE clause: a constant quoted triple + // lowers to a link with a constant term, written the same way. + let sparql = "PREFIX ex: \n\ + CONSTRUCT WHERE { << ex:alice ex:knows ex:bob ~ ?r >> ex:source ?src }"; + let shorthand = support::query_sparql(&fluree, &ledger, sparql) + .await + .expect("construct where"); + let ex = |l: &str| format!(""); + let mut expected = vec![ + format!("{} {} {} .", ex("alice"), ex("knows"), ex("bob")), + format!( + "{} {REIFIES} <<( {} {} {} )>> .", + ex("claim1"), + ex("alice"), + ex("knows"), + ex("bob") + ), + format!("{} {} {} .", ex("claim1"), ex("source"), ex("hr")), + ]; + expected.sort(); + assert_eq!(sorted(&shorthand), expected); + let about = construct("?r ex:about ?t", "?r rdf:reifies ?t ; ex:source ex:fr").await; let nt = render(&about, FormatterConfig::ntriples()); assert!( @@ -2197,3 +2200,109 @@ async fn construct_writes_triple_terms_as_reifications() { "{nt}" ); } + +/// An annotated ledger whose index predates links (simulated by dropping the +/// root's term dictionary) refuses link reads until a full rebuild links its +/// annotations. An incremental build in between does not start a term +/// dictionary, which would cover its window alone and lift the refusal. +#[tokio::test] +async fn link_reads_refuse_an_index_built_before_links() { + use fluree_db_binary_index::format::index_root::IndexRoot; + use fluree_db_core::{ContentKind, ContentStore}; + + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/triple-term-links:pre-link"; + let claim = |n: u32| { + format!( + "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:s{n} ex:knows ex:o{n} ~ ex:claim{n} {{| ex:src ex:x |}} .\n" + ) + }; + let current_root = || async { + let record = fluree + .nameservice() + .lookup(ledger_id) + .await + .expect("ns lookup") + .expect("ns record"); + let cid = record.index_head_id.clone().expect("index root"); + let bytes = fluree + .content_store(ledger_id) + .get(&cid) + .await + .expect("root"); + (IndexRoot::decode(&bytes).expect("decode"), record.index_t) + }; + let query = "PREFIX ex: \n\ + SELECT ?r WHERE { << ?s ex:knows ?o ~ ?r >> ex:src ex:x } ORDER BY ?r"; + + let ledger = support::genesis_ledger(&fluree, ledger_id); + fluree + .upsert_turtle(ledger, &claim(1)) + .await + .expect("claim 1"); + support::rebuild_and_publish_index(&fluree, ledger_id).await; + + let (mut root, index_t) = current_root().await; + assert!(root.has_annotations && root.term_dict.is_some()); + root.term_dict = None; + let cid = fluree + .content_store(ledger_id) + .put(ContentKind::IndexRoot, &root.encode()) + .await + .expect("put root"); + fluree + .publisher() + .expect("read-write nameservice") + .publish_index_allow_equal(ledger_id, index_t, &cid) + .await + .expect("publish root"); + + let ledger = fluree.ledger(ledger_id).await.expect("load"); + let err = support::query_sparql(&fluree, &ledger, query) + .await + .expect_err("a pre-link index refuses link reads"); + assert!(err.to_string().contains("fluree reindex"), "{err}"); + + fluree + .upsert_turtle(ledger, &claim(2)) + .await + .expect("claim 2"); + support::build_and_publish_index(&fluree, ledger_id).await; + let (root, _) = current_root().await; + assert!( + root.term_dict.is_none(), + "an incremental build over a pre-link index must not start a term dictionary" + ); + let ledger = fluree.ledger(ledger_id).await.expect("load"); + assert!(support::query_sparql(&fluree, &ledger, query) + .await + .is_err()); + + // Same `t` as the incremental root, so it publishes with allow-equal. + let record = fluree + .nameservice() + .lookup(ledger_id) + .await + .expect("ns lookup") + .expect("ns record"); + let rebuilt = fluree_db_indexer::rebuild_index_from_commits( + fluree.content_store(ledger_id), + ledger_id, + &record, + fluree_db_indexer::IndexerConfig::default(), + ) + .await + .expect("rebuild"); + fluree + .publisher() + .expect("read-write nameservice") + .publish_index_allow_equal(ledger_id, rebuilt.index_t, &rebuilt.root_id) + .await + .expect("publish rebuild"); + let ledger = fluree.ledger(ledger_id).await.expect("load"); + let result = support::query_sparql_formatted(&fluree, &ledger, query) + .await + .expect("a rebuilt index answers"); + assert_eq!(rows(&result), strings(&[&["ex:claim1"], &["ex:claim2"]])); +} diff --git a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs index 7553cd287b..c750975b71 100644 --- a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs +++ b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs @@ -287,8 +287,14 @@ pub async fn resolve_incremental_commits_v6( // 4. Seed SharedResolverState from V6 root. let mut shared = SharedResolverState::from_index_root(&root)?; - // Incremental builds resolve term ordinals in step 9a, so links are synthesized. - shared.link_synth.enable(); + // Incremental builds resolve term ordinals in step 9a, so links are + // synthesized — except over an annotated index built before links, whose + // earlier annotations only a full rebuild links: a term dictionary started + // here would cover the window alone, and readers take its presence to mean + // every annotation is linked. + if !(root.has_annotations && root.term_dict.is_none()) { + shared.link_synth.enable(); + } // Enable spatial hook for non-POINT geometry detection. shared.spatial_hook = Some(crate::spatial_hook::SpatialHook::new()); diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index 291573d4a2..4aee635163 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -2111,6 +2111,9 @@ impl Operator for BinaryScanOperator { return Err(QueryError::OperatorAlreadyOpened); } + if matches!(&self.pattern.p, Ref::Sid(p) if fluree_db_core::is_rdf_reifies(p)) { + crate::term_components::require_indexed_links(ctx)?; + } // Resolve store and g_id from context. self.store = ctx.binary_store.clone(); self.g_id = ctx.binary_g_id; diff --git a/fluree-db-query/src/datalog_rules/parse.rs b/fluree-db-query/src/datalog_rules/parse.rs index ec6aee3dc7..109c25621a 100644 --- a/fluree-db-query/src/datalog_rules/parse.rs +++ b/fluree-db-query/src/datalog_rules/parse.rs @@ -305,8 +305,7 @@ fn collect_predicates(patterns: &[Pattern], out: &mut Vec) { } } } - Pattern::EdgeAnnotation { edge, body, .. } - | Pattern::AnnotationTarget { edge, body, .. } => { + Pattern::EdgeAnnotation { edge, body, .. } => { if let Ref::Sid(p) = &edge.p { if !out.contains(p) { out.push(p.clone()); diff --git a/fluree-db-query/src/datalog_rules/validate.rs b/fluree-db-query/src/datalog_rules/validate.rs index 26de19669e..2792b87e32 100644 --- a/fluree-db-query/src/datalog_rules/validate.rs +++ b/fluree-db-query/src/datalog_rules/validate.rs @@ -97,8 +97,7 @@ fn walk_patterns( patterns: inner, .. } | Pattern::DefaultGraphSource { patterns: inner } - | Pattern::EdgeAnnotation { body: inner, .. } - | Pattern::AnnotationTarget { body: inner, .. } => { + | Pattern::EdgeAnnotation { body: inner, .. } => { walk_patterns(inner, snapshot, rule, origin)?; } // Graph-source patterns read data from outside the ledger's own @@ -550,8 +549,7 @@ fn collect_positions( literal_capable.insert(*v); } } - Pattern::EdgeAnnotation { edge, body, .. } - | Pattern::AnnotationTarget { edge, body, .. } => { + Pattern::EdgeAnnotation { edge, body, .. } => { collect_positions( std::slice::from_ref(&Pattern::Triple(edge.clone())), iri_position, diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index c83ce8f3f9..8b2aa21a6a 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -50,18 +50,16 @@ use super::pushdown::extract_bounds_from_filters; // Edge-annotation IR expansion (M1b) // ============================================================================ // -// `Pattern::EdgeAnnotation { edge, annotation, body }` and -// `Pattern::AnnotationTarget { annotation, edge, body }` are flattened +// `Pattern::EdgeAnnotation { edge, annotation, body }` is flattened // at planner time into the equivalent triple chain over the // `f:reifies*` system predicates. The standard scan / join machinery // handles the rest. This avoids a custom operator and exercises the // existing visibility / policy / dedup paths automatically — the -// base edge triple's standard scan provides the visibility check for -// the reverse direction "for free". +// base edge triple's standard scan provides the visibility check +// "for free". -/// Expand every `Pattern::EdgeAnnotation` / `Pattern::AnnotationTarget` -/// in `patterns` into its triple-chain equivalent, recursing through -/// every container pattern (`Optional`, `Union`, `Minus`, `Exists`, +/// Expand every `Pattern::EdgeAnnotation` in `patterns` into its +/// triple-chain equivalent, recursing through every container pattern (`Optional`, `Union`, `Minus`, `Exists`, /// `NotExists`, `Graph`, `Service`, `Subquery`). /// /// Each expanded triple chain is wrapped in @@ -84,15 +82,15 @@ pub fn expand_edge_annotation_patterns(patterns: &[Pattern]) -> Vec { out } -/// Cheap, allocation-free check for any `Pattern::EdgeAnnotation` / -/// `Pattern::AnnotationTarget` anywhere in the tree. Lets the WHERE +/// Cheap, allocation-free check for any `Pattern::EdgeAnnotation` +/// anywhere in the tree. Lets the WHERE /// planner skip the `expand_edge_annotation_patterns` clone+rebuild on /// the common non-RDF-1.2 path. Exhaustive over `Pattern` so a new /// container variant forces a decision here rather than silently hiding /// an annotation from expansion. pub(crate) fn pattern_tree_has_edge_annotation(patterns: &[Pattern]) -> bool { patterns.iter().any(|p| match p { - Pattern::EdgeAnnotation { .. } | Pattern::AnnotationTarget { .. } => true, + Pattern::EdgeAnnotation { .. } => true, Pattern::Optional(inner) | Pattern::Minus(inner) | Pattern::Exists(inner) @@ -124,11 +122,6 @@ fn expand_one_into(pattern: Pattern, out: &mut Vec, inside_graph: bool) edge, annotation, body, - } - | Pattern::AnnotationTarget { - annotation, - edge, - body, } => { // Build the triple chain (base edge + three f:reifies* // triples + recursively expanded body) into a local @@ -139,10 +132,9 @@ fn expand_one_into(pattern: Pattern, out: &mut Vec, inside_graph: bool) // which provides graph correlation by construction. let mut chain: Vec = Vec::new(); - // 1. Base edge triple: provides visibility for both - // directions. The standard scan applies snapshot rules - // + policy filters here, so an `AnnotationTarget` - // operator-style visibility check is redundant. + // 1. Base edge triple: provides visibility. The standard + // scan applies snapshot rules + policy filters here, so + // an operator-style visibility check is redundant. chain.push(Pattern::Triple(edge.clone())); // 2. Three required `f:reifies*` lookup triples that bind @@ -533,9 +525,7 @@ fn has_outer_correlated_graph_var(patterns: &[Pattern], outer: &HashSet) // re-opened the #1443 cross-encoding hash mismatch this fallback // exists to close. Pattern::Service(sp) => has_outer_correlated_graph_var(&sp.patterns, outer), - Pattern::EdgeAnnotation { body, .. } | Pattern::AnnotationTarget { body, .. } => { - has_outer_correlated_graph_var(body, outer) - } + Pattern::EdgeAnnotation { body, .. } => has_outer_correlated_graph_var(body, outer), Pattern::Filter(expr) => expr_embeds(expr, outer), Pattern::Bind { expr, .. } => expr_embeds(expr, outer), Pattern::Unwind { var: _, list } => expr_embeds(list, outer), @@ -819,7 +809,7 @@ pub fn collect_var_stats( // positions, the reifier and the body: the chain elision // treats a variable they miss as unread and drops the // `f:reifies*` lookup that binds it. - Pattern::EdgeAnnotation { .. } | Pattern::AnnotationTarget { .. } => { + Pattern::EdgeAnnotation { .. } => { for v in p.referenced_vars() { bump_count(counts, v); vars.insert(v); @@ -2642,15 +2632,13 @@ pub fn build_where_operators_seeded_with_needed( return Ok(seed.unwrap_or_else(|| Box::new(EmptyOperator::new()))); } - // Edge-annotation expansion (M1b): Pattern::EdgeAnnotation / - // Pattern::AnnotationTarget are flattened into the equivalent base - // edge plus four `f:reifies*` triple lookups plus the body. The - // standard scan/join machinery handles the rest. The base-edge - // triple is always emitted, which gives the - // `Pattern::AnnotationTarget` reverse direction its required - // visibility check for free: the base edge must be currently - // asserted under the snapshot's normal policy/visibility rules, - // or no row survives the join. + // Edge-annotation expansion (M1b): Pattern::EdgeAnnotation is + // flattened into the equivalent base edge plus the `f:reifies*` + // triple lookups plus the body. The standard scan/join machinery + // handles the rest. The base-edge triple is always emitted, which + // gives the annotation its visibility check for free: the base edge + // must be currently asserted under the snapshot's normal + // policy/visibility rules, or no row survives the join. // Skip the clone+rebuild entirely when the block has no edge-annotation // patterns (the overwhelmingly common, non-RDF-1.2 case): borrow the // original slice instead of allocating an expanded copy. The presence @@ -3476,7 +3464,7 @@ pub fn build_where_operators_seeded_with_needed( // dispatch arm is unreachable from any standard entry // point. Keep the guard for safety in case a future // caller bypasses the expansion pass. - Pattern::EdgeAnnotation { .. } | Pattern::AnnotationTarget { .. } => { + Pattern::EdgeAnnotation { .. } => { return Err(QueryError::Internal( "edge-annotation pattern reached the operator dispatch \ without being flattened by expand_edge_annotation_patterns" diff --git a/fluree-db-query/src/explain.rs b/fluree-db-query/src/explain.rs index b2822ccc98..6be3f09018 100644 --- a/fluree-db-query/src/explain.rs +++ b/fluree-db-query/src/explain.rs @@ -628,13 +628,6 @@ pub fn format_general_pattern(pattern: &Pattern) -> String { body.len() ) } - Pattern::AnnotationTarget { edge, body, .. } => { - format!( - "ANNOTATION-TARGET {{ {} | {} body patterns }}", - format_pattern(edge), - body.len() - ) - } Pattern::DefaultGraphSource { patterns } => { format!("DEFAULT-GRAPH-SOURCE {{ {} patterns }}", patterns.len()) } diff --git a/fluree-db-query/src/ir.rs b/fluree-db-query/src/ir.rs index 528299cd8e..2c451ab186 100644 --- a/fluree-db-query/src/ir.rs +++ b/fluree-db-query/src/ir.rs @@ -62,5 +62,5 @@ pub use projection::{ }; pub use query::{ConstructTemplate, Query, QueryOutput, Restriction, TemplateReification}; pub use reasoning::{ReasoningConfig, ReasoningModes}; -pub use term_components::{Component, TermComponentsPattern}; +pub use term_components::{lower_reified_link, Component, TermComponentsPattern}; pub use triple::{Ref, Term, TriplePattern}; diff --git a/fluree-db-query/src/ir/pattern.rs b/fluree-db-query/src/ir/pattern.rs index cf48c8705a..17e39917af 100644 --- a/fluree-db-query/src/ir/pattern.rs +++ b/fluree-db-query/src/ir/pattern.rs @@ -513,32 +513,6 @@ pub enum Pattern { /// Patterns about the annotation subject. body: Vec, }, - - /// Annotation-rooted pattern — reverse direction of [`Pattern::EdgeAnnotation`]. - /// - /// Lowered from JSON-LD `@reifies`. The enclosing node-map is the - /// annotation subject; `@reifies` names the base triple it reifies. - /// - /// # Semantics - /// - /// One row per `(annotation, base_edge)` pair. The base edge must be - /// **currently asserted and policy-visible** before a row is emitted - /// (M1 enforcement) — otherwise this operator would leak hidden - /// edges via annotation existence. - /// - /// # Execution - /// - /// Planning expands this variant into the base edge plus the - /// corresponding `f:reifies*` lookup chain before operator-tree - /// assembly. - AnnotationTarget { - /// The annotation subject — variable or constant ref. - annotation: Ref, - /// The base edge being reified. - edge: TriplePattern, - /// Patterns about the annotation subject. - body: Vec, - }, } /// Columns of a VALUES table that bind their variable on every row: those with @@ -611,15 +585,6 @@ impl Pattern { annotation, body: f(body), }, - Pattern::AnnotationTarget { - annotation, - edge, - body, - } => Pattern::AnnotationTarget { - annotation, - edge, - body: f(body), - }, Pattern::DefaultGraphSource { patterns } => Pattern::DefaultGraphSource { patterns: f(patterns), }, @@ -722,11 +687,6 @@ impl Pattern { edge, annotation, body, - } - | Pattern::AnnotationTarget { - annotation, - edge, - body, } => { edge.substitute_var(old, new); if let Ref::Var(v) = annotation { @@ -806,11 +766,6 @@ impl Pattern { edge, annotation, body, - } - | Pattern::AnnotationTarget { - annotation, - edge, - body, } => { let mut vars = edge.referenced_vars(); if let Ref::Var(v) = annotation { @@ -867,11 +822,6 @@ impl Pattern { edge, annotation, body, - } - | Pattern::AnnotationTarget { - annotation, - edge, - body, } => { let mut vars = edge.produced_vars(); if let Ref::Var(v) = annotation { @@ -908,7 +858,7 @@ impl Pattern { .any(|branch| branch.iter().any(|p| p.contains_function(target))), Pattern::Graph { patterns, .. } => patterns.iter().any(|p| p.contains_function(target)), Pattern::Subquery(sq) => sq.patterns.iter().any(|p| p.contains_function(target)), - Pattern::EdgeAnnotation { body, .. } | Pattern::AnnotationTarget { body, .. } => { + Pattern::EdgeAnnotation { body, .. } => { body.iter().any(|p| p.contains_function(target)) } Pattern::DefaultGraphSource { patterns, .. } => { diff --git a/fluree-db-query/src/ir/term_components.rs b/fluree-db-query/src/ir/term_components.rs index 703f7e6c03..ce2c49016e 100644 --- a/fluree-db-query/src/ir/term_components.rs +++ b/fluree-db-query/src/ir/term_components.rs @@ -6,7 +6,11 @@ //! is a join the planner sees, and a bound subject can drive the term //! dictionary's subject-first reverse tree instead of reading every link. -use crate::var_registry::VarId; +use crate::binding::Binding; +use crate::ir::triple::{Ref, Term, TriplePattern}; +use crate::ir::{Expression, Function, Pattern}; +use crate::parse::encode::IriEncoder; +use crate::var_registry::{VarId, VarRegistry}; use fluree_db_core::{DatatypeConstraint, FlakeValue, Sid}; /// One component position of a [`TermComponentsPattern`]. @@ -56,3 +60,159 @@ impl TermComponentsPattern { self.referenced_vars() } } + +/// Lower a reified-triple pattern to the link form: one +/// `annotation rdf:reifies ?__term` triple whose object is a triple-term +/// handle, and `TermComponents(?__term, s, p, o)` relating the term to its +/// components. A variable component is bound by that relation, which joins +/// on it when another pattern bound it first, in every scope, without the +/// lowering tracking who binds what; a bound subject lets the planner +/// drive the relation through the dictionary's subject prefix. A constant +/// component is also a filter on the link +/// (`FILTER(sameTerm(PREDICATE(?__term),

))`), which the planner turns +/// into the scan's handle interval and key check when the link leads. A +/// fully constant edge composes to a constant term the scan looks up +/// directly. +pub fn lower_reified_link( + annotation_ref: Ref, + edge: TriplePattern, + encoder: &E, + vars: &mut VarRegistry, + out: &mut Vec, +) { + let reifies = encoder.encode_ref(fluree_vocab::rdf::REIFIES); + + // Fully constant edge: compose the term itself. + if let Some(term) = constant_term(&edge) { + out.push(Pattern::Triple(TriplePattern { + s: annotation_ref, + p: reifies, + o: Term::Value(FlakeValue::TripleTerm(Box::new(term))), + dtc: None, + })); + return; + } + + let t = fresh_term_var(vars); + out.push(Pattern::Triple(TriplePattern { + s: annotation_ref, + p: reifies, + o: Term::Var(t), + dtc: None, + })); + + let accessor = |f: Function| Expression::call(f, vec![Expression::Var(t)]); + let same_term = |f: Function, constant: Expression| { + Pattern::Filter(Expression::call( + Function::SameTerm, + vec![accessor(f), constant], + )) + }; + let TriplePattern { s, p, o, dtc } = edge; + // Constant positions are filters on the link (the scan narrows on + // them) and constants of the components relation (a constant subject + // anchors it). Variable positions are bound by the components relation, + // which joins on them when another pattern bound them first. + let component = |func: Function, term: Term, out: &mut Vec| match term { + Term::Var(v) => Component::Var(v), + Term::Sid(sid) => { + out.push(same_term( + func, + Expression::Const(FlakeValue::Ref(sid.clone())), + )); + Component::Node(sid) + } + Term::Iri(iri) => match encoder.encode_iri(&iri) { + Some(sid) => { + out.push(same_term( + func, + Expression::Const(FlakeValue::Ref(sid.clone())), + )); + Component::Node(sid) + } + // An IRI in no registered namespace names nothing in this + // ledger, so the pattern cannot match. + None => { + out.push(Pattern::Filter(Expression::Const(FlakeValue::Boolean( + false, + )))); + Component::Any + } + }, + // A literal matches by term identity: its datatype or language + // tag is part of what `<< ?s :p "chat"@fr >>` asks for. + Term::Value(v) => match &dtc { + Some(dtc) => { + out.push(same_term( + func, + Expression::Resolved(Box::new(Binding::Lit { + val: v.clone(), + dtc: dtc.clone(), + t: None, + op: None, + p_id: None, + })), + )); + Component::Literal(v, dtc.clone()) + } + None => { + out.push(Pattern::Filter(Expression::call( + Function::Eq, + vec![accessor(func), Expression::Const(v)], + ))); + Component::Any + } + }, + }; + let tc = TermComponentsPattern { + term: t, + subject: component(Function::TripleSubject, Term::from(s), out), + predicate: component(Function::TriplePredicate, Term::from(p), out), + object: component(Function::TripleObject, o, out), + }; + if tc.components().into_iter().any(|c| c.var().is_some()) + || matches!(tc.subject, Component::Node(_)) + { + out.push(Pattern::TermComponents(tc)); + } +} + +/// A `?__term_N` variable no pattern uses yet. +fn fresh_term_var(vars: &mut VarRegistry) -> VarId { + let name = (vars.len()..) + .map(|n| format!("?__term_{n}")) + .find(|name| vars.get(name).is_none()) + .expect("an unused name"); + vars.get_or_insert(&name) +} + +/// The materialized term for an edge whose three positions are constants +/// and whose object datatype is known; `None` otherwise. +fn constant_term(edge: &TriplePattern) -> Option { + let s = match &edge.s { + Ref::Sid(s) => s.clone(), + _ => return None, + }; + let p = match &edge.p { + Ref::Sid(p) => p.clone(), + _ => return None, + }; + let (o, dt, lang) = match (&edge.o, &edge.dtc) { + (Term::Sid(sid), _) => ( + FlakeValue::Ref(sid.clone()), + fluree_db_core::edge::id_datatype_sid(), + None, + ), + (Term::Value(v), Some(DatatypeConstraint::Explicit(dt))) => (v.clone(), dt.clone(), None), + (Term::Value(v), Some(DatatypeConstraint::LangTag(tag))) => ( + v.clone(), + fluree_db_core::Sid::new( + fluree_vocab::namespaces::RDF, + fluree_vocab::rdf_names::LANG_STRING, + ), + Some(tag.to_string()), + ), + _ => return None, + }; + Some(fluree_db_core::TripleTermValue { s, p, o, dt, lang }) +} diff --git a/fluree-db-query/src/parse/lower.rs b/fluree-db-query/src/parse/lower.rs index 245aa870bc..78971c634c 100644 --- a/fluree-db-query/src/parse/lower.rs +++ b/fluree-db-query/src/parse/lower.rs @@ -459,12 +459,16 @@ pub fn lower_unresolved_pattern( } => { let lowered_annotation = lower_ref_term(annotation, encoder, vars)?; let lowered_edge = lower_triple_pattern(edge, encoder, vars)?; - let lowered_body = lower_unresolved_patterns(body, encoder, vars, pp_counter)?; - Ok(vec![Pattern::AnnotationTarget { - annotation: lowered_annotation, - edge: lowered_edge, - body: lowered_body, - }]) + let mut out = Vec::new(); + crate::ir::lower_reified_link( + lowered_annotation, + lowered_edge, + encoder, + vars, + &mut out, + ); + out.extend(lower_unresolved_patterns(body, encoder, vars, pp_counter)?); + Ok(out) } } } diff --git a/fluree-db-query/src/planner.rs b/fluree-db-query/src/planner.rs index 734321036b..25ead35c14 100644 --- a/fluree-db-query/src/planner.rs +++ b/fluree-db-query/src/planner.rs @@ -1232,11 +1232,9 @@ pub fn estimate_pattern( // wrapped edge's cardinality as a first approximation. Real // cost-based selection between edge-first and annotation-first // scans arrives in M3 alongside `AnnotationStats`. - Pattern::EdgeAnnotation { edge, .. } | Pattern::AnnotationTarget { edge, .. } => { - PatternEstimate::Source { - row_count: estimate_triple_row_count(edge, bound_vars, stats), - } - } + Pattern::EdgeAnnotation { edge, .. } => PatternEstimate::Source { + row_count: estimate_triple_row_count(edge, bound_vars, stats), + }, // DefaultGraphSource wraps an expanded edge-annotation chain and // runs it once per default-graph source. The chain has its own @@ -1817,11 +1815,6 @@ pub(crate) fn must_bind_vars(pattern: &Pattern, bind_targets: BindTargets) -> Ha edge, annotation, body, - } - | Pattern::AnnotationTarget { - annotation, - edge, - body, } => { let mut vars: HashSet = edge.produced_vars().into_iter().collect(); if let Ref::Var(v) = annotation { @@ -1898,7 +1891,7 @@ fn left_join_introduced_vars(pattern: &Pattern, out: &mut HashSet) { let body = must_bind_vars(pattern, BindTargets::Bound); out.extend(sq.select.iter().copied().filter(|v| !body.contains(v))); } - Pattern::EdgeAnnotation { body, .. } | Pattern::AnnotationTarget { body, .. } => { + Pattern::EdgeAnnotation { body, .. } => { for p in body { left_join_introduced_vars(p, out); } diff --git a/fluree-db-query/src/r2rml/rewrite.rs b/fluree-db-query/src/r2rml/rewrite.rs index 82afce6a05..37de4ad0f1 100644 --- a/fluree-db-query/src/r2rml/rewrite.rs +++ b/fluree-db-query/src/r2rml/rewrite.rs @@ -113,7 +113,7 @@ pub fn unsupported_subscope_error(graph_iris: &[&str], kinds: &[&str]) -> crate: /// - `GRAPH` is its own scope — already covered by the rewrite guard; /// - `SERVICE` targets another ledger or endpoint, which may have a native /// index, so refusing on its body would reject a query that works; -/// - the RDF-star `EdgeAnnotation`/`AnnotationTarget` bodies are left untouched +/// - the RDF-star `EdgeAnnotation` bodies are left untouched /// here exactly as the rewriter leaves them (RDF-star over a graph source is /// undefined rather than confirmed silently-empty). pub fn unsupported_outside_graph_scopes(patterns: &[Pattern]) -> Vec<&'static str> { @@ -159,8 +159,7 @@ fn collect_unsupported_outside_graph_scopes(patterns: &[Pattern], kinds: &mut Ve | Pattern::S2Search(_) | Pattern::Graph { .. } | Pattern::Service(_) - | Pattern::EdgeAnnotation { .. } - | Pattern::AnnotationTarget { .. } => {} + | Pattern::EdgeAnnotation { .. } => {} } } } @@ -364,8 +363,8 @@ pub fn rewrite_patterns_for_r2rml( // IndexSearch/VectorSearch/GeoSearch/S2Search carry their own // `graph_source_id` and route independently; `R2rml` is already // converted; nested `Graph`/`DefaultGraphSource` re-enter routing - // for their own target; the RDF-star `EdgeAnnotation`/ - // `AnnotationTarget` are expanded during planning (RDF-star over + // for their own target; the RDF-star `EdgeAnnotation` is + // expanded during planning (RDF-star over // R2RML is undefined — left untouched here, not silently-empty in // the confirmed sense). Pattern::Filter(_) @@ -379,7 +378,6 @@ pub fn rewrite_patterns_for_r2rml( | Pattern::S2Search(_) | Pattern::Graph { .. } | Pattern::EdgeAnnotation { .. } - | Pattern::AnnotationTarget { .. } | Pattern::DefaultGraphSource { .. } => { result_patterns.push(pattern.clone()); origins.push(i); diff --git a/fluree-db-query/src/rewrite.rs b/fluree-db-query/src/rewrite.rs index daabcb4d01..f6aff8e582 100644 --- a/fluree-db-query/src/rewrite.rs +++ b/fluree-db-query/src/rewrite.rs @@ -266,7 +266,6 @@ fn rewrite_single_pattern( | Pattern::Graph { .. } | Pattern::Service(_) | Pattern::EdgeAnnotation { .. } - | Pattern::AnnotationTarget { .. } | Pattern::DefaultGraphSource { .. } => { rewrite_subpatterns(pattern.clone(), diag, |xs, diag| { rewrite_patterns_internal(&xs, hierarchy, ctx, diag, total_expansions) diff --git a/fluree-db-query/src/rewrite_owl_ql.rs b/fluree-db-query/src/rewrite_owl_ql.rs index c21abfefe3..59058667f6 100644 --- a/fluree-db-query/src/rewrite_owl_ql.rs +++ b/fluree-db-query/src/rewrite_owl_ql.rs @@ -552,7 +552,6 @@ fn rewrite_owl_ql_single_pattern( | Pattern::Graph { .. } | Pattern::Service(_) | Pattern::EdgeAnnotation { .. } - | Pattern::AnnotationTarget { .. } | Pattern::DefaultGraphSource { .. } => { rewrite_subpatterns(pattern.clone(), diag, |xs, diag| { rewrite_owl_ql_patterns_internal(&xs, ontology, ctx, diag, total_expansions) diff --git a/fluree-db-query/src/term_components.rs b/fluree-db-query/src/term_components.rs index 04d0bd14d8..efaeb0af8d 100644 --- a/fluree-db-query/src/term_components.rs +++ b/fluree-db-query/src/term_components.rs @@ -393,6 +393,26 @@ impl TermComponentsOperator { const OBJECT_SITE: &str = "term-object"; /// A component no dictionary can name is a miss, not an error. +/// Refuse to read links from an index built before them. Such an index holds +/// an annotated ledger's annotations only as `f:reifies*` bundles, so a link +/// read would answer as if they did not exist. +pub(crate) fn require_indexed_links(ctx: &ExecutionContext<'_>) -> Result<()> { + let pre_link = ctx.active_snapshot.has_annotations + && ctx + .binary_store + .as_ref() + .is_some_and(|store| !store.has_term_dict()); + if pre_link { + return Err(QueryError::UnsupportedFeature( + "this ledger's index predates RDF 1.2 triple-term links, so it cannot answer \ + reified-triple patterns over its annotations; rebuild the index \ + (`fluree reindex `)" + .to_string(), + )); + } + Ok(()) +} + fn missing(r: std::io::Result) -> std::io::Result> { match r { Ok(v) => Ok(Some(v)), @@ -517,6 +537,7 @@ impl Operator for TermComponentsOperator { } async fn open(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { + require_indexed_links(ctx)?; self.child.open(ctx).await?; self.store = ctx.binary_store.clone(); if let Some(store) = self.store.clone() { diff --git a/fluree-db-sparql/src/lower/annotation.rs b/fluree-db-sparql/src/lower/annotation.rs index f3855c8806..3b876fc489 100644 --- a/fluree-db-sparql/src/lower/annotation.rs +++ b/fluree-db-sparql/src/lower/annotation.rs @@ -3,19 +3,18 @@ //! Translates the AST shapes (`TriplePattern.annotation`, //! `GraphPattern::AnnotationTarget`, and reified-triple terms //! `SubjectTerm::QuotedTriple` / `Term::QuotedTriple`) into the -//! existing query IR (`Pattern::EdgeAnnotation` and -//! `Pattern::AnnotationTarget`). The IR's -//! `expand_edge_annotation_patterns` step (in `fluree-db-query`) -//! handles the f:reifies* fan-out from there. +//! query IR: annotation syntax to `Pattern::EdgeAnnotation`, whose +//! `f:reifies*` fan-out `expand_edge_annotation_patterns` (in +//! `fluree-db-query`) handles, and reified triples to the `rdf:reifies` +//! link (`fluree_db_query::ir::lower_reified_link`). //! //! Reified triples desugar exactly per SPARQL 1.2: a `<< s p o ~ r? >>` //! term denotes its reifier node `r` (fresh when unnamed) and adds the //! pattern `r rdf:reifies <<( s p o )>>` — emitted here as a sibling -//! `Pattern::AnnotationTarget`. Nested reified triples recurse. +//! link. Nested reified triples recurse. //! -//! Sibling triples about a reifier variable are NOT folded into -//! `body` — they sit in the surrounding scope and join via the -//! standard executor on the bound reifier var. See +//! Sibling triples about a reifier variable sit in the surrounding +//! scope and join via the standard executor on the bound reifier var. See //! `docs/concepts/edge-annotations.md` "SPARQL 1.2 / RDF 1.2 surface" //! for the rationale. @@ -25,7 +24,7 @@ use crate::span::SourceSpan; use fluree_db_core::DatatypeConstraint; use fluree_db_query::ir::triple::{Ref, Term as IrTerm, TriplePattern as IrTriplePattern}; -use fluree_db_query::ir::Pattern; +use fluree_db_query::ir::{lower_reified_link, Pattern}; use fluree_db_query::parse::encode::IriEncoder; use std::collections::HashMap; @@ -71,10 +70,9 @@ impl LoweringContext<'_, E> { /// Lower a `GraphPattern::AnnotationTarget` (the /// `?ann rdf:reifies <<( s p o )>>` form and the standalone - /// reified-triple statement it desugars from) into - /// `Pattern::AnnotationTarget` IR (plus any sibling targets from - /// nested reified triples inside the triple term). Emits an empty - /// body — surrounding sibling triples about the reifier join + /// reified-triple statement it desugars from) into the link (plus + /// sibling links from nested reified triples inside the triple + /// term). Surrounding sibling triples about the reifier join /// through the standard executor. pub(super) fn lower_annotation_target_pattern( &mut self, @@ -84,21 +82,13 @@ impl LoweringContext<'_, E> { let mut out = Vec::new(); let annotation_ref = self.lower_subject(reifier)?; let edge = self.lower_triple_term(triple_term, &mut out)?; - if link_terms_enabled() { - self.lower_reified_link(annotation_ref, edge, &mut out); - return Ok(out); - } - out.push(Pattern::AnnotationTarget { - annotation: annotation_ref, - edge, - body: Vec::new(), - }); + lower_reified_link(annotation_ref, edge, self.encoder, self.vars, &mut out); Ok(out) } /// Desugar an RDF 1.2 reified triple `<< s p o ~ r? >>` used as a - /// term: emit `r rdf:reifies <<( s p o )>>` (as - /// `Pattern::AnnotationTarget`) into `out` and return the reifier + /// term: emit the link `r rdf:reifies <<( s p o )>>` into `out` and + /// return the reifier /// ref that stands in the reified triple's position. Nested /// reified triples in `s`/`o` recurse; repeated occurrences (same /// source span) reuse the memoized reifier. @@ -122,19 +112,13 @@ impl LoweringContext<'_, E> { let p = self.lower_predicate(&qt.predicate)?; let (o, dtc) = self.lower_object_desugared(&qt.object, cache, out, true)?; - if link_terms_enabled() { - self.lower_reified_link( - annotation_ref.clone(), - IrTriplePattern { s, p, o, dtc }, - out, - ); - } else { - out.push(Pattern::AnnotationTarget { - annotation: annotation_ref.clone(), - edge: IrTriplePattern { s, p, o, dtc }, - body: Vec::new(), - }); - } + lower_reified_link( + annotation_ref.clone(), + IrTriplePattern { s, p, o, dtc }, + self.encoder, + self.vars, + out, + ); cache.insert(qt.span, annotation_ref.clone()); Ok(annotation_ref) } @@ -198,8 +182,7 @@ impl LoweringContext<'_, E> { /// Lower the `{| verb obj ; verb obj |}` body to a flat list of /// patterns whose subject is the reifier: `Pattern::Triple` for /// simple predicates, property-path patterns for path verbs - /// (`{| :r/:q 'x' |}`), plus sibling `Pattern::AnnotationTarget`s - /// for reified-triple objects. + /// (`{| :r/:q 'x' |}`), plus sibling links for reified-triple objects. /// /// Each entry's object is lowered through /// `lower_object_with_constraint` so literal objects pin the scan @@ -269,165 +252,3 @@ impl LoweringContext<'_, E> { Ok(IrTriplePattern { s, p, o, dtc }) } } - -/// `FLUREE_ANNOTATION_TERMS=1` routes reified-triple patterns through the -/// `rdf:reifies` link flake and the term dictionary instead of the -/// `f:reifies*` bundle chain. Read per lowering so tests can flip it. -pub(super) fn link_terms_enabled() -> bool { - std::env::var("FLUREE_ANNOTATION_TERMS").is_ok_and(|v| v == "1") -} - -impl LoweringContext<'_, E> { - /// Lower a reified-triple pattern to the link form: one - /// `annotation rdf:reifies ?__term` triple whose object is a triple-term - /// handle, and `TermComponents(?__term, s, p, o)` relating the term to its - /// components. A variable component is bound by that relation, which joins - /// on it when another pattern bound it first, in every scope, without the - /// lowering tracking who binds what; a bound subject lets the planner - /// drive the relation through the dictionary's subject prefix. A constant - /// component is also a filter on the link - /// (`FILTER(sameTerm(PREDICATE(?__term),

))`), which the planner turns - /// into the scan's handle interval and key check when the link leads. A - /// fully constant edge composes to a constant term the scan looks up - /// directly. - pub(super) fn lower_reified_link( - &mut self, - annotation_ref: Ref, - edge: IrTriplePattern, - out: &mut Vec, - ) { - use fluree_db_core::FlakeValue; - use fluree_db_query::binding::Binding; - use fluree_db_query::ir::{Component, Expression, Function, TermComponentsPattern}; - - let reifies = self.encoder.encode_ref(fluree_vocab::rdf::REIFIES); - - // Fully constant edge: compose the term itself. - if let Some(term) = constant_term(&edge) { - out.push(Pattern::Triple(IrTriplePattern { - s: annotation_ref, - p: reifies, - o: IrTerm::Value(FlakeValue::TripleTerm(Box::new(term))), - dtc: None, - })); - return; - } - - let name = format!("?__term_{}", self.term_counter); - self.term_counter += 1; - let t = self.vars.get_or_insert(&name); - out.push(Pattern::Triple(IrTriplePattern { - s: annotation_ref, - p: reifies, - o: IrTerm::Var(t), - dtc: None, - })); - - let accessor = |f: Function| Expression::call(f, vec![Expression::Var(t)]); - let same_term = |f: Function, constant: Expression| { - Pattern::Filter(Expression::call( - Function::SameTerm, - vec![accessor(f), constant], - )) - }; - let IrTriplePattern { s, p, o, dtc } = edge; - // Constant positions are filters on the link (the scan narrows on - // them) and constants of the components relation (a constant subject - // anchors it). Variable positions are bound by the components relation, - // which joins on them when another pattern bound them first. - let component = |func: Function, term: IrTerm, out: &mut Vec| match term { - IrTerm::Var(v) => Component::Var(v), - IrTerm::Sid(sid) => { - out.push(same_term( - func, - Expression::Const(FlakeValue::Ref(sid.clone())), - )); - Component::Node(sid) - } - IrTerm::Iri(iri) => match self.encoder.encode_iri(&iri) { - Some(sid) => { - out.push(same_term( - func, - Expression::Const(FlakeValue::Ref(sid.clone())), - )); - Component::Node(sid) - } - // An IRI in no registered namespace names nothing in this - // ledger, so the pattern cannot match. - None => { - out.push(Pattern::Filter(Expression::Const(FlakeValue::Boolean( - false, - )))); - Component::Any - } - }, - // A literal matches by term identity: its datatype or language - // tag is part of what `<< ?s :p "chat"@fr >>` asks for. - IrTerm::Value(v) => match &dtc { - Some(dtc) => { - out.push(same_term( - func, - Expression::Resolved(Box::new(Binding::Lit { - val: v.clone(), - dtc: dtc.clone(), - t: None, - op: None, - p_id: None, - })), - )); - Component::Literal(v, dtc.clone()) - } - None => { - out.push(Pattern::Filter(Expression::call( - Function::Eq, - vec![accessor(func), Expression::Const(v)], - ))); - Component::Any - } - }, - }; - let tc = TermComponentsPattern { - term: t, - subject: component(Function::TripleSubject, IrTerm::from(s), out), - predicate: component(Function::TriplePredicate, IrTerm::from(p), out), - object: component(Function::TripleObject, o, out), - }; - if tc.components().into_iter().any(|c| c.var().is_some()) - || matches!(tc.subject, Component::Node(_)) - { - out.push(Pattern::TermComponents(tc)); - } - } -} - -/// The materialized term for an edge whose three positions are constants -/// and whose object datatype is known; `None` otherwise. -fn constant_term(edge: &IrTriplePattern) -> Option { - use fluree_db_core::FlakeValue; - let s = match &edge.s { - Ref::Sid(s) => s.clone(), - _ => return None, - }; - let p = match &edge.p { - Ref::Sid(p) => p.clone(), - _ => return None, - }; - let (o, dt, lang) = match (&edge.o, &edge.dtc) { - (IrTerm::Sid(sid), _) => ( - FlakeValue::Ref(sid.clone()), - fluree_db_core::edge::id_datatype_sid(), - None, - ), - (IrTerm::Value(v), Some(DatatypeConstraint::Explicit(dt))) => (v.clone(), dt.clone(), None), - (IrTerm::Value(v), Some(DatatypeConstraint::LangTag(tag))) => ( - v.clone(), - fluree_db_core::Sid::new( - fluree_vocab::namespaces::RDF, - fluree_vocab::rdf_names::LANG_STRING, - ), - Some(tag.to_string()), - ), - _ => return None, - }; - Some(fluree_db_core::TripleTermValue { s, p, o, dt, lang }) -} diff --git a/fluree-db-sparql/src/lower/construct.rs b/fluree-db-sparql/src/lower/construct.rs index bcb6d035db..ee721a99c3 100644 --- a/fluree-db-sparql/src/lower/construct.rs +++ b/fluree-db-sparql/src/lower/construct.rs @@ -204,11 +204,6 @@ impl LoweringContext<'_, E> { edge, annotation, body, - } - | Pattern::AnnotationTarget { - edge, - annotation, - body, } => { let triple = out.push_pattern(edge.clone(), None); out.push_reification(triple, annotation.clone()); diff --git a/fluree-db-sparql/src/lower/mod.rs b/fluree-db-sparql/src/lower/mod.rs index da69ac3e1e..2c1581fd0f 100644 --- a/fluree-db-sparql/src/lower/mod.rs +++ b/fluree-db-sparql/src/lower/mod.rs @@ -238,10 +238,8 @@ pub fn resolve_dataset_clause(ast: &SparqlAst) -> Result Result<()> { use fluree_db_query::ir::triple::Ref; use fluree_vocab::reifies_iris; @@ -346,7 +344,7 @@ fn reject_direct_reifies_in_patterns(patterns: &[Pattern]) -> Result<()> { Pattern::Graph { patterns, .. } => walk(patterns)?, Pattern::Service(sp) => walk(&sp.patterns)?, Pattern::Subquery(sq) => walk(&sq.patterns)?, - Pattern::EdgeAnnotation { body, .. } | Pattern::AnnotationTarget { body, .. } => { + Pattern::EdgeAnnotation { body, .. } => { walk(body)?; } Pattern::DefaultGraphSource { patterns } => walk(patterns)?, @@ -407,9 +405,6 @@ struct LoweringContext<'a, E> { pp_counter: u32, /// Monotonic counter for generating expression-based ORDER BY bind variables (`?__order_by_0`, …). order_counter: u32, - /// Monotonic counter for the triple-term variables the link lowering - /// mints (`?__term_0`, …). - term_counter: u32, /// Original SPARQL source text (for extracting SERVICE body text). source_text: Option<&'a str>, } @@ -434,7 +429,6 @@ impl<'a, E: IriEncoder> LoweringContext<'a, E> { agg_counter: 0, pp_counter: 0, order_counter: 0, - term_counter: 0, source_text, } } @@ -4595,34 +4589,33 @@ mod tests { } } + /// The link `?ann rdf:reifies <<( s p o )>>` of a fully constant edge. + fn is_constant_link(p: &Pattern) -> bool { + matches!( + p, + Pattern::Triple(tp) + if matches!(&tp.p, Ref::Sid(sid) if fluree_db_core::is_rdf_reifies(sid)) + && matches!(&tp.o, Term::Value(fluree_db_core::FlakeValue::TripleTerm(_))) + ) + } + #[test] - fn m43_rdf_reifies_lowers_to_annotation_target_with_empty_body() { + fn m43_rdf_reifies_lowers_to_the_link() { let query = lower_query( "PREFIX rdf: PREFIX ex: SELECT * WHERE { ?ann rdf:reifies <<( ex:alice ex:worksFor ex:acme )>> . }", ) .unwrap(); - // The pattern list contains exactly one AnnotationTarget. Sibling - // triples about ?ann (none in this query) would join via the - // standard executor — the AnnotationTarget itself carries no body. - let n = query - .patterns - .iter() - .filter(|p| matches!(p, Pattern::AnnotationTarget { .. })) - .count(); - assert_eq!(n, 1, "expected exactly one AnnotationTarget IR pattern"); - for p in &query.patterns { - if let Pattern::AnnotationTarget { body, .. } = p { - assert!(body.is_empty(), "M4.3 emits empty body"); - } - } + // A constant edge composes to its term: one link triple, nothing else. + assert_eq!(query.patterns.len(), 1, "{:?}", query.patterns); + assert!(is_constant_link(&query.patterns[0]), "{:?}", query.patterns); } #[test] fn m43_sibling_triples_about_reifier_stay_in_outer_scope() { - // ?ann ex:role "Engineer" must remain a regular Pattern::Triple - // alongside the AnnotationTarget — NOT folded into body. + // ?ann ex:role "Engineer" stays a regular Pattern::Triple beside the + // link. let query = lower_query( "PREFIX rdf: PREFIX ex: @@ -4632,24 +4625,15 @@ mod tests { }", ) .unwrap(); - let n_target = query - .patterns - .iter() - .filter(|p| matches!(p, Pattern::AnnotationTarget { .. })) - .count(); - let n_triple = query - .patterns - .iter() - .filter(|p| matches!(p, Pattern::Triple(_))) - .count(); - assert_eq!(n_target, 1); - assert_eq!(n_triple, 1, "sibling stays as outer Pattern::Triple"); - // And the body of AnnotationTarget is empty. - for p in &query.patterns { - if let Pattern::AnnotationTarget { body, .. } = p { - assert!(body.is_empty()); - } - } + assert_eq!(query.patterns.len(), 2, "{:?}", query.patterns); + assert_eq!( + query + .patterns + .iter() + .filter(|p| is_constant_link(p)) + .count(), + 1 + ); } #[test] diff --git a/fluree-db-sparql/src/lower/pattern.rs b/fluree-db-sparql/src/lower/pattern.rs index 26258d8222..b37b258400 100644 --- a/fluree-db-sparql/src/lower/pattern.rs +++ b/fluree-db-sparql/src/lower/pattern.rs @@ -130,8 +130,8 @@ impl LoweringContext<'_, E> { } => self.lower_property_path(subject, path, object, *span), // RDF 1.2 reifier-rooted annotation pattern. - // `?ann rdf:reifies <<( s p o )>>` → - // `Pattern::AnnotationTarget` IR with empty body. Sibling + // `?ann rdf:reifies <<( s p o )>>` → the link and its term + // components (`lower_reified_link`). Sibling // triples about the reifier in surrounding scope join on // the bound annotation var via the standard executor. // diff --git a/fluree-db-sparql/src/lower/rdf_star.rs b/fluree-db-sparql/src/lower/rdf_star.rs index ef9945838a..6ff484b5ab 100644 --- a/fluree-db-sparql/src/lower/rdf_star.rs +++ b/fluree-db-sparql/src/lower/rdf_star.rs @@ -7,8 +7,7 @@ //! triple plus BIND expressions extracting the metadata. //! 2. **RDF 1.2 reified triple** — everything else. The term denotes //! its reifier node; the desugaring in `lower/annotation.rs` emits -//! `r rdf:reifies <<( s p o )>>` as `Pattern::AnnotationTarget` and -//! substitutes the reifier ref into the enclosing position. +//! the link `r rdf:reifies <<( s p o )>>` and substitutes the reifier ref into the enclosing position. use crate::ast::term::{ObjectTerm, PredicateTerm, SubjectTerm, Term as SparqlTerm}; use crate::ast::TriplePattern as SparqlTriplePattern; @@ -48,7 +47,7 @@ impl LoweringContext<'_, E> { /// /// Becomes (conceptually): /// ```text - /// ?r rdf:reifies <<( ex:a ex:b ex:c )>> . (AnnotationTarget) + /// ?r rdf:reifies <<( ex:a ex:b ex:c )>> . (the link) /// ?doc ex:cites ?r . (triple pattern) /// ``` pub(super) fn lower_bgp_with_rdf_star( @@ -86,7 +85,7 @@ impl LoweringContext<'_, E> { } // RDF 1.2 reading: reified-triple terms denote their - // reifier node (desugared as sibling AnnotationTargets). + // reifier node (desugared as sibling links). let s = match &tp.subject { SubjectTerm::QuotedTriple(qt) => { self.lower_reified_triple(qt, &mut reified_cache, &mut result)? From ef5ec8d309f9b73c77b9cc56a350cab5caccc504 Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 17:49:25 -0400 Subject: [PATCH 43/92] feat(query): triple-term functions in JSON-LD queries JSON-LD expressions gain the names of SPARQL's triple-term functions: `triple`, `subject`, `predicate`, `object` and `isTriple`, in `bind` and `filter` alike. `triple` joins the operators whose unquoted operands are RDF terms: an unquoted `ex:alice` elsewhere is demoted to the string it always was, so `(triple ex:alice ex:knows ex:bob)` built nothing from three strings. --- docs/concepts/edge-annotations.md | 2 +- docs/query/jsonld-query.md | 31 +++++++++-- fluree-db-api/tests/it_triple_term_links.rs | 58 +++++++++++++++++++++ fluree-db-query/src/parse/filter_sexpr.rs | 17 +++--- fluree-db-query/src/parse/lower.rs | 6 +++ 5 files changed, 102 insertions(+), 12 deletions(-) diff --git a/docs/concepts/edge-annotations.md b/docs/concepts/edge-annotations.md index e38c92e14f..d8a792a346 100644 --- a/docs/concepts/edge-annotations.md +++ b/docs/concepts/edge-annotations.md @@ -566,7 +566,7 @@ Today's surface covers the common LPG / RDF-star use cases. The following are no - **Annotations on list-occurrence triples.** `@list` membership is in scope as a future extension; the on-disk format already reserves space for it. Today, annotating a list element is rejected at parse time. - **Reifiers for unasserted triples.** `@reifies` must point at an asserted edge. Pure-proposition reification (claims about triples that are not in the graph) is deferred. - **Reifiers for multiple triples.** One annotation subject corresponds to one edge. Reifying several unrelated triples from a single annotation isn't allowed. -- **Triple terms as stored values.** `ex:doc ex:mentions <<( ex:s ex:p ex:o )>>` is not a representable value on any write surface (JSON-LD, SPARQL, Turtle): the only accepted position for `<<( ... )>>` is the object of `rdf:reifies`, where it names an edge rather than storing a triple. Use a separate annotation subject. A SPARQL query can still build and return one: `TRIPLE`, `SUBJECT`, `PREDICATE`, `OBJECT`, `isTRIPLE` and `BIND(<<( ?s ?p ?o )>> AS ?t)` work on terms, and a bound term renders in every result format (see [Output formats](../query/output-formats.md#triple-terms)). JSON-LD queries have no syntax for these functions yet. +- **Triple terms as stored values.** `ex:doc ex:mentions <<( ex:s ex:p ex:o )>>` is not a representable value on any write surface (JSON-LD, SPARQL, Turtle): the only accepted position for `<<( ... )>>` is the object of `rdf:reifies`, where it names an edge rather than storing a triple. Use a separate annotation subject. A query can still build and return one: `TRIPLE`, `SUBJECT`, `PREDICATE`, `OBJECT`, `isTRIPLE` and `BIND(<<( ?s ?p ?o )>> AS ?t)` work on terms, and a bound term renders in every result format (see [Output formats](../query/output-formats.md#triple-terms)). JSON-LD queries name them `triple`, `subject`, `predicate`, `object` and `isTriple` (see [JSON-LD query](../query/jsonld-query.md#triple-term-functions)). - **SPARQL UPDATE annotations inside named graphs.** Annotation tails under `GRAPH { }` / `WITH ` in SPARQL UPDATE are rejected; write named-graph annotations with JSON-LD `@annotation` or TriG-star. - **Unasserted reified triples.** RDF 1.2's `<< s p o >>` and `r rdf:reifies <<( s p o )>>` do not assert `s p o`; Fluree's do (the reifier is lifecycle-coupled to a live edge). A W3C test that depends on a reifier existing for a triple that is *not* in the graph therefore diverges. diff --git a/docs/query/jsonld-query.md b/docs/query/jsonld-query.md index 1980306679..f9bdaeb1d6 100644 --- a/docs/query/jsonld-query.md +++ b/docs/query/jsonld-query.md @@ -601,9 +601,9 @@ Apply conditions to filter results: **Comparing against IRIs:** -An unquoted prefixed name or `<...>` IRI is an IRI operand wherever RDF terms -are compared — `=`, `!=`, `in`, `not-in`, `sameTerm` — and compares by -identity, so `(= ?p ex:knows)` matches the predicate `ex:knows` — never the +An unquoted prefixed name or `<...>` IRI is an IRI operand wherever an +operator reads RDF terms — `=`, `!=`, `in`, `not-in`, `sameTerm`, and `triple` +— and compares by identity, so `(= ?p ex:knows)` matches the predicate `ex:knows` — never the string `"ex:knows"`. In every other position it is the string it has always been: @@ -1268,6 +1268,31 @@ Function names are case-insensitive. See [Vector Search](../indexing-and-search/ - `(isIRI ?x)` - Is an IRI - `(isBlank ?x)` - Is a blank node - `(isLiteral ?x)` - Is a literal +- `(isTriple ?x)` - Is a triple term + +### Triple-Term Functions + +The RDF 1.2 functions over triple terms, the JSON-LD names for SPARQL's +`TRIPLE`, `SUBJECT`, `PREDICATE` and `OBJECT`. A reifier's `rdf:reifies` value +is a triple term: + +- `(triple ex:alice ex:knows ex:bob)` - The triple term `<<( ex:alice ex:knows ex:bob )>>` +- `(subject ?t)`, `(predicate ?t)`, `(object ?t)` - A triple term's components + +```json +{ + "@context": { + "ex": "http://example.org/", + "rdf": "http://www.w3.org/1999/02/22-rdf-syntax-ns#" + }, + "select": ["?r", "?s", "?o"], + "where": [ + { "@id": "?r", "rdf:reifies": "?t" }, + ["filter", "(sameTerm (predicate ?t) ex:knows)"], + ["bind", "?s", "(subject ?t)", "?o", "(object ?t)"] + ] +} +``` ## Query Modifiers diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 3bf6b0668d..138011a5a1 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2306,3 +2306,61 @@ async fn link_reads_refuse_an_index_built_before_links() { .expect("a rebuilt index answers"); assert_eq!(rows(&result), strings(&[&["ex:claim1"], &["ex:claim2"]])); } + +/// The JSON-LD twin of `triple_constructs_the_terms_links_hold`: the +/// triple-term functions under their JSON-LD names, in `bind` and `filter`. +#[tokio::test] +async fn jsonld_triple_term_functions() { + let (fluree, ledger) = + import(&[("claims.ttl", CLAIMS)], "it/triple-term-links:jsonld-fns").await; + let ctx = json!({ + "ex": "http://example.org/", + "rdf": "http://www.w3.org/1999/02/22-rdf-syntax-ns#" + }); + let run = |query: JsonValue| { + let fluree = &fluree; + let ledger = &ledger; + async move { + let result = support::query_jsonld_formatted(fluree, ledger, &query) + .await + .unwrap_or_else(|e| panic!("{query}: {e:?}")); + rows(&result) + } + }; + + let built = run(json!({ + "@context": ctx, + "select": ["?r"], + "where": [ + ["bind", "?t", "(triple ex:alice ex:knows ex:bob)"], + {"@id": "?r", "rdf:reifies": "?t"} + ] + })) + .await; + assert_eq!(built, strings(&[&["ex:claim1"]])); + + let decomposed = run(json!({ + "@context": ctx, + "select": ["?s", "?p", "?o"], + "where": [ + {"@id": "ex:claim1", "rdf:reifies": "?t"}, + ["bind", "?s", "(subject ?t)"], + ["bind", "?p", "(predicate ?t)"], + ["bind", "?o", "(object ?t)"] + ] + })) + .await; + assert_eq!(decomposed, strings(&[&["ex:alice", "ex:knows", "ex:bob"]])); + + let filtered = run(json!({ + "@context": ctx, + "select": ["?r"], + "where": [ + {"@id": "?r", "rdf:reifies": "?t"}, + ["filter", "(isTriple ?t)"], + ["filter", "(sameTerm (predicate ?t) ex:age)"] + ] + })) + .await; + assert_eq!(filtered, strings(&[&["ex:claim2"]])); +} diff --git a/fluree-db-query/src/parse/filter_sexpr.rs b/fluree-db-query/src/parse/filter_sexpr.rs index 9e85b04964..4e393e7c7b 100644 --- a/fluree-db-query/src/parse/filter_sexpr.rs +++ b/fluree-db-query/src/parse/filter_sexpr.rs @@ -87,10 +87,10 @@ pub fn parse_s_expression(s: &str) -> Result { }); } - // Parse arguments, then keep the IRI classification only where an RDF - // term comparison is actually being made (see `compares_rdf_terms`). + // Parse arguments, then keep the IRI classification only where the + // operator reads RDF terms (see `takes_rdf_terms`). let mut args = parse_s_expression_args(rest)?; - if !compares_rdf_terms(&op_lower) { + if !takes_rdf_terms(&op_lower) { demote_iri_atoms(&mut args); } let args = args; @@ -271,8 +271,9 @@ fn classify_unquoted_atom(s: &str) -> UnresolvedExpression { } } -/// Operators whose operands are compared as RDF *terms*, and so are the only -/// places an unquoted atom may mean an IRI. +/// Operators whose operands are RDF *terms* (compared as terms, or, for +/// `triple`, built into one), and so are the only places an unquoted atom may +/// mean an IRI. /// /// Everywhere else an unquoted atom is the string it has always been. Without /// this restriction the classifier reaches every argument position, which is @@ -283,10 +284,10 @@ fn classify_unquoted_atom(s: &str) -> UnresolvedExpression { /// to write a literal. Identity comparison is the one position where an IRI /// operand is both meaningful and what the author must have meant — a string /// could never have matched an IRI-valued variable there. -fn compares_rdf_terms(op_lower: &str) -> bool { +fn takes_rdf_terms(op_lower: &str) -> bool { matches!( op_lower, - "=" | "eq" | "!=" | "<>" | "ne" | "in" | "not-in" | "notin" | "sameterm" + "=" | "eq" | "!=" | "<>" | "ne" | "in" | "not-in" | "notin" | "sameterm" | "triple" ) } @@ -510,7 +511,7 @@ fn list_tokens_to_expr(items: &[SexprToken]) -> Result { let parsed: Result> = arg_tokens.iter().map(expr_from_sexpr_token).collect(); let mut args = parsed?; - if !compares_rdf_terms(&op_lower) { + if !takes_rdf_terms(&op_lower) { demote_iri_atoms(&mut args); } let args = args; diff --git a/fluree-db-query/src/parse/lower.rs b/fluree-db-query/src/parse/lower.rs index 78971c634c..48bb64dda4 100644 --- a/fluree-db-query/src/parse/lower.rs +++ b/fluree-db-query/src/parse/lower.rs @@ -1816,6 +1816,12 @@ fn lower_function_name(name: &str) -> Function { "datatype" => Function::Datatype { strict: false }, "langmatches" => Function::LangMatches, "sameterm" => Function::SameTerm, + // RDF 1.2 triple terms (SPARQL's TRIPLE, SUBJECT, PREDICATE, OBJECT, isTRIPLE) + "triple" => Function::Triple, + "subject" => Function::TripleSubject, + "predicate" => Function::TriplePredicate, + "object" => Function::TripleObject, + "istriple" | "is-triple" => Function::IsTriple, // Fluree-specific: transaction time "t" => Function::T, "op" => Function::Op, From b5503424a0d8a8cc27db354764ae6a5503366bea Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 19:38:31 -0400 Subject: [PATCH 44/92] fix(import): annotations in TriG GRAPH blocks get their links Bulk import writes a TriG `GRAPH` block through its own loop, which spooled each reifier's f:reifies* bundle but not the rdf:reifies link the default-graph sink writes beside it. An imported ledger whose annotations were all in named graphs got no term dictionary, so its link reads were refused as an index built before links, and a mixed one silently missed the named graphs' annotations. The loop now spools the link through the sink's own link writer, under the block's graph. --- fluree-db-api/tests/it_triple_term_links.rs | 23 +++++++++++++++++++++ fluree-db-transact/src/import.rs | 14 +++++++++++-- fluree-db-transact/src/import_sink.rs | 17 +++++++++++++++ 3 files changed, 52 insertions(+), 2 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 138011a5a1..014de4a45d 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2364,3 +2364,26 @@ async fn jsonld_triple_term_functions() { .await; assert_eq!(filtered, strings(&[&["ex:claim2"]])); } + +/// TriG import writes an annotation inside a `GRAPH` block through its own +/// named-graph loop; that loop spools the link as the default-graph sink does. +#[tokio::test] +async fn trig_import_links_named_graph_annotations() { + const TRIG: &str = r#"@prefix ex: . +ex:alice ex:name "Alice" . +GRAPH { + ex:event1 ex:actor ex:alice {| ex:confidence "high" |} . + ex:event2 ex:actor ex:alice ~ ex:claim2 {| ex:confidence "low"@en |} . +} +"#; + let (fluree, ledger) = import(&[("data.trig", TRIG)], "it/triple-term-links:trig").await; + let got = run_link_query( + &fluree, + &ledger, + "SELECT ?s WHERE { GRAPH \ + { ?r rdf:reifies ?t BIND(SUBJECT(?t) AS ?s) } } ORDER BY ?s" + .to_string(), + ) + .await; + assert_eq!(got, strings(&[&["ex:event1"], &["ex:event2"]])); +} diff --git a/fluree-db-transact/src/import.rs b/fluree-db-transact/src/import.rs index 902c49eb43..23bc4f1e6d 100644 --- a/fluree-db-transact/src/import.rs +++ b/fluree-db-transact/src/import.rs @@ -704,13 +704,23 @@ mod inner { }; let bundle = crate::generate::flakes::reified_triple_bundle( Some(graph_sid.clone()), - s, - p, + s.clone(), + p.clone(), o, &dtc, &ann, new_t, )?; + // The RDF 1.2 link rides alongside the bundle, as on the + // default-graph path. + if let (Some(sc), Some(object)) = ( + spool_ctx.as_mut(), + bundle + .iter() + .find(|f| fluree_db_core::is_reifies_object(&f.p)), + ) { + sc.push_named_graph_link(g_id, &s, &p, object, new_t)?; + } for flake in bundle { if let Some(sc) = spool_ctx.as_mut() { sc.push_named_graph_record( diff --git a/fluree-db-transact/src/import_sink.rs b/fluree-db-transact/src/import_sink.rs index 49e73ac713..3d28f5a56c 100644 --- a/fluree-db-transact/src/import_sink.rs +++ b/fluree-db-transact/src/import_sink.rs @@ -696,6 +696,23 @@ mod inner { self.g_id = saved; result } + + /// Spool the `rdf:reifies` link of a reified edge in an explicit named + /// graph (`g_id`); see [`Self::write_link_record`]. + pub fn push_named_graph_link( + &mut self, + g_id: GraphId, + s: &Sid, + p: &Sid, + object: &Flake, + t: i64, + ) -> Result<(), CommitCodecError> { + let saved = self.g_id; + self.g_id = g_id; + let result = self.write_link_record(s, p, object, t); + self.g_id = saved; + result + } } // ----------------------------------------------------------------------- From 731dad91eadfcdfb9e691ee61995f594498177df Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 19:38:37 -0400 Subject: [PATCH 45/92] fix(query): a novelty-assigned id unifies with the term it names bind_unifies compares two bindings by their store form, so an id novelty assigned (a subject no index holds yet) never matched the same subject bound by name: a reifier re-pointed, over an index, at an edge to a new subject joined nothing once the edge's scan drove the term components. When the store forms differ and either side is encoded, the two now compare by what they decode to. --- fluree-db-api/tests/it_triple_term_links.rs | 37 +++++++++++++++++++++ fluree-db-query/src/object_binding.rs | 15 ++++++++- 2 files changed, 51 insertions(+), 1 deletion(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 014de4a45d..88c259c11d 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2387,3 +2387,40 @@ GRAPH { .await; assert_eq!(got, strings(&[&["ex:event1"], &["ex:event2"]])); } + +/// A reifier re-pointed, over an index, at an edge to a subject only novelty +/// holds: the term's object is that subject's provisional id while the edge's +/// scan binds the subject by name, and the two must still join. +#[tokio::test] +async fn links_join_a_subject_only_novelty_holds() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/triple-term-links:novelty-subject"; + let ttl = |o: &str| { + format!( + "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:alice ex:knows ex:{o} ~ ex:claim1 {{| ex:confidence 0.9 |}} .\n" + ) + }; + let ledger = support::genesis_ledger(&fluree, ledger_id); + fluree + .upsert_turtle(ledger, &ttl("bob")) + .await + .expect("bob"); + support::rebuild_and_publish_index(&fluree, ledger_id).await; + let ledger = fluree.ledger(ledger_id).await.expect("load"); + fluree + .upsert_turtle(ledger, &ttl("carol")) + .await + .expect("re-point to carol"); + let ledger = fluree.ledger(ledger_id).await.expect("load"); + for body in [ + "SELECT ?o ?r WHERE { ex:alice ex:knows ?o . ?r rdf:reifies <<( ex:alice ex:knows ?o )>> }", + ] { + assert_eq!( + run_link_query(&fluree, &ledger, body.to_string()).await, + strings(&[&["ex:carol", "ex:claim1"]]), + "{body}" + ); + } +} + diff --git a/fluree-db-query/src/object_binding.rs b/fluree-db-query/src/object_binding.rs index 3e63fc7cf5..d4e6060969 100644 --- a/fluree-db-query/src/object_binding.rs +++ b/fluree-db-query/src/object_binding.rs @@ -494,8 +494,21 @@ pub(crate) fn bind_unifies( let gv = (is_numbig_encoded(a) || is_numbig_encoded(b)) .then(|| ctx.graph_view()) .flatten(); - normalize_for_key_cow(a, Some(store), gv.as_ref()).as_ref() + if normalize_for_key_cow(a, Some(store), gv.as_ref()).as_ref() == normalize_for_key_cow(b, Some(store), gv.as_ref()).as_ref() + { + return true; + } + // An id novelty assigned (a subject or string no index holds yet) has no + // store form to normalize to, so compare what the two name. + if !(a.is_encoded() || b.is_encoded()) { + return false; + } + let Some(view) = ctx.graph_view() else { + return false; + }; + crate::group_aggregate::materialize_encoded(a, Some(&view)) + == crate::group_aggregate::materialize_encoded(b, Some(&view)) } /// True if this is an arena-backed (NUM_BIG) encoded literal. From 891abee6937dd3d3c88ef39df5b4602911339996 Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 19:39:48 -0400 Subject: [PATCH 46/92] feat(query): annotation syntax reads the rdf:reifies link `s p o ~ ?r {| ... |}`, JSON-LD `@annotation` and Cypher relationship properties still expanded to the f:reifies* bundle chain; reified-triple patterns already read the link. EdgeAnnotation now expands at planning to the body, the reifier's link with its term components, and the base edge: - The term variable is reserved where the pattern is built, because planning cannot mint one; `link_patterns` is the link lowering's core, taking that variable and an IRI encoder. An IRI planning cannot encode (no ledger at hand) is compared as the query runs, and the runner encodes annotation edges up front so a constant edge still narrows the link scan. - The chain is ordered for the planner's tie-break: body and link are probes by reifier, the components decode from the bound term, and the base edge last checks existence on three bound positions. Base edge first ran the body as a probe per edge, so a threshold on it no longer cut scan work (the annotation filter-pushdown test caught it). - The DefaultGraphSource wrapper is emitted only under a default-graph union (`PlanningContext::default_graph_union`, set with the existing multi-graph flag but in every mode), where it keeps the base edge and the link in one member; otherwise the chain joins its enclosing block. Tests that pinned the bundle chain's shape now pin the link's: the expansion unit tests check the term object's datatype or tag, the explain test checks the expanded triples, and the filter-pushdown test keeps its correctness, fast-path agreement and scan-work assertions without the lane pins the chain needed. The bundle-chain read lanes are no longer reached by queries. --- docs/design/edge-annotations.md | 4 +- fluree-db-api/src/explain.rs | 25 +- .../tests/it_annotation_filter_pushdown.rs | 267 +++--------------- .../tests/it_edge_annotations_indexed.rs | 122 ++------ fluree-db-api/tests/it_triple_term_links.rs | 46 +-- fluree-db-cypher/src/lower/pattern.rs | 5 + fluree-db-query/src/execute.rs | 2 +- fluree-db-query/src/execute/runner.rs | 39 +++ fluree-db-query/src/execute/where_plan.rs | 237 +++++++--------- fluree-db-query/src/ir/pattern.rs | 14 +- fluree-db-query/src/ir/term_components.rs | 41 ++- fluree-db-query/src/lib.rs | 5 +- fluree-db-query/src/parse/lower.rs | 1 + fluree-db-query/src/planner.rs | 2 + fluree-db-query/src/temporal_mode.rs | 8 + fluree-db-sparql/src/lower/annotation.rs | 1 + fluree-db-sparql/src/lower/construct.rs | 1 + 17 files changed, 311 insertions(+), 509 deletions(-) diff --git a/docs/design/edge-annotations.md b/docs/design/edge-annotations.md index f8375a4504..640b0c8ada 100644 --- a/docs/design/edge-annotations.md +++ b/docs/design/edge-annotations.md @@ -166,7 +166,9 @@ Annotation forward and reverse branch CIDs are returned by `IndexRoot::all_cas_i ## Reified-triple patterns read the link -Reified-triple patterns (`<< s p o ~ ?r >>`, `?r rdf:reifies <<( s p o )>>`, JSON-LD `@reifies`) do not use the chain below. `lower_reified_link` (`fluree-db-query/src/ir/term_components.rs`), shared by the SPARQL and JSON-LD lowerings, emits the reifier's `rdf:reifies` link, `?r rdf:reifies ?t`, whose object is a triple-term handle the index's term dictionary resolves, plus `TermComponents(?t, s, p, o)` relating the term to its components and a `sameTerm` filter per constant component, which the planner turns into the scan's handle interval. A fully constant edge composes to a constant term instead. +Every annotation pattern reads the link. Reified-triple patterns (`<< s p o ~ ?r >>`, `?r rdf:reifies <<( s p o )>>`, JSON-LD `@reifies`) lower to it directly: `lower_reified_link` (`fluree-db-query/src/ir/term_components.rs`), shared by the SPARQL and JSON-LD lowerings, emits the reifier's `rdf:reifies` link, `?r rdf:reifies ?t`, whose object is a triple-term handle the index's term dictionary resolves, plus `TermComponents(?t, s, p, o)` relating the term to its components and a `sameTerm` filter per constant component, which the planner turns into the scan's handle interval. A fully constant edge composes to a constant term instead. + +Annotation syntax (`s p o ~ ?r {| ... |}`, JSON-LD `@annotation`, Cypher relationship properties) keeps its `Pattern::EdgeAnnotation` through lowering, with the term variable reserved there, and `expand_edge_annotation_patterns_for` (`where_plan.rs`) expands it at planning into the body, the link with its term components (`link_patterns`), and the base edge, which asserts that the edge exists and is visible. The order is the planner's tie-break: the body and the link are probes by reifier, the components decode from the bound term, and the base edge last is an existence check on three bound positions, so a range filter on the body drives the scan when nothing is more selective. The chain joins its enclosing block, wrapped in `Pattern::DefaultGraphSource` only when the default graph is a union of two or more graphs (`PlanningContext::default_graph_union`), where the wrapper reads one member at a time so the base edge and the link come from the same graph. The runner encodes the edge's IRIs first, since planning has no ledger to encode against; an IRI still unencoded is compared as the query runs. The link names its triple without joining the base edge, so visibility is checked on the term: `QueryPolicyEnforcer` lets a flake whose object is a triple term through only when the triple that term names would be visible, recursively for nested terms. A policy hiding `ex:worksFor` therefore hides the links to `ex:worksFor` edges on every route, not only on annotation-syntax patterns. diff --git a/fluree-db-api/src/explain.rs b/fluree-db-api/src/explain.rs index 2bdb625e6b..e6707dbb5f 100644 --- a/fluree-db-api/src/explain.rs +++ b/fluree-db-api/src/explain.rs @@ -8,7 +8,7 @@ use crate::format::iri::IriCompactor; use crate::query::helpers::{parse_jsonld_query, parse_sparql_to_ir}; use fluree_db_core::{is_rdf_type, StatsView}; use fluree_db_query::{ - expand_edge_annotation_patterns, explain_execution_hints, parse_query, ExplainPlan, + expand_edge_annotation_patterns_for, explain_execution_hints, parse_query, ExplainPlan, OptimizationStatus, Pattern, Query, Ref, Term, TriplePattern, VarId, VarRegistry, }; use serde_json::{json, Map, Value as JsonValue}; @@ -516,12 +516,11 @@ fn explain_from_parsed( Pattern::Subquery(sq) => { collect_triples_in_order(out, &sq.patterns, normalize_ref, normalize_term); } - // The `DefaultGraphSource` wrapper is introduced by - // `expand_edge_annotation_patterns` to keep `f:reifies*` - // lookups per-source-correlated in multi-graph default - // contexts. For explain purposes the triples inside it - // are the ones the optimizer sees, so recurse like - // `Graph` does. + // The `DefaultGraphSource` wrapper is introduced by the + // edge-annotation expansion to keep a chain per-source- + // correlated under a default-graph union. For explain + // purposes the triples inside it are the ones the optimizer + // sees, so recurse like `Graph` does. Pattern::DefaultGraphSource { patterns, .. } => { collect_triples_in_order(out, patterns, normalize_ref, normalize_term); } @@ -532,12 +531,12 @@ fn explain_from_parsed( } } - // Expand edge-annotation IR into the same triple chain the - // executor uses (`Pattern::EdgeAnnotation` → base edge + 3 - // `f:reifies*` lookups + body). Without this, edge-annotation queries appear as empty - // in `/explain` output because `collect_triples_in_order` doesn't - // descend into those container patterns. - let expanded_patterns = expand_edge_annotation_patterns(&parsed.patterns); + // Expand edge-annotation IR into the same chain the executor uses + // (`Pattern::EdgeAnnotation` → body + the reifier's link + base edge). + // Without this, edge-annotation queries appear as empty in `/explain` + // output because `collect_triples_in_order` doesn't descend into those + // container patterns. A single ledger's default graph is one graph. + let expanded_patterns = expand_edge_annotation_patterns_for(&parsed.patterns, false); let mut triples_in_order = Vec::new(); collect_triples_in_order( diff --git a/fluree-db-api/tests/it_annotation_filter_pushdown.rs b/fluree-db-api/tests/it_annotation_filter_pushdown.rs index 7bda830a93..037d510cea 100644 --- a/fluree-db-api/tests/it_annotation_filter_pushdown.rs +++ b/fluree-db-api/tests/it_annotation_filter_pushdown.rs @@ -1,47 +1,21 @@ //! A threshold on an annotation body must reduce scan work. //! -//! `expand_edge_annotation_patterns` wraps the expanded chain in -//! `Pattern::DefaultGraphSource`, and `collect_inner_join_block` breaks on that -//! wrapper — it collects `Filter` alongside `Triple`, `Values` and `Bind`, and -//! stops only on its `_` arm. A `FILTER` written beside the annotation -//! therefore started a block with no triples in it, -//! `extract_bounds_from_filters` returned early on an empty `object_vars`, and -//! the threshold ran as a `FilterOperator` -//! *above* the wrapper — after every annotation had been read and -//! materialized. Filtering by confidence cost exactly what not filtering cost: -//! on a 60k-edge ledger, `?c > 0.7` (18,317 rows) and `?c > 0.97` (1,148 rows) -//! both burned 321.03 fuel, the same as no filter at all. +//! `expand_edge_annotation_patterns` wraps an annotated edge (the base triple, +//! the reifier's `rdf:reifies` link and its term components) in +//! `Pattern::DefaultGraphSource`, and `collect_inner_join_block` breaks on +//! that wrapper. A `FILTER` written beside the annotation therefore started a +//! block with no triples in it, `extract_bounds_from_filters` returned early, +//! and the threshold ran *above* the wrapper — after every annotation had been +//! read. On a 60k-edge ledger `?c > 0.7` and `?c > 0.97` both burned the same +//! fuel as no filter at all. The expansion now copies such a filter into the +//! wrapper (pinned structurally by the `sink_*` unit tests in `where_plan.rs`). //! -//! Two stamps, because a test that only checked row counts would pass against -//! the bug it is meant to pin — the bug never changed an answer. -//! -//! 1. **Structural**: `--explain` must report `body-filters: 1` on the -//! `DefaultGraphSourceOperator`, i.e. the threshold is *inside* the chain -//! where the block builder can see it. Before the rewrite this is 0. -//! 2. **Effect**: with the lane pinned to `chain`, a tighter threshold must -//! burn strictly less fuel than no threshold. Fuel charges rows the scan -//! emits and never charges rows an encoded pre-filter drops inside the -//! cursor (`binary_scan.rs`), so it measures the thing under test, is -//! bit-identical across runs, and cannot be moved by load on the build box. -//! Before the rewrite these were equal to the last decimal place. -//! -//! The lane is pinned deliberately. The `arena` lane the planner currently -//! prefers drives from the base edge and probes per row, and does not exploit -//! object bounds on the resulting bound-subject body lookup — a separate -//! defect, tracked with the lane-selection issue. Pinning `chain` runs the -//! generic path, which plans the whole chain as one block — that is where the -//! bounds bite. Note the naming: `ChainLane::Chain` means "neither arena nor -//! enumerate" and the engine reports the execution as `generic`, so seeing -//! `lane="generic"` after pinning `chain` is expected and not a demotion. -//! Pinning is what keeps this test from passing by silently taking a lane that -//! never had the problem — and the fired-lane assertion is what keeps the -//! pinning itself honest. -//! -//! Which check guards what, since three of them look redundant and are not: -//! stamp 1 pins that the threshold is planned inside the chain; stamp 2 -//! (`pinned_kept_fuel < pinned_all_fuel`) catches a dropped override, because a -//! demoted arm cannot show the threshold reducing scan work; and the fired-lane -//! assertion in the loop is the only one that certifies *which* lane ran. +//! The effect is pinned here: a tighter threshold must burn strictly less +//! fuel. Fuel charges rows the scan emits and never rows an encoded pre-filter +//! drops inside the cursor (`binary_scan.rs`), so it measures the thing under +//! test, is bit-identical across runs, and cannot be moved by load. A test that +//! only checked row counts would pass against the bug: it never changed an +//! answer. //! //! Twin surfaces: SPARQL and JSON-LD share the IR, and the rewrite lives in //! `fluree-db-query`, so both are covered here. @@ -106,12 +80,10 @@ fn annotated_count_sparql(filter: Option) -> String { ) } -/// The same COUNT, with the base **object** variable also in the filter. This -/// is the shape that couples the sink to `elide_redundant_chain`: `?o` is not -/// projected and not read outside the wrapper, so elision is free to drop the -/// `f:reifiesObject` lookup — and `?o` survives only because `collect_var_stats` -/// walks `Pattern::Filter` and puts it back in the referenced set. If that walk -/// ever goes away, `?o` unbinds inside the chain and every row drops silently. +/// The same COUNT, with the base **object** variable also in the filter: `?o` +/// is not projected and not read outside the wrapper, and survives only because +/// `collect_var_stats` walks `Pattern::Filter` and puts it back in the +/// referenced set. fn annotated_count_with_object_var_sparql() -> String { format!( "PREFIX ex: @@ -202,24 +174,19 @@ async fn annotation_body_threshold_reduces_scan_work_on_both_surfaces() { .ledger(ledger_id) .await .expect("reload after reindex"); - assert!( - post.snapshot.annotation_index.is_some(), - "fixture must have a sealed arena, or this exercises a different lane" - ); - // ---- correctness first ------------------------------------- - let (all, _) = + let (all, all_fuel) = sparql_count_and_fuel(&fluree, &post, &annotated_count_sparql(None)).await; - let (kept, _) = + let (kept, kept_fuel) = sparql_count_and_fuel(&fluree, &post, &annotated_count_sparql(Some(THRESHOLD))) .await; assert_eq!(all as usize, EDGES, "unfiltered count"); assert_eq!(kept as usize, KEPT, "thresholded count"); - // The same answer with every fast path disabled. This is a weak - // oracle on its own (a defect in the plan both lanes share is - // invisible to it), which is why the fuel assertion below is the - // one that pins the behaviour. + // The same answer with every fast path disabled. A weak oracle on + // its own (a defect in the plan both lanes share is invisible to + // it), which is why the fuel assertion below is the one that pins + // the behaviour. let generic = { let _g = DisableFastPaths::set(); sparql_count_and_fuel(&fluree, &post, &annotated_count_sparql(Some(THRESHOLD))) @@ -228,120 +195,26 @@ async fn annotation_body_threshold_reduces_scan_work_on_both_surfaces() { }; assert_eq!(generic, kept, "fast-path-disabled lane must agree"); - // ---- stamp 1: the threshold is INSIDE the chain ------------- - // `body-filters` counts `Pattern::Filter`s in the chain body. It - // is 0 without the rewrite: the FILTER stays a sibling of the - // wrapper, where no block builder can reach it. - let db = support::graphdb_from_ledger(&post); - let explained = fluree - .explain(&db, &annotated_rows_jsonld(Some(THRESHOLD))) - .await - .expect("explain"); - let wrapper = find_op(&explained["plan"]["physical"], "DefaultGraphSourceOperator") - .expect("the annotated BGP must appear as a DefaultGraphSourceOperator"); - assert_eq!( - wrapper["details"]["kind"], "edge-annotation", - "explain must name the chain: {explained}" - ); - assert_eq!( - wrapper["details"]["body-filters"], 1, - "the threshold must be planned inside the chain body: {explained}" - ); - - // ---- stamp 2: it reaches the scan --------------------------- - // Pinned to `chain` (which executes as `generic`); module header for why. - let (pinned_all, pinned_all_fuel, pinned_kept, pinned_kept_fuel) = { - let _lane = LanePin::chain(); - let (a, af) = - sparql_count_and_fuel(&fluree, &post, &annotated_count_sparql(None)).await; - let (k, kf) = - sparql_count_and_fuel(&fluree, &post, &annotated_count_sparql(Some(THRESHOLD))) - .await; - (a, af, k, kf) - }; - assert_eq!(pinned_all as usize, EDGES, "chain pin, unfiltered count"); - assert_eq!(pinned_kept as usize, KEPT, "chain pin, thresholded count"); + // ---- the effect: it reaches the scan ------------------------ assert!( - pinned_kept_fuel < pinned_all_fuel, + kept_fuel < all_fuel, "a threshold keeping {KEPT}/{EDGES} rows must cut scan work, not just \ - rows in the answer: unfiltered {pinned_all_fuel}, filtered \ - {pinned_kept_fuel} (these were EQUAL before the rewrite)" + rows in the answer: unfiltered {all_fuel}, filtered {kept_fuel}" ); - // ---- the cross-module invariant the sink creates ------------ // A filter naming the base OBJECT variable passes the sink gate, - // because `?o` is produced by the base triple inside the wrapper. - // It then lands in a chain whose `f:reifiesObject` lookup is an - // elision candidate: with a pure COUNT, `?o` is in neither the - // projection nor `needed_outside`, so `elide_redundant_chain` - // (`default_graph_source.rs`) may drop that lookup. `?o` stays - // bound only because `collect_var_stats` (`where_plan.rs`) walks - // `Pattern::Filter` into the referenced set. - // - // That traversal predates this rewrite and nothing else connects - // the two modules, so delete it and every row here drops silently - // with no other test going red. This pins it, on every lane. - // (pin, the lane the engine actually reports for that pin). - // - // `chain` maps to `generic`, and that is not a demotion: there is no - // execution branch keyed on `ChainLane::Chain`. The variant means only - // "neither arena nor enumerate", after which control reaches the hash - // branch (admitted at >= 256 driving rows) and otherwise the generic - // chain. Pinning `chain` therefore cannot produce a `chain`-labelled - // execution on any shape. Encoding that here rather than in prose is - // the point: if it ever changes, this fails. - let (store, _tracing_guard) = support::span_capture::init_test_tracing(); - for (lane, expected_fired) in [ - ("arena", "arena"), - ("enumerate", "enumerate"), - ("chain", "generic"), - ] { - let _pin = LanePin::lane(lane); - let before = store.find_events("annotation delegate lane").len(); - let (n, obj_fuel) = sparql_count_and_fuel( - &fluree, - &post, - &annotated_count_with_object_var_sparql(), - ) - .await; - assert_eq!( - n as usize, KEPT_EXCLUDING_ONE_OBJECT, - "lane={lane}: the base object variable must stay bound inside \ - the chain — 0 here means `f:reifiesObject` was elided while \ - the sunk filter still reads `?o`" - ); - // The plain threshold on the same lane, so `enumerate` (which - // neither the default lane nor the chain pin exercises) has a - // filter-carrying correctness assertion too. - let (plain, _) = - sparql_count_and_fuel(&fluree, &post, &annotated_count_sparql(Some(THRESHOLD))) - .await; - assert_eq!(plain as usize, KEPT, "lane={lane}: thresholded count"); - - // The lane that actually executed, from the engine's own event. - // This is what distinguishes "three lanes ran" from "one lane ran - // three times" — pairwise-distinct fuel cannot, because a pin that - // falls through to a fourth path with its own cost still yields - // three distinct numbers and a mislabelled cell. - let fired: Vec = store.find_events("annotation delegate lane")[before..] - .iter() - .filter_map(|e| e.fields.get("lane").cloned()) - .collect(); - assert!( - !fired.is_empty(), - "lane={lane}: no `annotation delegate lane` event was captured, so \ - nothing here establishes which lane ran. An absent marker must \ - fail rather than pass — if the event moved, this assertion is \ - what needs updating, not deleting." - ); - assert!( - fired.iter().all(|f| f == expected_fired), - "lane={lane}: expected the engine to report `{expected_fired}`, got \ - {fired:?} (fuel {obj_fuel}). Either a runtime gate demoted the \ - pin, or the FLUREE_ANNOTATION_LANE override never reached the \ - query process, or the pin-to-lane mapping changed." - ); - } + // because `?o` is produced inside the wrapper. With a pure COUNT, + // `?o` is in neither the projection nor `needed_outside`, so it + // stays bound only because `collect_var_stats` (`where_plan.rs`) + // walks `Pattern::Filter` into the referenced set; without that, + // every row here drops silently. + let (n, _) = + sparql_count_and_fuel(&fluree, &post, &annotated_count_with_object_var_sparql()) + .await; + assert_eq!( + n as usize, KEPT_EXCLUDING_ONE_OBJECT, + "the base object variable must stay bound inside the wrapper" + ); // ---- twin surface: JSON-LD --------------------------------- let (jl_all, _) = @@ -356,55 +229,6 @@ async fn annotation_body_threshold_reduces_scan_work_on_both_surfaces() { .await; } -/// First node with this `op` in a rendered physical plan, depth-first. -fn find_op<'a>(node: &'a JsonValue, op: &str) -> Option<&'a JsonValue> { - if node["op"] == op { - return Some(node); - } - node["children"] - .as_array()? - .iter() - .find_map(|edge| find_op(&edge["node"], op)) -} - -/// Scoped `FLUREE_ANNOTATION_LANE`, restored on drop, serialized against any -/// other test in this binary that pins a lane. Results are lane-invariant, so -/// a concurrent annotation test seeing a pinned lane still gets its answer; -/// only a test asserting a *lane* would be disturbed, and this is the only one. -struct LanePin { - _guard: std::sync::MutexGuard<'static, ()>, - prev: Option, -} - -impl LanePin { - fn chain() -> Self { - Self::lane("chain") - } - - fn lane(name: &str) -> Self { - static LOCK: std::sync::OnceLock> = std::sync::OnceLock::new(); - let guard = LOCK - .get_or_init(|| std::sync::Mutex::new(())) - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner); - let prev = std::env::var("FLUREE_ANNOTATION_LANE").ok(); - std::env::set_var("FLUREE_ANNOTATION_LANE", name); - Self { - _guard: guard, - prev, - } - } -} - -impl Drop for LanePin { - fn drop(&mut self) { - match self.prev.take() { - Some(v) => std::env::set_var("FLUREE_ANNOTATION_LANE", v), - None => std::env::remove_var("FLUREE_ANNOTATION_LANE"), - } - } -} - /// Scoped planner-fast-path disable, restored on drop. /// /// Programmatic rather than `FLUREE_DISABLE_QUERY_FAST_PATHS`, because a scoped @@ -417,14 +241,11 @@ impl Drop for LanePin { /// binary, which `Drop` cannot undo — `it_join_batched_overlay` is a module of /// the same `grp_misc` binary and asserts specific lanes fired. /// -/// Which of those two happens is decided by test scheduling. `LanePin` above is -/// safe with an env var only because `forced_chain_lane()` re-reads per call: -/// same shape, opposite correctness, decided by the reader rather than the -/// setter. `set_fast_paths_disabled` is an `AtomicBool`, so it is both effective -/// and reversible; its own doc comment names this footgun. +/// Which of those two happens is decided by test scheduling. +/// `set_fast_paths_disabled` is an `AtomicBool`, so it is both effective and +/// reversible; its own doc comment names this footgun. /// -/// The mutex is for the same reason `LanePin` has one: process-wide state in a -/// parallel binary. +/// The mutex is there because this is process-wide state in a parallel binary. /// /// # Residual, and the full remedy if it ever bites /// diff --git a/fluree-db-api/tests/it_edge_annotations_indexed.rs b/fluree-db-api/tests/it_edge_annotations_indexed.rs index 8ac6dc92d3..45a500e87f 100644 --- a/fluree-db-api/tests/it_edge_annotations_indexed.rs +++ b/fluree-db-api/tests/it_edge_annotations_indexed.rs @@ -839,19 +839,17 @@ async fn non_annotation_ledger_skips_inject_annotations() { } #[tokio::test] -async fn explain_tags_annotation_role_and_uses_arena_stats() { - // M3.2: `/explain` must (a) expand `@annotation` patterns the same - // way the executor does, (b) tag the resulting - // `f:reifies*` triples with their slot name so the chosen ordering - // is observable, and (c) report stats as available when the - // annotation arena is sealed even if no other property stats - // exist yet. +async fn explain_expands_annotations_as_the_executor_does() { + // `/explain` must expand an `@annotation` pattern the way the executor + // does (the base edge and the reifier's `rdf:reifies` link), or an + // annotated query explains as nearly empty, and must report the stats the + // index build wrote. use crate::support::graphdb_from_ledger; let fluree = FlureeBuilder::memory() .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build_memory(); - let ledger_id = "it/edge-annotations-indexed:explain-tags"; + let ledger_id = "it/edge-annotations-indexed:explain-expansion"; let (local, handle) = support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); @@ -869,17 +867,8 @@ async fn explain_tags_annotation_role_and_uses_arena_stats() { .expect("pre-reindex cached load"); support::trigger_index_and_wait(&handle, ledger_id, after.receipt.t).await; support::wait_for_index_application(&fluree, ledger_id, after.receipt.t).await; - let post = fluree.ledger(ledger_id).await.expect("reload"); - assert!( - post.snapshot.annotation_index.is_some(), - "arena must be sealed for the explain test to exercise M3.1 stats" - ); - // Edge-rooted query filtered by annotation metadata. Lowering - // produces a `Pattern::EdgeAnnotation`, which `/explain` - // expands into a base edge triple + 3 `f:reifies*` lookups. - // (`@reifies` reads the `rdf:reifies` link instead.) let query = json!({ "@context": ctx(), "select": ["?person", "?org"], @@ -891,104 +880,39 @@ async fn explain_tags_annotation_role_and_uses_arena_stats() { } } }); - - // Pin the annotation-role tags on the normally-indexed - // snapshot first. - let db = graphdb_from_ledger(&post); - let resp = fluree.explain(&db, &query).await.expect("explain"); - + let resp = fluree + .explain(&graphdb_from_ledger(&post), &query) + .await + .expect("explain"); assert_ne!( resp["plan"]["optimization"], "none", - "with annotation arena present, explain must report stats availability \ - (got plan: {})", + "an indexed ledger must report stats availability (got plan: {})", resp["plan"] ); let optimized = resp["plan"]["optimized"] .as_array() .expect("optimized order is an array"); - - // Collect the annotation-role tags we saw, in optimized - // order, so the test pins both presence and ordering. - let roles: Vec = optimized + let properties: Vec<&str> = optimized .iter() - .filter_map(|entry| { - entry - .get("annotation-role") - .and_then(|v| v.as_str()) - .map(String::from) - }) + .filter_map(|entry| entry["pattern"]["property"].as_str()) .collect(); - - // The expansion emits exactly three `f:reifies*` triples - // per edge-annotation pattern (subject + predicate + - // object). The optimizer may reorder them but their - // count and slot identities are fixed. - let mut sorted = roles.clone(); - sorted.sort(); - assert_eq!( - sorted, - vec!["object", "predicate", "subject"], - "optimized order must contain exactly the three required \ - f:reifies* slots (got: {roles:?})" - ); - - // Sanity: every entry in the optimized order has a - // selectivity score field — we're not silently dropping - // patterns the planner doesn't have inputs for. + for expected in [ + "http://www.w3.org/1999/02/22-rdf-syntax-ns#reifies", + "ex:worksFor", + "ex:role", + ] { + assert!( + properties.contains(&expected), + "the expansion's {expected} triple must be planned: {properties:?}" + ); + } for entry in optimized { assert!( entry.get("selectivity").is_some(), "optimized entry missing selectivity: {entry}" ); } - - // M3.1 review fix: prove the arena-only stats path. Clone - // the LedgerState and strip `snapshot.stats` so the - // ordinary `IndexStats` is unavailable. With the M3.1 - // wiring, `/explain` should still report optimization - // (not "none") because `merge_annotation_stats` populates - // the view from `annotation_index` alone. - let mut arena_only = post.clone(); - assert!( - arena_only.snapshot.stats.is_some(), - "preconditions: post-reindex snapshot has IndexStats" - ); - std::sync::Arc::make_mut(&mut arena_only.snapshot).stats = None; - let db_arena_only = graphdb_from_ledger(&arena_only); - let resp_arena_only = fluree - .explain(&db_arena_only, &query) - .await - .expect("explain (arena-only stats)"); - assert_ne!( - resp_arena_only["plan"]["optimization"], "none", - "with snapshot.stats=None but annotation_index=Some, explain must \ - still report optimization via merged arena stats (got plan: {})", - resp_arena_only["plan"] - ); - assert_eq!( - resp_arena_only["plan"]["statistics"]["total-flakes"], 0, - "no IndexStats → total-flakes reports zero, but stats are still available" - ); - // Roles still present in the arena-only path. - let arena_only_roles: Vec = resp_arena_only["plan"]["optimized"] - .as_array() - .expect("optimized array") - .iter() - .filter_map(|entry| { - entry - .get("annotation-role") - .and_then(|v| v.as_str()) - .map(String::from) - }) - .collect(); - let mut arena_only_sorted = arena_only_roles.clone(); - arena_only_sorted.sort(); - assert_eq!( - arena_only_sorted, - vec!["object", "predicate", "subject"], - "arena-only path must still tag the three required slots (got: {arena_only_roles:?})" - ); }) .await; } diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 88c259c11d..038fbb55c0 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2365,29 +2365,6 @@ async fn jsonld_triple_term_functions() { assert_eq!(filtered, strings(&[&["ex:claim2"]])); } -/// TriG import writes an annotation inside a `GRAPH` block through its own -/// named-graph loop; that loop spools the link as the default-graph sink does. -#[tokio::test] -async fn trig_import_links_named_graph_annotations() { - const TRIG: &str = r#"@prefix ex: . -ex:alice ex:name "Alice" . -GRAPH { - ex:event1 ex:actor ex:alice {| ex:confidence "high" |} . - ex:event2 ex:actor ex:alice ~ ex:claim2 {| ex:confidence "low"@en |} . -} -"#; - let (fluree, ledger) = import(&[("data.trig", TRIG)], "it/triple-term-links:trig").await; - let got = run_link_query( - &fluree, - &ledger, - "SELECT ?s WHERE { GRAPH \ - { ?r rdf:reifies ?t BIND(SUBJECT(?t) AS ?s) } } ORDER BY ?s" - .to_string(), - ) - .await; - assert_eq!(got, strings(&[&["ex:event1"], &["ex:event2"]])); -} - /// A reifier re-pointed, over an index, at an edge to a subject only novelty /// holds: the term's object is that subject's provisional id while the edge's /// scan binds the subject by name, and the two must still join. @@ -2414,6 +2391,7 @@ async fn links_join_a_subject_only_novelty_holds() { .expect("re-point to carol"); let ledger = fluree.ledger(ledger_id).await.expect("load"); for body in [ + "SELECT ?o ?r WHERE { ex:alice ex:knows ?o ~ ?r }", "SELECT ?o ?r WHERE { ex:alice ex:knows ?o . ?r rdf:reifies <<( ex:alice ex:knows ?o )>> }", ] { assert_eq!( @@ -2424,3 +2402,25 @@ async fn links_join_a_subject_only_novelty_holds() { } } +/// TriG import writes an annotation inside a `GRAPH` block through its own +/// named-graph loop; that loop spools the link as the default-graph sink does. +#[tokio::test] +async fn trig_import_links_named_graph_annotations() { + const TRIG: &str = r#"@prefix ex: . +ex:alice ex:name "Alice" . +GRAPH { + ex:event1 ex:actor ex:alice {| ex:confidence "high" |} . + ex:event2 ex:actor ex:alice ~ ex:claim2 {| ex:confidence "low"@en |} . +} +"#; + let (fluree, ledger) = import(&[("data.trig", TRIG)], "it/triple-term-links:trig").await; + let got = run_link_query( + &fluree, + &ledger, + "SELECT ?s WHERE { GRAPH \ + { ?r rdf:reifies ?t BIND(SUBJECT(?t) AS ?s) } } ORDER BY ?s" + .to_string(), + ) + .await; + assert_eq!(got, strings(&[&["ex:event1"], &["ex:event2"]])); +} diff --git a/fluree-db-cypher/src/lower/pattern.rs b/fluree-db-cypher/src/lower/pattern.rs index 2770b41f0b..1e8f2b74a1 100644 --- a/fluree-db-cypher/src/lower/pattern.rs +++ b/fluree-db-cypher/src/lower/pattern.rs @@ -1120,6 +1120,7 @@ fn build_fixed_chain( edge: TriplePattern::new(gs, ctx.iri_ref(type_iri.to_string()), go.into()), annotation: ann, body, + term: ctx.fresh_synth(), }); } else { push_hop( @@ -1212,6 +1213,7 @@ fn build_rel_list_expr( edge, annotation: Ref::Var(ann), body: Vec::new(), + term: ctx.fresh_synth(), }])); rels.push(Expression::call( Function::Coalesce, @@ -1416,6 +1418,7 @@ fn push_rel_triple( edge: TriplePattern::new(s, pred, edge_o), annotation: Ref::Var(ann), body: Vec::new(), + term: ctx.fresh_synth(), }])); Expression::call(Function::Coalesce, vec![Expression::Var(ann), rel_value]) } else { @@ -1445,6 +1448,7 @@ fn push_rel_triple( edge, annotation: ann, body, + term: ctx.fresh_synth(), }); Ok(()) } @@ -1457,6 +1461,7 @@ fn push_rel_triple( edge, annotation: ann, body, + term: ctx.fresh_synth(), }); Ok(()) } diff --git a/fluree-db-query/src/execute.rs b/fluree-db-query/src/execute.rs index 72fc992869..cb78225d32 100644 --- a/fluree-db-query/src/execute.rs +++ b/fluree-db-query/src/execute.rs @@ -49,8 +49,8 @@ pub use runner::ExecutableQuery; // Re-export internal helpers for use in lib.rs pub use where_plan::build_where_operators_seeded; pub(crate) use where_plan::collect_var_stats; -pub use where_plan::expand_edge_annotation_patterns; pub(crate) use where_plan::{analyze_property_join_plan, collect_inner_join_block}; +pub use where_plan::{expand_edge_annotation_patterns, expand_edge_annotation_patterns_for}; // Re-export operator tree builder and runner for custom execution pipelines pub use operator_tree::{build_operator_tree, fast_paths_disabled, set_fast_paths_disabled}; diff --git a/fluree-db-query/src/execute/runner.rs b/fluree-db-query/src/execute/runner.rs index 47b0efd606..6be5b84449 100644 --- a/fluree-db-query/src/execute/runner.rs +++ b/fluree-db-query/src/execute/runner.rs @@ -397,6 +397,39 @@ pub async fn prepare_execution_with_config( .collect() } + // Annotation edges become links at planning, which encodes no IRIs; + // encoding theirs here lets a constant edge narrow the link scan. + fn encode_annotation_edges( + snapshot: &LedgerSnapshot, + patterns: &[Pattern], + ) -> Vec { + patterns + .iter() + .map(|p| match p { + Pattern::EdgeAnnotation { + edge, + annotation, + body, + term, + } => Pattern::EdgeAnnotation { + edge: TriplePattern { + s: encode_ref(snapshot, &edge.s), + p: encode_ref(snapshot, &edge.p), + o: encode_term(snapshot, &edge.o), + dtc: edge.dtc.clone(), + }, + annotation: encode_ref(snapshot, annotation), + body: encode_annotation_edges(snapshot, body), + term: *term, + }, + Pattern::Service(_) => p.clone(), + other => other + .clone() + .map_subpatterns(&mut |xs| encode_annotation_edges(snapshot, &xs)), + }) + .collect() + } + let rewritten_query = { let _rewrite_span = tracing::debug_span!( "pattern_rewrite", @@ -410,6 +443,12 @@ pub async fn prepare_execution_with_config( } else { query.query.patterns.clone() }; + let patterns_for_rewrite = + if super::where_plan::pattern_tree_has_edge_annotation(&patterns_for_rewrite) { + encode_annotation_edges(db.snapshot, &patterns_for_rewrite) + } else { + patterns_for_rewrite + }; let (rewritten_patterns, _diag) = rewrite_query_patterns( &patterns_for_rewrite, hierarchy.clone(), diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index 8b2aa21a6a..864b4ddcbd 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -74,9 +74,17 @@ use super::pushdown::extract_bounds_from_filters; /// the executor does — otherwise edge-annotation queries would /// surface as empty in the explain output. pub fn expand_edge_annotation_patterns(patterns: &[Pattern]) -> Vec { + expand_edge_annotation_patterns_for(patterns, true) +} + +/// [`expand_edge_annotation_patterns`], wrapping each chain in +/// [`Pattern::DefaultGraphSource`] only when `union` says the default graph +/// is a union of two or more graphs. Otherwise the chain stays in the +/// enclosing block, where the planner orders it with everything else. +pub fn expand_edge_annotation_patterns_for(patterns: &[Pattern], union: bool) -> Vec { let mut out = Vec::with_capacity(patterns.len()); for p in patterns { - expand_one_into(p.clone(), &mut out, false); + expand_one_into(p.clone(), &mut out, !union); } sink_filters_into_annotation_chains(&mut out); out @@ -116,99 +124,63 @@ pub(crate) fn pattern_tree_has_edge_annotation(patterns: &[Pattern]) -> bool { }) } -fn expand_one_into(pattern: Pattern, out: &mut Vec, inside_graph: bool) { +fn expand_one_into(pattern: Pattern, out: &mut Vec, one_graph: bool) { match pattern { Pattern::EdgeAnnotation { edge, annotation, body, + term, } => { - // Build the triple chain (base edge + three f:reifies* - // triples + recursively expanded body) into a local - // vector. The whole chain then gets wrapped in - // `Pattern::DefaultGraphSource` so the per-source - // iteration correlates them; the wrapper is skipped when - // we're already inside an explicit `Pattern::Graph`, - // which provides graph correlation by construction. + // Build the chain (the body, the annotation's link to the edge, + // and the base edge) into a local vector. It is wrapped in + // `Pattern::DefaultGraphSource` when the default graph is a union, + // so the per-source iteration correlates them; otherwise one graph + // is in scope and the chain joins its enclosing block. + // + // The order is the planner's tie-break: the body and the link are + // probes by reifier, the term's components then decode from the + // bound term, and the base edge last is an existence check on + // three bound positions. Estimates still decide where they differ + // (a constant subject makes the base edge the cheaper entry). let mut chain: Vec = Vec::new(); - // 1. Base edge triple: provides visibility. The standard - // scan applies snapshot rules + policy filters here, so - // an operator-style visibility check is redundant. - chain.push(Pattern::Triple(edge.clone())); - - // 2. Three required `f:reifies*` lookup triples that bind - // the annotation to the edge. - let ann_ref = annotation.clone(); - // `f:reifiesSubject` / `f:reifiesPredicate` objects are refs by - // construction, so on a VARIABLE object the `@id` constraint is - // a no-op filter — and it costs the batched subject-join lane - // (`is_batched_eligible` needs no dtc), which is what turns the - // per-reifier probes into one sorted SPOT walk. A constant - // object keeps the constraint so the lookup key encodes as a - // ref rather than a same-lexical string. - let id_dt = fluree_db_core::edge::id_datatype_sid(); - let id_dtc_for = |o: &Ref| match o { - Ref::Var(_) => None, - _ => Some(fluree_db_core::DatatypeConstraint::Explicit(id_dt.clone())), - }; - chain.push(Pattern::Triple(TriplePattern { - s: ann_ref.clone(), - p: reifies_subject_ref(), - o: edge.s.clone().into(), - dtc: id_dtc_for(&edge.s), - })); - chain.push(Pattern::Triple(TriplePattern { - s: ann_ref.clone(), - p: reifies_predicate_ref(), - o: edge.p.clone().into(), - dtc: id_dtc_for(&edge.p), - })); - // f:reifiesObject — preserves the original object's - // datatype constraint via `dtc` so typed-equality matches - // round-trip. For language-tagged literals - // (`DatatypeConstraint::LangTag`) this same clone is the - // intended per-language disambiguator: the writer DOES - // store the language tag on the f:reifiesObject flake's - // `m.lang` (verified by `it_edge_annotations:: - // cross_language_annotation_does_not_cross_match` — - // both flakes carry `dt=rdf:langString, - // m.lang=Some()`), so the LangTag dtc filter in - // `binary_scan` should pick exactly one annotation per - // language. This per-language disambiguation is exercised - // and green via `it_edge_annotations:: - // cross_language_annotation_does_not_cross_match` (the - // executor honors the `dtc` LangTag filter; the earlier - // `#[ignore]`d gap was fixed in commit c3117574e). - chain.push(Pattern::Triple(TriplePattern { - s: ann_ref, - p: reifies_object_ref(), - o: edge.o.clone(), - dtc: edge.dtc.clone(), - })); - - // 3. Body patterns (recursively expanded so nested - // annotations — though M0 rejects them — flatten too). - // The body inherits this expansion's wrapper context - // (already inside the chain we'll wrap below). + // 1. Body patterns (recursively expanded so nested annotations — + // though M0 rejects them — flatten too). The body inherits + // this expansion's wrapper context. for inner in body { expand_one_into(inner, &mut chain, true); } - // f:reifiesGraph is NOT emitted as a separate constraint - // triple — the `DefaultGraphSource` wrapper handles - // per-source correlation by switching execution context to - // one source at a time, so all f:reifies* triples scope to - // the same graph per iteration. The cross-graph misjoin - // (N×M cross-product under `from: [g1, g2]`) that - // motivated this wrapper is resolved by the per-source - // iteration. - - if inside_graph { - // Already inside a `Pattern::Graph` wrapper — the - // existing wrapper scopes the inner subplan to one - // graph per iteration, so no extra correlation - // layer is needed. + // 2. The annotation's `rdf:reifies` link to the edge's triple + // term, related to its components. + let reifies = Ref::Sid(fluree_db_core::Sid::new( + fluree_vocab::namespaces::RDF, + fluree_vocab::rdf_names::REIFIES, + )); + crate::ir::term_components::link_patterns( + annotation, + edge.clone(), + reifies, + || term, + &|_| None, + &mut chain, + ); + + // 3. Base edge triple: provides visibility. The standard scan + // applies snapshot rules + policy filters here, so an + // operator-style visibility check is redundant. + chain.push(Pattern::Triple(edge)); + + // Under a default-graph union the `DefaultGraphSource` wrapper + // switches the execution context to one member at a time, so the + // base edge and the link come from the same graph; without it a + // base edge from `g1` would pair with a link from `g2`. + + if one_graph { + // One graph is in scope (an enclosing `Pattern::Graph`, or a + // default graph that is not a union), so no correlation layer + // is needed and the chain joins its enclosing block. out.extend(chain); } else { out.push(Pattern::DefaultGraphSource { patterns: chain }); @@ -222,7 +194,7 @@ fn expand_one_into(pattern: Pattern, out: &mut Vec, inside_graph: bool) Pattern::Graph { .. } => { // Inside an explicit Pattern::Graph, expansion must // suppress the DefaultGraphSource wrapper — propagate - // `inside_graph = true`. + // `one_graph = true`. let expanded = pattern .map_subpatterns(&mut |inner| expand_edge_annotation_patterns_inside_graph(&inner)); out.push(expanded); @@ -235,9 +207,8 @@ fn expand_one_into(pattern: Pattern, out: &mut Vec, inside_graph: bool) | Pattern::Service(_) | Pattern::Subquery(_) | Pattern::DefaultGraphSource { .. } => { - // Container patterns inherit the current `inside_graph` - // context. - let was_inside = inside_graph; + // Container patterns inherit the current `one_graph` context. + let was_inside = one_graph; let expanded = pattern.map_subpatterns(&mut |inner| { let mut out = Vec::with_capacity(inner.len()); for p in inner { @@ -253,7 +224,7 @@ fn expand_one_into(pattern: Pattern, out: &mut Vec, inside_graph: bool) } } -/// Recursive entry that propagates `inside_graph = true`. Used by the +/// Recursive entry that propagates `one_graph = true`. Used by the /// `Pattern::Graph` arm above so its inner subtree doesn't synthesize /// a redundant `DefaultGraphSource`. fn expand_edge_annotation_patterns_inside_graph(patterns: &[Pattern]) -> Vec { @@ -404,27 +375,6 @@ fn sink_filters_into_annotation_chains(patterns: &mut Vec) { } } -fn reifies_subject_ref() -> Ref { - Ref::Sid(fluree_db_core::Sid::new( - fluree_vocab::namespaces::FLUREE_DB, - fluree_vocab::db::REIFIES_SUBJECT, - )) -} - -fn reifies_predicate_ref() -> Ref { - Ref::Sid(fluree_db_core::Sid::new( - fluree_vocab::namespaces::FLUREE_DB, - fluree_vocab::db::REIFIES_PREDICATE, - )) -} - -fn reifies_object_ref() -> Ref { - Ref::Sid(fluree_db_core::Sid::new( - fluree_vocab::namespaces::FLUREE_DB, - fluree_vocab::db::REIFIES_OBJECT, - )) -} - #[inline] fn filter_not_bound_var(expr: &Expression) -> Option { match expr { @@ -2646,7 +2596,10 @@ pub fn build_where_operators_seeded_with_needed( // whenever annotations are actually present. let expanded_storage: std::borrow::Cow<'_, [Pattern]> = if pattern_tree_has_edge_annotation(patterns) { - std::borrow::Cow::Owned(expand_edge_annotation_patterns(patterns)) + std::borrow::Cow::Owned(expand_edge_annotation_patterns_for( + patterns, + planning.default_graph_union, + )) } else { std::borrow::Cow::Borrowed(patterns) }; @@ -6537,18 +6490,14 @@ mod tests { // Edge-annotation expansion — `f:reifiesObject` dtc propagation // --------------------------------------------------------------------- - fn find_reifies_object_triple(chain: &[Pattern]) -> &TriplePattern { - let reifies_object = Ref::Sid(Sid::new( - fluree_vocab::namespaces::FLUREE_DB, - fluree_vocab::db::REIFIES_OBJECT, - )); + fn find_term_components(chain: &[Pattern]) -> &crate::ir::TermComponentsPattern { chain .iter() .find_map(|p| match p { - Pattern::Triple(tp) if tp.p == reifies_object => Some(tp), + Pattern::TermComponents(tc) => Some(tc), _ => None, }) - .expect("synthesized f:reifiesObject lookup triple must be present in chain") + .expect("the link's term components must be present in chain") } fn unwrap_default_graph_source(p: &Pattern) -> &[Pattern] { @@ -6559,7 +6508,7 @@ mod tests { } #[test] - fn expand_edge_annotation_propagates_explicit_dtc_to_reifies_object() { + fn expand_edge_annotation_keeps_explicit_dtc_on_the_term_object() { let xsd_string = fluree_db_core::DatatypeConstraint::Explicit(Sid::new(2, "string")); let edge = TriplePattern { s: Ref::Var(VarId(0)), @@ -6571,6 +6520,7 @@ mod tests { edge, annotation: Ref::Var(VarId(1)), body: Vec::new(), + term: VarId(3), }]; let expanded = expand_edge_annotation_patterns(&patterns); assert_eq!( @@ -6579,15 +6529,16 @@ mod tests { "expansion produces one DefaultGraphSource wrapper" ); let chain = unwrap_default_graph_source(&expanded[0]); - let reifies_obj = find_reifies_object_triple(chain); - assert_eq!(reifies_obj.dtc, Some(xsd_string)); + assert_eq!( + find_term_components(chain).object, + crate::ir::Component::Literal(FlakeValue::String("Alice".to_string()), xsd_string) + ); } #[test] - fn expand_edge_annotation_propagates_lang_tag_dtc_to_reifies_object() { - // The regression that motivates per-language disambiguation: - // a language-tagged literal on the base edge must constrain the - // synthesized f:reifiesObject lookup by the same lang tag, or + fn expand_edge_annotation_keeps_the_lang_tag_on_the_term_object() { + // Per-language disambiguation: a language-tagged literal on the base + // edge must constrain the link's term by the same tag, or // same-lexical strings in different languages would cross-match. let lang_fr = fluree_db_core::DatatypeConstraint::LangTag(std::sync::Arc::from("fr")); let edge = TriplePattern { @@ -6600,21 +6551,19 @@ mod tests { edge, annotation: Ref::Var(VarId(1)), body: Vec::new(), + term: VarId(3), }]; let expanded = expand_edge_annotation_patterns(&patterns); let chain = unwrap_default_graph_source(&expanded[0]); - let reifies_obj = find_reifies_object_triple(chain); - match &reifies_obj.dtc { - Some(fluree_db_core::DatatypeConstraint::LangTag(tag)) => { - assert_eq!(tag.as_ref(), "fr"); - } - other => panic!("expected LangTag dtc on f:reifiesObject, got {other:?}"), - } + assert_eq!( + find_term_components(chain).object, + crate::ir::Component::Literal(FlakeValue::String("chat".to_string()), lang_fr) + ); } #[test] - fn expand_edge_annotation_no_dtc_when_edge_has_none() { - // Ref-object edge: no constraint on either side. + fn expand_edge_annotation_variable_object_stays_a_variable() { + // Ref-object edge: the term's object joins the edge's variable. let edge = TriplePattern { s: Ref::Var(VarId(0)), p: Ref::Sid(Sid::new(100, "worksFor")), @@ -6625,11 +6574,14 @@ mod tests { edge, annotation: Ref::Var(VarId(1)), body: Vec::new(), + term: VarId(3), }]; let expanded = expand_edge_annotation_patterns(&patterns); let chain = unwrap_default_graph_source(&expanded[0]); - let reifies_obj = find_reifies_object_triple(chain); - assert!(reifies_obj.dtc.is_none()); + assert_eq!( + find_term_components(chain).object, + crate::ir::Component::Var(VarId(2)) + ); } // --------------------------------------------------------------------- @@ -6658,6 +6610,7 @@ mod tests { Ref::Sid(Sid::new(100, "confidence")), Term::Var(VarId(body_var)), ))], + term: VarId(100 + ann), } } @@ -6671,10 +6624,20 @@ mod tests { )) } + /// FILTERs other than the link's own `sameTerm(PREDICATE(?t),

)`. fn filter_count(patterns: &[Pattern]) -> usize { + let link_filter = |e: &Expression| { + matches!(e, Expression::Call { func: crate::ir::Function::SameTerm, args } + if matches!(args.first(), Some(Expression::Call { + func: crate::ir::Function::TripleSubject + | crate::ir::Function::TriplePredicate + | crate::ir::Function::TripleObject, + .. + }))) + }; patterns .iter() - .filter(|p| matches!(p, Pattern::Filter(_))) + .filter(|p| matches!(p, Pattern::Filter(e) if !link_filter(e))) .count() } @@ -6683,7 +6646,8 @@ mod tests { let patterns = vec![annotated_hop(0, 1, 2, 3), gt(3, 0.97)]; let expanded = expand_edge_annotation_patterns(&patterns); let chain = unwrap_default_graph_source(&expanded[0]); - // base + 3 f:reifies* + body triple + the sunk FILTER + // body triple + link + its predicate filter + term components + + // base + the sunk FILTER assert_eq!(chain.len(), 6); assert_eq!(filter_count(chain), 1); assert!( @@ -7003,6 +6967,7 @@ mod tests { name: crate::ir::GraphName::Var(g), patterns: vec![Pattern::Triple(make_pattern(VarId(13), "q", VarId(14)))], }], + term: VarId(15), }]; let strategy = choose_exists_strategy(&outer_schema, &inner); diff --git a/fluree-db-query/src/ir/pattern.rs b/fluree-db-query/src/ir/pattern.rs index 17e39917af..f9b4baa4c7 100644 --- a/fluree-db-query/src/ir/pattern.rs +++ b/fluree-db-query/src/ir/pattern.rs @@ -503,8 +503,7 @@ pub enum Pattern { /// # Execution /// /// Planning expands this variant into the base edge plus the - /// corresponding `f:reifies*` lookup chain before operator-tree - /// assembly. + /// annotation's `rdf:reifies` link to it before operator-tree assembly. EdgeAnnotation { /// The annotated edge (base triple). edge: TriplePattern, @@ -512,6 +511,9 @@ pub enum Pattern { annotation: Ref, /// Patterns about the annotation subject. body: Vec, + /// The variable the link binds to the edge's triple term, reserved + /// at lowering because planning cannot mint one. + term: VarId, }, } @@ -580,10 +582,12 @@ impl Pattern { edge, annotation, body, + term, } => Pattern::EdgeAnnotation { edge, annotation, body: f(body), + term, }, Pattern::DefaultGraphSource { patterns } => Pattern::DefaultGraphSource { patterns: f(patterns), @@ -687,11 +691,13 @@ impl Pattern { edge, annotation, body, + term, } => { edge.substitute_var(old, new); if let Ref::Var(v) = annotation { rename(v); } + rename(term); for p in body { p.substitute_var(old, new); } @@ -766,11 +772,13 @@ impl Pattern { edge, annotation, body, + term, } => { let mut vars = edge.referenced_vars(); if let Ref::Var(v) = annotation { vars.push(*v); } + vars.push(*term); vars.extend(body.iter().flat_map(Pattern::referenced_vars)); vars } @@ -822,11 +830,13 @@ impl Pattern { edge, annotation, body, + term, } => { let mut vars = edge.produced_vars(); if let Ref::Var(v) = annotation { vars.push(*v); } + vars.push(*term); vars.extend(body.iter().flat_map(Pattern::produced_vars)); vars } diff --git a/fluree-db-query/src/ir/term_components.rs b/fluree-db-query/src/ir/term_components.rs index ce2c49016e..8243166589 100644 --- a/fluree-db-query/src/ir/term_components.rs +++ b/fluree-db-query/src/ir/term_components.rs @@ -80,8 +80,27 @@ pub fn lower_reified_link( vars: &mut VarRegistry, out: &mut Vec, ) { - let reifies = encoder.encode_ref(fluree_vocab::rdf::REIFIES); + link_patterns( + annotation_ref, + edge, + encoder.encode_ref(fluree_vocab::rdf::REIFIES), + || fresh_term_var(vars), + &|iri| encoder.encode_iri(iri), + out, + ); +} +/// The patterns of [`lower_reified_link`], given the `rdf:reifies` ref, the +/// term variable (asked for only when the edge is not constant), and how to +/// encode an IRI. +pub(crate) fn link_patterns( + annotation_ref: Ref, + edge: TriplePattern, + reifies: Ref, + term_var: impl FnOnce() -> VarId, + encode_iri: &dyn Fn(&str) -> Option, + out: &mut Vec, +) { // Fully constant edge: compose the term itself. if let Some(term) = constant_term(&edge) { out.push(Pattern::Triple(TriplePattern { @@ -93,7 +112,7 @@ pub fn lower_reified_link( return; } - let t = fresh_term_var(vars); + let t = term_var(); out.push(Pattern::Triple(TriplePattern { s: annotation_ref, p: reifies, @@ -122,7 +141,7 @@ pub fn lower_reified_link( )); Component::Node(sid) } - Term::Iri(iri) => match encoder.encode_iri(&iri) { + Term::Iri(iri) => match encode_iri(&iri) { Some(sid) => { out.push(same_term( func, @@ -130,12 +149,16 @@ pub fn lower_reified_link( )); Component::Node(sid) } - // An IRI in no registered namespace names nothing in this - // ledger, so the pattern cannot match. + // Not encodable here (no ledger at hand, or not this one's): + // compared as the query runs, against the ledger it reads. None => { - out.push(Pattern::Filter(Expression::Const(FlakeValue::Boolean( - false, - )))); + out.push(same_term( + func, + Expression::call( + Function::Iri, + vec![Expression::Const(FlakeValue::String(iri.to_string()))], + ), + )); Component::Any } }, @@ -178,7 +201,7 @@ pub fn lower_reified_link( } /// A `?__term_N` variable no pattern uses yet. -fn fresh_term_var(vars: &mut VarRegistry) -> VarId { +pub fn fresh_term_var(vars: &mut VarRegistry) -> VarId { let name = (vars.len()..) .map(|n| format!("?__term_{n}")) .find(|name| vars.get(name).is_none()) diff --git a/fluree-db-query/src/lib.rs b/fluree-db-query/src/lib.rs index 8db0d62eb4..9507089b2f 100644 --- a/fluree-db-query/src/lib.rs +++ b/fluree-db-query/src/lib.rs @@ -127,8 +127,9 @@ pub use dataset_operator::{DatasetBuilder, DatasetOperator, ScanDatasetBuilder}; pub use distinct::DistinctOperator; pub use error::{QueryError, Result}; pub use execute::{ - build_operator_tree, execute, expand_edge_annotation_patterns, fast_paths_disabled, - run_operator, set_fast_paths_disabled, ContextConfig, ExecutableQuery, + build_operator_tree, execute, expand_edge_annotation_patterns, + expand_edge_annotation_patterns_for, fast_paths_disabled, run_operator, + set_fast_paths_disabled, ContextConfig, ExecutableQuery, }; pub use exists::ExistsOperator; pub use explain::{ diff --git a/fluree-db-query/src/parse/lower.rs b/fluree-db-query/src/parse/lower.rs index 48bb64dda4..e44ade62c3 100644 --- a/fluree-db-query/src/parse/lower.rs +++ b/fluree-db-query/src/parse/lower.rs @@ -450,6 +450,7 @@ pub fn lower_unresolved_pattern( edge: lowered_edge, annotation: lowered_annotation, body: lowered_body, + term: crate::ir::term_components::fresh_term_var(vars), }]) } UnresolvedPattern::AnnotationTarget { diff --git a/fluree-db-query/src/planner.rs b/fluree-db-query/src/planner.rs index 25ead35c14..a1414ffac0 100644 --- a/fluree-db-query/src/planner.rs +++ b/fluree-db-query/src/planner.rs @@ -1815,11 +1815,13 @@ pub(crate) fn must_bind_vars(pattern: &Pattern, bind_targets: BindTargets) -> Ha edge, annotation, body, + term, } => { let mut vars: HashSet = edge.produced_vars().into_iter().collect(); if let Ref::Var(v) = annotation { vars.insert(*v); } + vars.insert(*term); vars.extend(all(body)); vars } diff --git a/fluree-db-query/src/temporal_mode.rs b/fluree-db-query/src/temporal_mode.rs index 8546844eeb..cd780ee610 100644 --- a/fluree-db-query/src/temporal_mode.rs +++ b/fluree-db-query/src/temporal_mode.rs @@ -80,6 +80,11 @@ pub struct PlanningContext { /// (which assume bag cardinality over the union). Only ever `true` in /// current mode — history datasets keep per-event (assert/retract) rows. pub multi_default_graph: bool, + /// Whether the default graph is a union of two or more graphs, in any mode + /// (unlike [`Self::multi_default_graph`]). An annotated edge then reads its + /// base edge and its link one member at a time, so the two come from the + /// same graph. + pub default_graph_union: bool, /// What an unmatched OPTIONAL binds its optional-only variables to — the /// surface language's null semantics. Folded in from `Query` at the plan /// root; defaults to SPARQL's `Unbound`. @@ -99,6 +104,7 @@ impl PlanningContext { mode: TemporalMode::Current, allow_semantic_elision: false, multi_default_graph: false, + default_graph_union: false, unmatched_optional: UnmatchedOptional::Unbound, row_goal: None, } @@ -111,6 +117,7 @@ impl PlanningContext { mode: TemporalMode::History, allow_semantic_elision: false, multi_default_graph: false, + default_graph_union: false, unmatched_optional: UnmatchedOptional::Unbound, row_goal: None, } @@ -135,6 +142,7 @@ impl PlanningContext { #[inline] pub const fn with_multi_default_graph(mut self, multi: bool) -> Self { self.multi_default_graph = multi && self.mode.is_current(); + self.default_graph_union = multi; self } diff --git a/fluree-db-sparql/src/lower/annotation.rs b/fluree-db-sparql/src/lower/annotation.rs index 3b876fc489..a8faf5ba29 100644 --- a/fluree-db-sparql/src/lower/annotation.rs +++ b/fluree-db-sparql/src/lower/annotation.rs @@ -63,6 +63,7 @@ impl LoweringContext<'_, E> { edge: edge.clone(), annotation: annotation_ref, body, + term: fluree_db_query::ir::term_components::fresh_term_var(self.vars), }); } Ok(()) diff --git a/fluree-db-sparql/src/lower/construct.rs b/fluree-db-sparql/src/lower/construct.rs index ee721a99c3..f49da18889 100644 --- a/fluree-db-sparql/src/lower/construct.rs +++ b/fluree-db-sparql/src/lower/construct.rs @@ -204,6 +204,7 @@ impl LoweringContext<'_, E> { edge, annotation, body, + .. } => { let triple = out.push_pattern(edge.clone(), None); out.push_reification(triple, annotation.clone()); From 988e669375c675235e1905ddf0c4c9ea50984be6 Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 20:37:00 -0400 Subject: [PATCH 47/92] perf(query): an annotated edge is not re-checked against its base triple The link expansion joined the base edge to every annotation it read, so an annotation-syntax count over every annotated edge of the annotation benchmark dev slice took 11 s, against 0.8 s on the bundle chain, which elided the same check. Every write that attaches a reifier asserts its edge in the same graph and commit, and retracting the edge retracts the attachment, so a live link names a live edge; a policy hiding the edge hides the link. The expansion now emits the body and the link only: that count takes 0.4 s, and every other annotation-syntax shape timed is level or faster than the chain, with the same answers. A reifier that does not assert its triple ends the invariant, and the check then returns as a join. --- docs/design/edge-annotations.md | 2 +- .../tests/it_edge_annotations_indexed.rs | 7 ++-- fluree-db-query/src/execute/where_plan.rs | 36 ++++++++++--------- 3 files changed, 23 insertions(+), 22 deletions(-) diff --git a/docs/design/edge-annotations.md b/docs/design/edge-annotations.md index 640b0c8ada..93782fae4e 100644 --- a/docs/design/edge-annotations.md +++ b/docs/design/edge-annotations.md @@ -168,7 +168,7 @@ Annotation forward and reverse branch CIDs are returned by `IndexRoot::all_cas_i Every annotation pattern reads the link. Reified-triple patterns (`<< s p o ~ ?r >>`, `?r rdf:reifies <<( s p o )>>`, JSON-LD `@reifies`) lower to it directly: `lower_reified_link` (`fluree-db-query/src/ir/term_components.rs`), shared by the SPARQL and JSON-LD lowerings, emits the reifier's `rdf:reifies` link, `?r rdf:reifies ?t`, whose object is a triple-term handle the index's term dictionary resolves, plus `TermComponents(?t, s, p, o)` relating the term to its components and a `sameTerm` filter per constant component, which the planner turns into the scan's handle interval. A fully constant edge composes to a constant term instead. -Annotation syntax (`s p o ~ ?r {| ... |}`, JSON-LD `@annotation`, Cypher relationship properties) keeps its `Pattern::EdgeAnnotation` through lowering, with the term variable reserved there, and `expand_edge_annotation_patterns_for` (`where_plan.rs`) expands it at planning into the body, the link with its term components (`link_patterns`), and the base edge, which asserts that the edge exists and is visible. The order is the planner's tie-break: the body and the link are probes by reifier, the components decode from the bound term, and the base edge last is an existence check on three bound positions, so a range filter on the body drives the scan when nothing is more selective. The chain joins its enclosing block, wrapped in `Pattern::DefaultGraphSource` only when the default graph is a union of two or more graphs (`PlanningContext::default_graph_union`), where the wrapper reads one member at a time so the base edge and the link come from the same graph. The runner encodes the edge's IRIs first, since planning has no ledger to encode against; an IRI still unencoded is compared as the query runs. +Annotation syntax (`s p o ~ ?r {| ... |}`, JSON-LD `@annotation`, Cypher relationship properties) keeps its `Pattern::EdgeAnnotation` through lowering, with the term variable reserved there, and `expand_edge_annotation_patterns_for` (`where_plan.rs`) expands it at planning into the body and the link with its term components (`link_patterns`). The base edge is not joined: every write that attaches a reifier asserts its edge in the same graph and commit, and retracting the edge retracts the attachment, so a live link names a live edge, and a policy that hides the edge hides the link. Joining it probed the edge once per annotation (an annotation-syntax count over every annotated edge of the StarBench dev slice took 11 s against 0.8 s on the bundle chain, which elided the same check). A reifier that does not assert its triple ends that invariant, and the check then has to come back as a join. The order is the planner's tie-break: the body and the link are probes by reifier and the components decode from the bound term, so a range filter on the body drives the scan when nothing is more selective. The chain joins its enclosing block, wrapped in `Pattern::DefaultGraphSource` only when the default graph is a union of two or more graphs (`PlanningContext::default_graph_union`). The runner encodes the edge's IRIs first, since planning has no ledger to encode against; an IRI still unencoded is compared as the query runs. The link names its triple without joining the base edge, so visibility is checked on the term: `QueryPolicyEnforcer` lets a flake whose object is a triple term through only when the triple that term names would be visible, recursively for nested terms. A policy hiding `ex:worksFor` therefore hides the links to `ex:worksFor` edges on every route, not only on annotation-syntax patterns. diff --git a/fluree-db-api/tests/it_edge_annotations_indexed.rs b/fluree-db-api/tests/it_edge_annotations_indexed.rs index 45a500e87f..1f214604f2 100644 --- a/fluree-db-api/tests/it_edge_annotations_indexed.rs +++ b/fluree-db-api/tests/it_edge_annotations_indexed.rs @@ -841,9 +841,9 @@ async fn non_annotation_ledger_skips_inject_annotations() { #[tokio::test] async fn explain_expands_annotations_as_the_executor_does() { // `/explain` must expand an `@annotation` pattern the way the executor - // does (the base edge and the reifier's `rdf:reifies` link), or an - // annotated query explains as nearly empty, and must report the stats the - // index build wrote. + // does (the body and the reifier's `rdf:reifies` link), or an annotated + // query explains as nearly empty, and must report the stats the index + // build wrote. use crate::support::graphdb_from_ledger; let fluree = FlureeBuilder::memory() @@ -899,7 +899,6 @@ async fn explain_expands_annotations_as_the_executor_does() { .collect(); for expected in [ "http://www.w3.org/1999/02/22-rdf-syntax-ns#reifies", - "ex:worksFor", "ex:role", ] { assert!( diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index 864b4ddcbd..bd4451b91d 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -132,17 +132,25 @@ fn expand_one_into(pattern: Pattern, out: &mut Vec, one_graph: bool) { body, term, } => { - // Build the chain (the body, the annotation's link to the edge, - // and the base edge) into a local vector. It is wrapped in + // Build the chain (the body and the annotation's link to the + // edge) into a local vector. It is wrapped in // `Pattern::DefaultGraphSource` when the default graph is a union, // so the per-source iteration correlates them; otherwise one graph // is in scope and the chain joins its enclosing block. // // The order is the planner's tie-break: the body and the link are - // probes by reifier, the term's components then decode from the - // bound term, and the base edge last is an existence check on - // three bound positions. Estimates still decide where they differ - // (a constant subject makes the base edge the cheaper entry). + // probes by reifier, and the term's components then decode from + // the bound term. Estimates still decide where they differ (a + // constant subject anchors the components through the term + // dictionary). + // + // The base edge is not joined: every write that attaches a + // reifier asserts its edge in the same graph and commit, and + // retracting the edge retracts the attachment, so a live link + // names a live edge; a policy that hides the edge hides the link. + // Joining it anyway probed the edge once per annotation. A + // reifier that does not assert its triple would end that + // invariant, and with it this elision. let mut chain: Vec = Vec::new(); // 1. Body patterns (recursively expanded so nested annotations — @@ -160,22 +168,16 @@ fn expand_one_into(pattern: Pattern, out: &mut Vec, one_graph: bool) { )); crate::ir::term_components::link_patterns( annotation, - edge.clone(), + edge, reifies, || term, &|_| None, &mut chain, ); - // 3. Base edge triple: provides visibility. The standard scan - // applies snapshot rules + policy filters here, so an - // operator-style visibility check is redundant. - chain.push(Pattern::Triple(edge)); - // Under a default-graph union the `DefaultGraphSource` wrapper // switches the execution context to one member at a time, so the - // base edge and the link come from the same graph; without it a - // base edge from `g1` would pair with a link from `g2`. + // body and the link come from the same graph. if one_graph { // One graph is in scope (an enclosing `Pattern::Graph`, or a @@ -6647,11 +6649,11 @@ mod tests { let expanded = expand_edge_annotation_patterns(&patterns); let chain = unwrap_default_graph_source(&expanded[0]); // body triple + link + its predicate filter + term components + - // base + the sunk FILTER - assert_eq!(chain.len(), 6); + // the sunk FILTER + assert_eq!(chain.len(), 5); assert_eq!(filter_count(chain), 1); assert!( - matches!(chain[5], Pattern::Filter(_)), + matches!(chain[4], Pattern::Filter(_)), "the FILTER sinks to the end of the chain, where the body is: {chain:?}" ); // Copied, not relocated: the original keeps feeding From ec9d5cdb52b7433c3b03e87929324ca068fbedf0 Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 21:38:32 -0400 Subject: [PATCH 48/92] refactor(query): remove the bundle-chain read lanes Every annotation pattern now reads the rdf:reifies link, so nothing built the `[base edge + f:reifies*]` chain the lanes recognized: the forward-arena probe, the hash sidecar probe, the annotation-first enumeration, the chain elision and lane choice (FLUREE_ANNOTATION_LANE), the value-only OPTIONAL builder and the chain's cardinality model. annotation_edge_probe.rs goes, and DefaultGraphSourceOperator keeps only the per-source iteration and its build-once single-graph path. A value-only Cypher relationship binding (`OPTIONAL { EdgeAnnotation }`) had the OPTIONAL builder; its chain now holds term components, which the batched hash left-join admits, so it keeps the batched path instead of a per-row rebuild. --- docs/design/edge-annotations.md | 55 +- fluree-db-query/src/annotation_edge_probe.rs | 2088 ------------------ fluree-db-query/src/context.rs | 21 - fluree-db-query/src/default_graph_source.rs | 1040 +-------- fluree-db-query/src/execute.rs | 2 +- fluree-db-query/src/execute/where_plan.rs | 40 +- fluree-db-query/src/lib.rs | 1 - fluree-db-query/src/optional.rs | 262 +-- fluree-db-query/src/planner.rs | 171 +- 9 files changed, 67 insertions(+), 3613 deletions(-) delete mode 100644 fluree-db-query/src/annotation_edge_probe.rs diff --git a/docs/design/edge-annotations.md b/docs/design/edge-annotations.md index 93782fae4e..b17c4fc2c4 100644 --- a/docs/design/edge-annotations.md +++ b/docs/design/edge-annotations.md @@ -135,7 +135,7 @@ fn current_annotations_merged(edge: &EdgeKey, novelty_events: &[(Sid, i64, bool) fn current_targets_merged(ann: &Sid, novelty_events: &[(EdgeKey, i64, bool)], as_of_t: i64) -> Vec ``` -Callers (the hydration injector, the cascade pass, the planner-flattened `f:reifies*` triples) feed the arena's `collect_*_events` output into a single `(t, op)` visibility pass that resolves to currently-live state. Arena and novelty events can interleave arbitrarily — the merge applies one latest-wins pass over the union, so an arena `op = true` followed by a novelty `op = false` (or vice versa) resolves correctly without the caller doing any pre-merging. +Callers (the hydration injector and the cascade pass) feed the arena's `collect_*_events` output into a single `(t, op)` visibility pass that resolves to currently-live state. Arena and novelty events can interleave arbitrarily — the merge applies one latest-wins pass over the union, so an arena `op = true` followed by a novelty `op = false` (or vice versa) resolves correctly without the caller doing any pre-merging. The scan fallback (used when no arena is sealed) runs `db.range(POST, f:reifiesSubject, edge.s)` to find candidate annotation SIDs, then walks each candidate's bundle via `db.range(SPOT, s=ann_sid)` for structural decode. Bundle decode at the SPOT step bypasses view policy: the `f:reifies*` flakes are system-controlled discriminators, not user data, and policy-filtering them at decode would let an incidental policy that hides FLUREE_DB-namespace predicates collapse the bundle and drop the annotation entirely. The annotation **body** continues through the policy-filtered `format_subject` path, so user-data visibility is unchanged. @@ -164,7 +164,7 @@ Annotation forward and reverse branch CIDs are returned by `IndexRoot::all_cas_i - Old annotation CIDs become garbage only when no retained root references them. - Leaf-level CIDs behind a branch are walked during the GC-diff pass; `drop.rs` and `gc/collector.rs` both call into the expanded-CAS expansion helpers so a strict GC pass never deletes a still-reachable leaf. -## Reified-triple patterns read the link +## Annotation patterns read the link Every annotation pattern reads the link. Reified-triple patterns (`<< s p o ~ ?r >>`, `?r rdf:reifies <<( s p o )>>`, JSON-LD `@reifies`) lower to it directly: `lower_reified_link` (`fluree-db-query/src/ir/term_components.rs`), shared by the SPARQL and JSON-LD lowerings, emits the reifier's `rdf:reifies` link, `?r rdf:reifies ?t`, whose object is a triple-term handle the index's term dictionary resolves, plus `TermComponents(?t, s, p, o)` relating the term to its components and a `sameTerm` filter per constant component, which the planner turns into the scan's handle interval. A fully constant edge composes to a constant term instead. @@ -174,56 +174,7 @@ The link names its triple without joining the base edge, so visibility is checke Links exist only in indexes built since the link form. An annotated ledger (`has_annotations`) whose root has no term dictionary predates them, and its annotations live only in the `f:reifies*` bundles: link reads there (`BinaryScanOperator` on `rdf:reifies`, `TermComponentsOperator`) fail asking for a rebuild rather than answering without those annotations, and an incremental build over such a root leaves link synthesis off, so it never acquires a term dictionary that covers only its window. A full rebuild interns a term for every attachment in history and links them all. -## Query IR expansion - -Annotation syntax (`s p o ~ ?r`, `{| |}`, JSON-LD `@annotation`, Cypher relationship properties) lowers to `Pattern::EdgeAnnotation`, which is not executed as a dedicated operator. `expand_edge_annotation_patterns` (`fluree-db-query/src/execute/where_plan.rs`) flattens it into a base-edge `Pattern::Triple` plus three required `f:reifies*` triples (`f:reifiesSubject` / `f:reifiesPredicate` / `f:reifiesObject`) plus body patterns before the join planner runs. The standard scan/join/dedup machinery handles the rest, and the planner's `reorder_patterns` picks the cheaper direction (edge-first vs annotation-first) based on the merged `AnnotationStats` selectivity. - -The expansion does **not** emit an `f:reifiesGraph` constraint triple. Graph correlation is structural instead: each chain is scoped to one source at a time (the `Pattern::DefaultGraphSource` wrapper below, or an enclosing `Pattern::Graph`), so every `f:reifies*` lookup resolves against the same flake-level `g` per iteration — matching how `EdgeKey` derives graph identity. **Consequence:** the strict bundle-wellformedness rules under [`f:reifies*` durable encoding](#freifies-durable-encoding) above (required-slot, duplicate-slot, and `f:reifiesGraph`-vs-`g` `GraphMismatch` rejection) are guaranteed only on the **decode paths** — warmup / arena-build / hydration / cascade. Query-side annotation matching keys off the subject/predicate/object slots within the scoped graph; a malformed bundle introduced by bulk import (e.g. a named-graph bundle missing `f:reifiesGraph`) that decode would skip can still satisfy an `@annotation` / `rdf:reifies` query. Well-formed Fluree writes never produce such bundles — the JSON-LD writer emits `f:reifiesGraph` for named-graph edges and the SPARQL UPDATE surface is default-graph only — so this boundary matters only for hand-authored or externally-imported `f:reifies*` data. - -Multi-source default-graph datasets (`from: [g1, g2]`) wrap each expanded chain in `Pattern::DefaultGraphSource` so the base-edge match correlates per source — otherwise a base-edge from `g1` would cross-join with annotations from `g2`. `collect_var_stats` walks into `DefaultGraphSource` so bridge variables (e.g. `?ann` shared between the wrapper's inner chain and a sibling external triple) are correctly counted as join vars. - -### Read lanes for the expanded chain - -`DefaultGraphSourceOperator`'s single-graph delegate (`build_single_graph_delegate`, `fluree-db-query/src/default_graph_source.rs`) recognizes the canonical `[base edge + 3 f:reifies*]` shape (`recognize_annotation_edge`) and picks one of four physical lanes: - -1. **Forward-arena probe** (`AnnotationEdgeProbeOperator`) — when a sealed annotation arena exists, the overlay is empty, the query is current-state, and policy is root. One sorted merge-scan of the forward arena per query. Handles untyped (`-[p]->`) relationships (a late-materialized `Binding::EncodedPid` predicate decodes through the store's `p_sid_table`) and literal reified objects (the probe keys them by value + effective datatype + language tag, exactly as `EdgeKey::from_flake` stored them). -2. **Hash sidecar probe** (`HashAnnotationEdgeProbeOperator`, `fluree-db-query/src/annotation_edge_probe.rs`) — no arena needed. Drains the three `f:reifies*` predicates ONCE through ordinary planned scans (overlay-merged, policy-filtered) into one map keyed by the full reified edge `s → p → o → reifiers` (`AnnotationSidecarMaps`), then answers every base-edge row by `(s, p, o)` lookup. The map must be keyed by the whole edge: keyed by subject alone, every lookup walked all reifiers sharing that subject, a sum of squared reified out-degrees on hub-heavy data (StarBench BKR carries one subject with 3,345 `TREATS` edges). Object keys are representation-normalized `GroupKeyOwned`s, so ref and literal reified objects both match. Admitted at plan time when the driving estimate is ≥ 256 rows or unknown and the child does not already bind the reifier, then **re-decided at runtime**. With a bound reifier the chain is one point probe per row, while the lane walks every swept edge per row once the edge subject is unbound: a constant annotation value driving `<< ?s :p ?o >> :q "v"` over 20k edges took 55 s on the lane and 0.33 s on the chain (debug build, novelty only). At runtime: the lane buffers its whole driving stream before it can sweep anyway, so fewer than 256 buffered rows, or fewer than 256 distinct bound driving subjects, run the generic chain over the buffered rows (`probe_generic`, seeded through `BatchReplayOperator`) instead of draining the sidecar and sweeping the base predicate — StarBench P1 swept the whole graph for 9 s to enrich 3 rows. The value-only OPTIONAL lane (`AnnotationValueOptionalBuilder`, `fluree-db-query/src/optional.rs`) shares the same map-building machinery. -3. **Annotation-first enumeration** (`AnnotationEnumerateOperator`) — chosen when the whole arena is the cheapest entry point (a wildcard `<< ?s ?p ?o >>`, or a predicate wider than the arena), both endpoints are variables the child leaves unbound, and the child binds none of the chain's variables. Streams the **reverse** arena leaf by leaf (`AnnotationArenaReader::reverse_leaf_entries` + `live_pairs_in_reverse_leaf`, uncached so the walk retains nothing) and emits `(s, p, o, ann)` per live pair, filtered by the predicate when it is constant. Reifier order is dictionary order, and the lane encodes each ref through the subject dictionary (`find_subject_id_by_parts`, then dictionary novelty) so it emits `EncodedSid` / `EncodedPid` like scan output: the materialized form cost 40% of P2's time in `Arc` clone/drop churn downstream, and random-order encodes of 21M blank-node reifiers would thrash the dictionary. Edge subjects and objects recur across the walk far more often than they are distinct, so the lane memoizes their ids per operator (P4 resolves 33.6M positions over far fewer nodes); reifiers are unique per edge and take the direct lookup. No base-edge scan and no per-reifier probe: on full-scale StarBench P2 the generic chain re-probed the base edge 33.6M times (757 s, single-threaded, no IO). No base-edge visibility check is needed under this lane's gates (root policy, current state, empty overlay) because every write surface asserts the base triple it annotates, `@reifies`-rooted inserts are rejected, and a base-edge retract cascades to its attachments — a live arena row implies a live edge. -4. **Generic join chain** — everything else. `reorder_patterns` orders the chain; the broad-sidecar demotion (`is_broad_annotation_sidecar`) applies only while a connected non-chain alternative can seed **and the sidecar's reifier subject is still unbound** (`reifier_subject_bound`). Once `?ann` is bound the sidecar is a one-row subject probe; demoting it handed the seed to a fan-out step — the base edge with only `(s, p)` bound, or a body triple such as `?ann :derives_from ?x` at ~20 rows per reifier — so every later per-row probe ran once per fanned row (StarBench P2 / P11). An untyped relationship's chain (where `f:reifiesPredicate` joins on the base predicate VAR) still drives from a bound-object `f:reifiesSubject`/`f:reifiesObject` probe (per-row fan ≈ node degree) and never from `f:reifiesPredicate` (fan ≈ sidecar / #relationship-types). - -#### Choosing between the lanes - -`choose_chain_lane` (`default_graph_source.rs`) decides by what the child binds, then by cost in rows walked: - -- **Reifier bound by the child** → the chain: one point probe per driving row, which no sweep beats (P22 / C7 / C10 regressed 200× when the arena probe was taken there). -- **Base-edge subject bound** (a constant, or a variable the child binds) → the arena probe (below the 20M-row buffer ceiling). The probe keys into one forward leaf per driving row, while the chain's `f:reifiesSubject ` lookup walks every reifier of that subject — a hub tail the mean-based estimates cannot see. On the full StarBench ledger the fully-bound P23 went from 46 ms to 192 s on the chain and the child-driven P19 ran out of memory. -- **Subject unbound** → the cheapest of three costs. The arena probe pays the base-edge rows plus `LEAF_ROW_EQUIV` (≈400 rows) per leaf it touches; forward leaves are 4,096-row CBOR blobs ordered by `(s, p, o)`, so edges with an unbound subject land in about one leaf each, and a predicate-wide probe decodes the whole arena whatever the predicate's size (P11 / S5 / C8 / C10 all cost the same 13–15 s, 82% of it in leaf decode). The enumeration costs the arena's live pairs times ~12: each pair is a CBOR row decode plus two dictionary re-encodes of the endpoints. The chain costs the reifier candidates its cheapest `f:reifies*` lookup yields — `f:reifiesSubject`, `f:reifiesPredicate` or `f:reifiesObject` with the child's bindings — times ~16 for the generic chain (one scattered point probe per sequential row, the crossover measured on the slice) or ~2 when the chain is elided (below). `f:reifiesPredicate` counts here even though its uniform per-predicate estimate is too coarse to size the wrapper's output by: a predicate-only chain really does drive from that POST range. - -#### Eliding the redundant chain checks - -`elide_redundant_chain` rewrites the generic chain before it is planned. The write invariants — every reifier carries exactly one `f:reifiesSubject` / `f:reifiesPredicate` / `f:reifiesObject`, written into the edge's own graph, with the base triple asserted (`@reifies` without it is rejected; retracting the base cascades) — mean the base edge never removes a row once a reifies lookup has bound the reifier, and a lookup whose position is a variable nothing reads (neither a later sibling, nor the post-WHERE pipeline, nor the wrapper's own body) is a cardinality-one no-op. So the base edge is dropped, each unread variable position drops its lookup, constant positions keep theirs, and at least one lookup always remains (a reifier bound by the body still has to be a reifier: P3 has three `:derives_from` rows on plain subjects that P2 must not count). A read *variable predicate* blocks the whole rewrite — the base scan binds it as a predicate id, the reifies lookup as a plain ref, and those are not interchangeable downstream. The rewrite is gated on root policy, a single ledger and a non-history query; a past-`t` snapshot qualifies, because a reifier's bundle and its base edge are written and cascaded in the same commit. - -Two things count as a read that are easy to miss. A variable in two positions, or naming the reifier, keeps its lookups: once the base edge is gone they are the only thing equating those positions (`<< ?s :p ?s >>` must not match `:a :p :b`). And under `SELECT *` there is no projection-pushdown set, so the post-WHERE read set is every variable `collect_var_stats` finds in the un-expanded WHERE clause. That walk has to look inside `EdgeAnnotation`, or the annotated triple's own variables are dropped from every row (W3C `sparql12/eval-triple-terms` basic-4, basic-5, pattern-7, pattern-8). - -Measured on the full StarBench ledger the predicate-only chain went from one SPOT point scan per reifier to one POST range plus the body's batched probe: P11 4.9 s → 0.42 s, C11 8.4 s → 0.61 s, S5 6.9 s → 1.8 s, and P2 on the elided chain (a single `f:reifiesSubject` range over all 21.4M reifiers) 10.6 s against 65 s on the enumeration — which is why the elided chain is costed at ~2 rows per reifier and the enumeration at ~12 per pair. - -`FLUREE_ANNOTATION_LANE=arena|enumerate|chain` pins the lane for A/B timing; the runtime gates still apply. - -#### Placing the wrapper among its siblings - -`estimate_pattern` costs `Pattern::DefaultGraphSource` with `estimate_annotation_chain_cardinality`, not the generic branch model. The branch model multiplies the chain's triples in standalone-selectivity order with no regard for connectivity, so `<< ?s :P ?o >>` came out as `f:reifiesPredicate × base edge` (3,846 × 102,555 ≈ 4e8 on StarBench P11) and the wrapper sorted *behind* its own 6.5M-row `:derives_from` body triple, which then drove the chain once per row — the 60-second timeouts on every `<< ?s :P ?o >> :q ?x` shape. The chain binds one reifier per matching base edge, so its cardinality is the cheaper of its two entry points (the base edge, or the most selective `f:reifies*` lookup) times the body's expansion with the edge and reifier bound. The `/explain` output still shows the un-expanded `annotation-target` estimate in its logical plan; the executor plans the expanded wrapper. - -`f:reifiesSubject` / `f:reifiesPredicate` carry the `@id` datatype constraint only when their object is a constant. On a variable object it is a no-op filter (those slots are refs by construction) and it disqualified the batched subject-join lane (`is_batched_eligible` requires no dtc), turning the per-reifier probes into scattered scan opens. On a constant ref object the batched existence lane admits it (`is_batched_subject_exists_eligible`): the probe compares a ref constant against ref values only, so the constraint is already enforced, and without the admission every reifier candidate of a predicate-only or object-bound quoted triple opened its own point scan (~150 µs each). - -A rel var whose properties are read anywhere (`r.prop`, `properties(r)`, …) lowers to this REQUIRED chain — unreified edges do not match, per Cypher's relationship semantics. Value-only rel vars keep the plain base triple + OPTIONAL probe so unreified edges match with a synthesized relationship value. - -#### Buffering and the sweep ceiling - -**Both probe lanes are pipeline breakers.** `next_batch` runs the probe to completion before emitting the first row, so a query with a small `LIMIT` over a large driving side pays the full probe rather than stopping early. This is inherited behavior — the arena lane has always buffered — but the hash lane adds a sweep of the *base* edges on top of it, so the over-read is larger there. `PlanningContext` carries no limit, so a LIMIT-aware gate would need real plumbing; until then, treat the buffering as the lane's cost model rather than an oversight. - -`base_sweep_bounded` caps that sweep at `BASE_SWEEP_MAX_ROWS = 20_000_000` rows, measured as the predicate's flake count for a typed relationship and `stats.total_property_flakes()` (the whole default graph) for an untyped one. The number is a **backstop against pathological graphs, not a tuned optimum**: the lane was validated at ~190k reified edges over 72k nodes, so 20M is roughly 100× the measured scale, chosen as the point where one sequential sweep (tens of millions of rows through a planned scan) stops being obviously cheaper than the per-row probes it replaces. Nothing between the validated scale and the ceiling has been measured; a future tuner wanting a defensible value should benchmark the sweep against the nested-loop alternative at 1M/5M/20M and set the crossover from data. - -One sharp edge inside the sweep: `keep_all` disables the driving-subject filter entirely if **any single** driving row leaves the subject unbound, flipping the sweep from "filtered to the driving subjects" to "every base edge in the graph." That is correct — an unbound subject can match any edge — but it is a cliff, not a gradient, and it is not visible from the gate names. +A rel var whose properties are read anywhere (`r.prop`, `properties(r)`, …) lowers to a REQUIRED annotation pattern — unreified edges do not match, per Cypher's relationship semantics. Value-only rel vars keep the plain base triple + an OPTIONAL annotation pattern (batched as a hash left-join) so unreified edges match with a synthesized relationship value. ## See also diff --git a/fluree-db-query/src/annotation_edge_probe.rs b/fluree-db-query/src/annotation_edge_probe.rs deleted file mode 100644 index 4c22b80505..0000000000 --- a/fluree-db-query/src/annotation_edge_probe.rs +++ /dev/null @@ -1,2088 +0,0 @@ -//! Forward-arena probe operator — the physical counterpart to a Cypher -//! relationship binding (`MATCH (a)-[r:P]->(b) … RETURN r.prop`). -//! -//! ## Why this exists -//! -//! Binding a relationship variable reifies the edge: Fluree's RDF-star -//! lowering turns `[r:P]` into a base edge plus a `f:reifies*` sidecar -//! (`?r f:reifiesSubject a`, `?r f:reifiesPredicate P`, `?r -//! f:reifiesObject b`). Executed as generic triple joins those three -//! lookups scatter across the *whole* annotation sidecar in the base -//! index — the cost behind IC5's timeout. -//! -//! This operator replaces those three joins with one **forward-arena -//! merge-scan**: it drives a stream of fully-bound base edges, builds an -//! [`EdgeKey`] per row, and probes the annotation arena's forward index -//! (`EdgeKey → ann`) in a single sorted pass. The reifier variable `?r` -//! is bound directly; the relationship-property reads downstream -//! (`?r joinDate ?d`) then plan as ordinary subject-keyed lookups. -//! -//! ## Where it slots -//! -//! Recognized and built inside [`crate::default_graph_source`]'s -//! single-graph delegate, where the expanded chain `[base edge + 3 -//! f:reifies* + body]` is already grouped. The base edge plans normally -//! (so visibility + policy filtering still happen on it), this operator -//! enriches each surviving edge with its reifier, and the body plans -//! normally on top. When any gate fails the caller keeps the generic -//! join chain — a slower but identical-result fallback. -//! -//! ## Gates (all checked before this operator is built) -//! -//! - a forward annotation arena is sealed on the snapshot, -//! - current-state query (history falls back — the arena reader's -//! visibility model is `as_of_t`, but ranged history is out of scope), -//! - the attachment overlay is empty (so the indexed arena is -//! authoritative; with annotation novelty the per-edge merged path is -//! required and we fall back), -//! - root / no policy (the base edge and body stay policy-filtered via -//! their own scans; the structural `f:reifies*` binding comes from the -//! arena, so we gate it to root to avoid leaking a reifier a policy -//! would hide). - -use crate::binding::{Batch, Binding}; -use crate::context::ExecutionContext; -use crate::error::{QueryError, Result}; -use crate::group_aggregate::{binding_to_group_key_normalized, GroupKeyOwned}; -use crate::ir::{Pattern, Ref, Term, TriplePattern}; -use crate::operator::{BoxedOperator, Operator, OperatorState}; -use crate::temporal_mode::PlanningContext; -use crate::var_registry::VarId; -use async_trait::async_trait; -use fluree_db_binary_index::annotation_arena::AnnotationArenaReader; -use fluree_db_core::edge::{id_datatype_sid, EdgeKey}; -use fluree_db_core::storage::ContentStore; -use fluree_db_core::{AnnotationIndexRoot, FlakeValue, Sid, StatsView}; -use rustc_hash::FxHashMap; -use std::collections::HashMap; -use std::sync::Arc; - -/// The recognized `[base edge + 3 f:reifies* + body]` shape, decomposed -/// into what the probe operator and its surrounding plan need. -pub(crate) struct AnnotationEdgeShape { - /// Base-edge triple, planned normally (visibility + policy). - pub base: Pattern, - /// Reifier variable bound by the probe. - pub ann_var: VarId, - pub s_pos: EdgePos, - /// Constant relationship predicate. Cypher lowers relationship types - /// to `Ref::Iri`; the caller resolves it to a `Sid` (with `ctx`) - /// before constructing the operator, falling back if it can't. - pub p_pred: Ref, - pub o_pos: EdgePos, - /// Remaining patterns (relationship-property reads, filters), planned - /// normally on top of the probe with `ann_var` bound. - pub body: Vec, -} - -/// True iff `r` is the constant `f:` predicate ref. -fn is_reifies_pred(r: &Ref, name: &str) -> bool { - matches!(r, Ref::Sid(sid) - if sid.namespace_code == fluree_vocab::namespaces::FLUREE_DB - && sid.name.as_ref() == name) -} - -/// Recognize the expanded edge-annotation chain produced by -/// `expand_edge_annotation_patterns`: a base-edge triple followed -/// immediately by the three `f:reifies{Subject,Predicate,Object}` -/// triples (all sharing one reifier var, their objects matching the -/// base edge's s/p/o), then arbitrary body patterns. -/// -/// Returns `None` (→ generic-join fallback) unless every structural and -/// fast-path-eligibility condition holds: a constant relationship -/// predicate, ref-valued subject and object (node-to-node edge), and the -/// three reifies triples in canonical position. -pub(crate) fn recognize_annotation_edge(patterns: &[Pattern]) -> Option { - use fluree_vocab::db::{REIFIES_OBJECT, REIFIES_PREDICATE, REIFIES_SUBJECT}; - - if patterns.len() < 4 { - return None; - } - let Pattern::Triple(base) = &patterns[0] else { - return None; - }; - let (Pattern::Triple(r_subj), Pattern::Triple(r_pred), Pattern::Triple(r_obj)) = - (&patterns[1], &patterns[2], &patterns[3]) - else { - return None; - }; - - // Constant relationship type (a typed Cypher relationship lowers to - // `Ref::Iri`; a `Ref::Sid` is also accepted) or a variable predicate - // (untyped `-[p]->`): the probe resolves a variable per row from the - // base-edge binding, so both are probeable. - let p_pred = base.p.clone(); - // Subject and object must be node refs (no IRI/literal objects in v1). - let s_pos = EdgePos::from_ref(&base.s)?; - let o_pos = EdgePos::from_term(&base.o)?; - - // The three reifies triples share one reifier var as subject. - let ann_var = r_subj.s.as_var()?; - if r_pred.s.as_var() != Some(ann_var) || r_obj.s.as_var() != Some(ann_var) { - return None; - } - if !is_reifies_pred(&r_subj.p, REIFIES_SUBJECT) - || !is_reifies_pred(&r_pred.p, REIFIES_PREDICATE) - || !is_reifies_pred(&r_obj.p, REIFIES_OBJECT) - { - return None; - } - // Their objects must reference the base edge's s / p / o. - if r_subj.o != Term::from(base.s.clone()) - || r_pred.o != Term::from(base.p.clone()) - || r_obj.o != base.o - { - return None; - } - - Some(AnnotationEdgeShape { - base: patterns[0].clone(), - ann_var, - s_pos, - p_pred, - o_pos, - body: patterns[4..].to_vec(), - }) -} - -/// How to obtain one position (subject / object) of the base edge for a -/// given child row. Predicate is always a constant for a typed Cypher -/// relationship, so it is stored directly as a `Sid` on the operator. -#[derive(Clone)] -pub(crate) enum EdgePos { - /// Constant ref taken straight from the pattern. - Const(Sid), - /// Ref carried by a child-row variable binding. - Var(VarId), -} - -impl EdgePos { - /// A subject/predicate ref position. `None` (→ recognition falls - /// back) for cross-ledger `Iri` refs, which this single-ledger fast - /// path cannot probe. - pub(crate) fn from_ref(r: &Ref) -> Option { - match r { - Ref::Sid(sid) => Some(EdgePos::Const(sid.clone())), - Ref::Var(v) => Some(EdgePos::Var(*v)), - Ref::Iri(_) => None, - } - } - - /// An object ref position. `None` for literal/IRI objects — only - /// node-ref edges are handled in v1. - pub(crate) fn from_term(t: &Term) -> Option { - match t { - Term::Sid(sid) => Some(EdgePos::Const(sid.clone())), - Term::Var(v) => Some(EdgePos::Var(*v)), - Term::Iri(_) | Term::Value(_) => None, - } - } -} - -/// Probe the forward annotation arena to bind a reifier variable from a -/// stream of base edges. See the module docs for the recognized shape. -pub struct AnnotationEdgeProbeOperator { - child: BoxedOperator, - /// Reifier variable to bind (`?r`). - ann_var: VarId, - /// Base-edge subject source. - s_pos: EdgePos, - /// Base-edge predicate source: constant for a typed relationship, - /// per-row for an untyped `-[p]->` (the base scan binds it as a Sid). - p_pos: EdgePos, - /// Base-edge object source. - o_pos: EdgePos, - schema: Arc<[VarId]>, - state: OperatorState, - /// Owned arena root + store, captured at `open` from the snapshot. - root: Option, - store: Option>, - as_of_t: i64, - /// Output rows, filled by a single probe pass over the whole child - /// stream on the first `next_batch`, then drained in chunks. One pass - /// = one arena reader, so the forward branch/leaves decode once rather - /// than per child batch. - probed: bool, - result_buffer: Vec>, - buffer_pos: usize, -} - -/// Output rows emitted per `next_batch` once the probe pass has filled -/// the buffer. Keeps any single output batch bounded. -const PROBE_OUTPUT_CHUNK: usize = 4096; - -/// Below this many bound driving subjects the generic (well-ordered) -/// `f:reifies*` chain — three narrow per-row probes — beats draining the -/// whole annotation sidecar into hash maps and sweeping the base predicate. -/// Checked twice: at plan time against the driving estimate (an unknown -/// estimate admits the lane) and again at runtime by -/// [`HashAnnotationEdgeProbeOperator`] against the buffered stream itself. -pub(crate) const HASH_ANNOTATION_MIN_DRIVING_ROWS: usize = 256; - -impl AnnotationEdgeProbeOperator { - pub(crate) fn new( - child: BoxedOperator, - ann_var: VarId, - s_pos: EdgePos, - p_pos: EdgePos, - o_pos: EdgePos, - ) -> Self { - let mut schema_vec: Vec = child.schema().to_vec(); - if !schema_vec.contains(&ann_var) { - schema_vec.push(ann_var); - } - let schema = Arc::from(schema_vec.into_boxed_slice()); - - Self { - child, - ann_var, - s_pos, - p_pos, - o_pos, - schema, - state: OperatorState::Created, - root: None, - store: None, - as_of_t: 0, - probed: false, - result_buffer: Vec::new(), - buffer_pos: 0, - } - } - - /// Build the `EdgeKey` for one child row. Returns `None` only when a - /// position binding is absent (unbound/poisoned) — those rows can - /// carry no reifier and are dropped, matching the generic-join - /// semantics (an unbound edge position matches no `f:reifies*` row). - fn edge_key_for_row( - &self, - batch: &Batch, - row: usize, - view: Option<&fluree_db_binary_index::BinaryGraphView>, - ) -> Result> { - let Some(s) = resolve_pos_ref(batch, row, &self.s_pos, view)? else { - return Ok(None); - }; - let Some(p) = resolve_pos_pred(batch, row, &self.p_pos, view)? else { - return Ok(None); - }; - // Object: any FlakeValue. Ref objects key with the `@id` - // datatype; literal objects (a wildcard `?s ?p ?o` base scan - // also delivers literal-valued triples, and SPARQL quoted - // triples reify them) key by value + effective datatype + - // language tag — exactly as `EdgeKey::from_flake` stored them, - // so reified literal triples are probeable rather than dropped. - let Some((o, dt, lang)) = resolve_obj_key_parts(batch, row, &self.o_pos, view)? else { - return Ok(None); - }; - Ok(Some(EdgeKey { - g: None, - s, - p, - o, - dt, - lang, - list_i: None, - })) - } -} - -/// Resolve the base edge's OBJECT position into arena `EdgeKey` parts -/// (`o`, `dt`, `lang`) exactly as [`EdgeKey::from_flake`] recorded them -/// at write time. `None` drops the row: an unbound position matches no -/// `f:reifies*` row, and Poisoned follows join semantics (a failed -/// OPTIONAL row joins nothing). Value/dt/lang decode through the same -/// dictionaries the writer used — sound here because the arena gates -/// require an empty overlay, so every binding comes from the indexed -/// scan. -fn resolve_obj_key_parts( - batch: &Batch, - row: usize, - pos: &EdgePos, - view: Option<&fluree_db_binary_index::BinaryGraphView>, -) -> Result)>> { - let ref_parts = |sid: Sid| (FlakeValue::Ref(sid), id_datatype_sid(), None); - match pos { - EdgePos::Const(sid) => Ok(Some(ref_parts(sid.clone()))), - EdgePos::Var(v) => { - let Some(b) = batch.get(row, *v) else { - return Ok(None); - }; - match b { - Binding::Sid { sid, .. } => Ok(Some(ref_parts(sid.clone()))), - Binding::EncodedSid { .. } => { - Ok(resolve_pos_ref(batch, row, pos, view)?.map(ref_parts)) - } - Binding::Unbound | Binding::Poisoned => Ok(None), - other => { - // Literal object: decode to (value, effective dt, lang). - match crate::group_aggregate::materialize_encoded(other, view) { - Binding::Lit { val, dtc, .. } => Ok(Some(( - val, - dtc.datatype().clone(), - dtc.lang_tag().map(str::to_string), - ))), - // Anything else (cross-ledger Iri, list/path/rel - // values) has no arena key — the row matches no - // reified edge. - _ => Ok(None), - } - } - } - } - } -} - -/// Resolve the predicate position of a base edge. A variable predicate -/// (untyped `-[p]->`) arrives from the base-edge scan either as an eager -/// `Binding::Sid` or late-materialized as `Binding::EncodedPid` — decode -/// the latter through the store's predicate table. An id absent from the -/// persisted table is a shape violation (both probe paths run on planned -/// scans whose predicate ids come from that table) — surface it loudly -/// rather than silently dropping the row. -pub(crate) fn resolve_pos_pred( - batch: &Batch, - row: usize, - pos: &EdgePos, - view: Option<&fluree_db_binary_index::BinaryGraphView>, -) -> Result> { - match pos { - EdgePos::Const(sid) => Ok(Some(sid.clone())), - EdgePos::Var(v) => match batch.get(row, *v) { - Some(Binding::Sid { sid, .. }) => Ok(Some(sid.clone())), - Some(Binding::EncodedPid { p_id }) => { - let view = view.ok_or_else(|| { - QueryError::execution( - "annotation edge probe: encoded predicate with no binary graph view", - ) - })?; - view.store() - .p_sid_table() - .get(*p_id as usize) - .cloned() - .map(Some) - .ok_or_else(|| { - QueryError::execution(format!( - "annotation edge probe: resolve encoded predicate {p_id}" - )) - }) - } - Some(Binding::Unbound | Binding::Poisoned) | None => Ok(None), - Some(other) => Err(QueryError::execution(format!( - "annotation edge probe: predicate position bound to non-Sid {other:?}" - ))), - }, - } -} - -/// Resolve an edge ref position to a concrete `Sid`. Handles the two -/// ref-valued binding representations a base-edge scan can emit: -/// eagerly-resolved `Sid` and late-materialized `EncodedSid`. The -/// latter is decoded **directly** through the subject dictionary -/// (`BinaryGraphView::resolve_subject_sid`) — an IRI round-trip -/// (`resolve_subject_iri` + `encode_iri`) silently returns `None` for -/// subjects whose IRI doesn't re-encode, which would drop rows -/// non-deterministically (a subject may arrive eager or late depending -/// on scan timing). A failure here is a loud error, never a dropped row. -pub(crate) fn resolve_pos_ref( - batch: &Batch, - row: usize, - pos: &EdgePos, - view: Option<&fluree_db_binary_index::BinaryGraphView>, -) -> Result> { - match pos { - EdgePos::Const(sid) => Ok(Some(sid.clone())), - EdgePos::Var(v) => match batch.get(row, *v) { - Some(Binding::Sid { sid, .. }) => Ok(Some(sid.clone())), - Some(Binding::EncodedSid { s_id, .. }) => { - let view = view.ok_or_else(|| { - QueryError::execution( - "annotation edge probe: encoded subject with no binary graph view", - ) - })?; - let sid = view.resolve_subject_sid(*s_id).map_err(|e| { - QueryError::execution(format!( - "annotation edge probe: resolve encoded subject {s_id}: {e}" - )) - })?; - Ok(Some(sid)) - } - Some(Binding::Unbound | Binding::Poisoned) | None => Ok(None), - // A non-ref binding in an edge ref position means the - // recognized shape's invariant was violated. Surface it - // loudly rather than silently dropping the row. - Some(other) => Err(QueryError::execution(format!( - "annotation edge probe: edge ref position bound to non-ref {other:?}" - ))), - }, - } -} - -/// Normalize a binding to a raw `Sid` — decoding late-materialized -/// subjects through the graph view and late-materialized predicates (an -/// untyped `-[p]->` base scan can emit `EncodedPid`) through the store's -/// predicate table. `None` for unbound/poisoned or non-ref bindings — -/// such a position matches no `f:reifies*` row. -pub(crate) fn binding_sid( - b: &Binding, - view: Option<&fluree_db_binary_index::BinaryGraphView>, -) -> Result> { - match b { - Binding::Sid { sid, .. } => Ok(Some(sid.clone())), - Binding::EncodedSid { s_id, .. } => { - let view = view.ok_or_else(|| { - QueryError::execution("annotation probe: encoded subject with no binary graph view") - })?; - view.resolve_subject_sid(*s_id).map(Some).map_err(|e| { - QueryError::execution(format!( - "annotation probe: resolve encoded subject {s_id}: {e}" - )) - }) - } - Binding::EncodedPid { p_id } => { - let view = view.ok_or_else(|| { - QueryError::execution( - "annotation probe: encoded predicate with no binary graph view", - ) - })?; - Ok(view.store().p_sid_table().get(*p_id as usize).cloned()) - } - _ => Ok(None), - } -} - -/// The three `f:reifies*` lookups, drained once through ordinary planned -/// scans (overlay-merged and policy-filtered like any scan) and indexed by -/// the full reified edge `(s, p, o)`. Shared by the value-only OPTIONAL lane -/// ([`crate::optional::AnnotationValueOptionalBuilder`]) and the required -/// lane ([`HashAnnotationEdgeProbeOperator`]). -/// -/// Keyed by the whole edge, not the subject alone: a subject-keyed map made -/// every lookup walk all reifiers sharing that subject and test each one's -/// predicate and object, which on a hub-heavy graph is a sum of squared -/// reified out-degrees (one StarBench BKR subject carries 3,345 `TREATS` -/// edges by itself) — the P11 timeout. -pub(crate) struct AnnotationSidecarMaps { - /// `s → p → o → live reifiers`. The object level is keyed by - /// [`GroupKeyOwned`] (representation-normalized), so ref objects AND - /// literal objects (a literal-valued reified edge) both match across - /// encoded/materialized binding representations. - edge_to_anns: HashMap>>>, -} - -impl AnnotationSidecarMaps { - /// Live reifiers of the edge `(s, p, o)`; empty when it has none. - pub(crate) fn anns_for(&self, s: &Sid, p: &Sid, o: &GroupKeyOwned) -> &[Sid] { - self.edge_to_anns - .get(s) - .and_then(|by_p| by_p.get(p)) - .and_then(|by_o| by_o.get(o)) - .map_or(&[], Vec::as_slice) - } - - /// Compose the per-slot `(reifier, value)` pairs into the edge-keyed - /// map. A reifier missing any slot attaches to nothing. A multi-target - /// reifier (legacy / replayed-from-corrupt-history ledgers) attaches to - /// every `subject × predicate × object` combination — the same answer - /// the per-slot containment test it replaces gave. - pub(crate) fn from_slot_pairs( - subj_pairs: Vec<(Sid, Sid)>, - pred_pairs: Vec<(Sid, Sid)>, - obj_pairs: Vec<(Sid, GroupKeyOwned)>, - ) -> Self { - let mut ann_subjs: HashMap> = HashMap::new(); - for (ann, s) in subj_pairs { - ann_subjs.entry(ann).or_default().push(s); - } - let mut ann_preds: HashMap> = HashMap::new(); - for (ann, p) in pred_pairs { - ann_preds.entry(ann).or_default().push(p); - } - let mut edge_to_anns: HashMap>>> = - HashMap::new(); - for (ann, o) in obj_pairs { - let (Some(subjs), Some(preds)) = (ann_subjs.get(&ann), ann_preds.get(&ann)) else { - continue; - }; - for s in subjs { - for p in preds { - edge_to_anns - .entry(s.clone()) - .or_default() - .entry(p.clone()) - .or_default() - .entry(o.clone()) - .or_default() - .push(ann.clone()); - } - } - } - Self { edge_to_anns } - } - - /// Drain the three reifies triples and build the lookup map. - pub(crate) async fn build( - r_subj: &TriplePattern, - r_pred: &TriplePattern, - r_obj: &TriplePattern, - stats: Option>, - planning: &PlanningContext, - ctx: &ExecutionContext<'_>, - view: Option<&fluree_db_binary_index::BinaryGraphView>, - ) -> Result { - let subj_pairs = drain_pairs(r_subj, stats.clone(), planning, ctx, view).await?; - let pred_pairs = drain_pairs(r_pred, stats.clone(), planning, ctx, view).await?; - let obj_pairs = drain_object_keys(r_obj, stats, planning, ctx, view).await?; - Ok(Self::from_slot_pairs(subj_pairs, pred_pairs, obj_pairs)) - } - - /// The drained maps for this shape, built once per execution context and - /// shared by every probe operator in the run that would drain the same - /// thing. - /// - /// This is the entry point; [`build`](Self::build) is the miss path. A - /// drain costs O(#annotations in the ledger) regardless of how many rows - /// the probe answers, so the operator is the wrong unit to cache it at: - /// one bounded variable-length Cypher range emits a probe operator per hop - /// of per chain (`*1..3` six, `*1..5` fifteen), and caching per operator - /// pays the whole sidecar once for each of them. - pub(crate) async fn shared( - r_subj: &TriplePattern, - r_pred: &TriplePattern, - r_obj: &TriplePattern, - stats: Option>, - planning: &PlanningContext, - ctx: &ExecutionContext<'_>, - view: Option<&fluree_db_binary_index::BinaryGraphView>, - ) -> Result> { - let key = AnnotationSidecarKey::new(r_subj, r_pred, r_obj, planning, ctx.binary_g_id); - if let Some(hit) = ctx.annotation_sidecar_cache.get(&key) { - return Ok(hit); - } - let built = Arc::new(Self::build(r_subj, r_pred, r_obj, stats, planning, ctx, view).await?); - // Two probes racing on a miss both drain; `publish` keeps whichever - // landed first so every caller ends up on one `Arc`. The maps are - // equal either way — the key is the whole of what a drain depends on. - Ok(ctx.annotation_sidecar_cache.publish(key, built)) - } -} - -/// The scan-visible identity of a sidecar drain: what two probe operators must -/// agree on for one drain's maps to answer both of them. -/// -/// The three reifies triples differ between probes mostly in their **variable -/// ids** — a bounded variable-length range mints a fresh reifier variable per -/// hop, over its own synthetic endpoint variables — and a variable contributes -/// nothing to what a drain returns. [`drain_pairs`] and [`drain_object_keys`] -/// plan each triple as a standalone *unseeded* single-pattern scan, so a -/// variable position filters nothing, and the maps they build are keyed by -/// `Sid` / [`GroupKeyOwned`], never by the variable a value was read out of. A -/// **constant** position is a real scan filter (`f:reifiesPredicate ` -/// for a typed relationship, an IRI-anchored endpoint), and so is a `dtc` — a -/// `LangTag` constraint narrows the object drain to one language. Erasing -/// variable identity and keeping everything else is therefore exactly the -/// equivalence the drain observes. -/// -/// Two things are deliberately absent because [`recognize_annotation_edge`] -/// fixes them by construction: the three predicates (checked to be -/// `f:reifies{Subject,Predicate,Object}`, so position carries them), and the -/// subject positions (checked to be one shared reifier variable, so the -/// sharing structure cannot differ between probes). -#[derive(Clone, PartialEq)] -pub(crate) struct AnnotationSidecarKey { - /// `(object position, datatype constraint)` for the `f:reifiesSubject`, - /// `f:reifiesPredicate` and `f:reifiesObject` triples in that order, with - /// a variable object collapsed to `None`. - filters: [(Option, Option); 3], - /// The graph the three drains scan. `with_active_graph` keeps the store - /// but can retarget `binary_g_id`, so keying on it makes a cross-graph - /// alias impossible outright, rather than resting on the structural reason - /// the value-only lane is confined to the default graph source today (it - /// requires the `DefaultGraphSource` wrapper, which expansion omits inside - /// an explicit `Pattern::Graph`). - g_id: fluree_db_core::GraphId, - /// Temporal planning mode: `is_history()` changes what a scan returns. - planning: PlanningContext, -} - -impl AnnotationSidecarKey { - fn new( - r_subj: &TriplePattern, - r_pred: &TriplePattern, - r_obj: &TriplePattern, - planning: &PlanningContext, - g_id: fluree_db_core::GraphId, - ) -> Self { - fn filter(t: &TriplePattern) -> (Option, Option) { - let o = match &t.o { - Term::Var(_) => None, - other => Some(other.clone()), - }; - (o, t.dtc.clone()) - } - Self { - filters: [filter(r_subj), filter(r_pred), filter(r_obj)], - g_id, - planning: *planning, - } - } -} - -/// Per-query-run memo of drained [`AnnotationSidecarMaps`], keyed by -/// [`AnnotationSidecarKey`]. -/// -/// Held on [`ExecutionContext`] and scoped exactly like -/// [`ConstSidCache`](crate::context::ConstSidCache): shared across -/// `with_active_graph` / `with_default_graph` derivations, which keep the -/// store, and reset by `with_graph_ref`, which swaps snapshot, overlay and -/// `to_t`. -/// -/// A `Vec` scanned linearly rather than a `HashMap`: one query has a handful -/// of distinct probe shapes at most (every hop of one range shares a key), and -/// the key holds a [`Term`], whose `Value` arm is a `FlakeValue` — `PartialEq` -/// but deliberately not `Hash`/`Eq`, since a float object has no lawful hash. -#[derive(Clone, Default)] -pub struct AnnotationSidecarCache(Arc>); - -/// The cache's entries: one per distinct drain the run has paid for. -type SidecarEntries = Vec<(AnnotationSidecarKey, Arc)>; - -impl AnnotationSidecarCache { - fn get(&self, key: &AnnotationSidecarKey) -> Option> { - let guard = self.0.lock().expect("annotation sidecar cache lock"); - guard - .iter() - .find(|(k, _)| k == key) - .map(|(_, maps)| maps.clone()) - } - - /// Record `maps` under `key`, returning whatever is canonical afterwards — - /// an entry another task published first, else `maps`. - fn publish( - &self, - key: AnnotationSidecarKey, - maps: Arc, - ) -> Arc { - let mut guard = self.0.lock().expect("annotation sidecar cache lock"); - if let Some((_, existing)) = guard.iter().find(|(k, _)| *k == key) { - return existing.clone(); - } - guard.push((key, maps.clone())); - maps - } -} - -/// The probe-side counterpart to [`drain_object_keys`]'s normalization: -/// one base-edge row's object position as a [`GroupKeyOwned`]. -/// `GroupKeyOwned::Absent` (never matched by the maps — absent pairs are -/// skipped at build) for unbound/poisoned positions. -pub(crate) fn row_obj_key( - batch: &Batch, - row: usize, - pos: &EdgePos, - view: Option<&fluree_db_binary_index::BinaryGraphView>, -) -> GroupKeyOwned { - let store = view.map(fluree_db_binary_index::BinaryGraphView::store); - match pos { - EdgePos::Const(sid) => { - binding_to_group_key_normalized(&Binding::sid(sid.clone()), store, view) - } - EdgePos::Var(v) => match batch.get(row, *v) { - Some(b) => binding_to_group_key_normalized(b, store, view), - None => GroupKeyOwned::Absent, - }, - } -} - -/// Drain one reifies triple's whole predicate through a planned scan, -/// yielding `(reifier Sid, object Sid)` pairs. Non-ref objects (a -/// literal-valued reified edge) are skipped — the probe only matches -/// ref positions, mirroring `EdgePos::from_term`. -async fn drain_pairs( - triple: &TriplePattern, - stats: Option>, - planning: &PlanningContext, - ctx: &ExecutionContext<'_>, - view: Option<&fluree_db_binary_index::BinaryGraphView>, -) -> Result> { - let ann_v = match &triple.s { - Ref::Var(v) => *v, - _ => { - return Err(QueryError::execution( - "annotation probe: reifies subject must be the reifier var", - )) - } - }; - let o_v = match &triple.o { - Term::Var(v) => Some(*v), - _ => None, - }; - let mut op = crate::execute::build_where_operators_seeded( - None, - std::slice::from_ref(&Pattern::Triple(triple.clone())), - stats, - None, - planning, - )?; - op.open(ctx).await?; - let mut out = Vec::new(); - while let Some(batch) = op.next_batch(ctx).await? { - ctx.check_cancelled()?; - for r in 0..batch.len() { - let Some(ann) = batch - .get(r, ann_v) - .map(|b| binding_sid(b, view)) - .transpose()? - .flatten() - else { - continue; - }; - let obj = match o_v { - Some(v) => batch - .get(r, v) - .map(|b| binding_sid(b, view)) - .transpose()? - .flatten(), - // Constant object (typed relationship / fixed endpoint): - // the scan already filtered to it; record the constant. - None => match &triple.o { - Term::Sid(sid) => Some(sid.clone()), - _ => None, - }, - }; - if let Some(obj) = obj { - out.push((ann, obj)); - } - } - } - op.close(); - Ok(out) -} - -/// Drain the `f:reifiesObject` triple, yielding `(reifier Sid, object -/// key)` pairs with the object normalized via -/// [`binding_to_group_key_normalized`] — ref objects and literal objects -/// (literal-valued reified edges) both key canonically. Absent objects -/// are skipped, so `GroupKeyOwned::Absent` never enters the maps. -async fn drain_object_keys( - triple: &TriplePattern, - stats: Option>, - planning: &PlanningContext, - ctx: &ExecutionContext<'_>, - view: Option<&fluree_db_binary_index::BinaryGraphView>, -) -> Result> { - let ann_v = match &triple.s { - Ref::Var(v) => *v, - _ => { - return Err(QueryError::execution( - "annotation probe: reifies subject must be the reifier var", - )) - } - }; - let o_v = match &triple.o { - Term::Var(v) => Some(*v), - _ => None, - }; - let store = view.map(fluree_db_binary_index::BinaryGraphView::store); - let mut op = crate::execute::build_where_operators_seeded( - None, - std::slice::from_ref(&Pattern::Triple(triple.clone())), - stats, - None, - planning, - )?; - op.open(ctx).await?; - let mut out = Vec::new(); - while let Some(batch) = op.next_batch(ctx).await? { - ctx.check_cancelled()?; - for r in 0..batch.len() { - let Some(ann) = batch - .get(r, ann_v) - .map(|b| binding_sid(b, view)) - .transpose()? - .flatten() - else { - continue; - }; - let obj = match o_v { - Some(v) => batch - .get(r, v) - .map(|b| binding_to_group_key_normalized(b, store, view)) - .unwrap_or(GroupKeyOwned::Absent), - // Constant object (typed relationship / fixed endpoint): - // the scan already filtered to it; record the constant. - None => match &triple.o { - Term::Sid(sid) => { - binding_to_group_key_normalized(&Binding::sid(sid.clone()), store, view) - } - _ => GroupKeyOwned::Absent, - }, - }; - if !matches!(obj, GroupKeyOwned::Absent) { - out.push((ann, obj)); - } - } - } - op.close(); - Ok(out) -} - -#[async_trait] -impl Operator for AnnotationEdgeProbeOperator { - fn schema(&self) -> &[VarId] { - &self.schema - } - - async fn open(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { - self.root = ctx.active_snapshot.annotation_index.clone(); - self.store = ctx.active_snapshot.content_store.clone(); - self.as_of_t = ctx.to_t; - self.child.open(ctx).await?; - self.state = OperatorState::Open; - self.probed = false; - self.result_buffer.clear(); - self.buffer_pos = 0; - Ok(()) - } - - async fn next_batch(&mut self, ctx: &ExecutionContext<'_>) -> Result> { - if self.state != OperatorState::Open { - return Ok(None); - } - if !self.probed { - self.probe_all(ctx).await?; - self.probed = true; - } - let out = self.drain_chunk(); - if out.is_none() { - self.state = OperatorState::Exhausted; - } - Ok(out) - } - - fn close(&mut self) { - self.child.close(); - self.result_buffer.clear(); - self.state = OperatorState::Closed; - } - - fn estimated_rows(&self) -> Option { - // ~1 reifier per base edge. - self.child.estimated_rows() - } -} - -impl AnnotationEdgeProbeOperator { - /// Drain the whole child stream once, probe the forward arena in a - /// single merge-scan (one reader → one branch/leaf decode), and fill - /// `result_buffer` with the fanned-out output rows. The base-edge - /// stream is bounded by the relationship's cardinality, so - /// materializing it is cheap relative to the per-batch reader rebuild - /// it replaces. - async fn probe_all(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { - // The reifier binding is appended as the final schema column — - // unless the child already binds it (nested reifier patterns like - // `?t :p :z . << ?s ?p ?o ~ ?t >>`), in which case candidates - // UNIFY with the existing column instead of appending. - let ann_col = self.child.schema().iter().position(|v| *v == self.ann_var); - debug_assert!(ann_col.is_some() || self.schema.last() == Some(&self.ann_var)); - let parent_schema_len = self.child.schema().len(); - - // Collect every child row's parent-schema bindings plus its - // EdgeKey (None → the row carries no probeable edge). - let mut saved_rows: Vec> = Vec::new(); - let mut edges: Vec = Vec::new(); - let mut edge_of_row: Vec> = Vec::new(); - // One subject-dictionary view for the whole pass — decoding an - // EncodedSid edge endpoint goes straight through it. - let view = ctx.graph_view(); - while let Some(batch) = self.child.next_batch(ctx).await? { - if batch.is_empty() { - continue; - } - for row in 0..batch.len() { - let mut rb = Vec::with_capacity(parent_schema_len); - for var in self.child.schema() { - rb.push(batch.get(row, *var).cloned().unwrap_or(Binding::Unbound)); - } - match self.edge_key_for_row(&batch, row, view.as_ref())? { - Some(ek) => { - edge_of_row.push(Some(edges.len())); - edges.push(ek); - } - None => edge_of_row.push(None), - } - saved_rows.push(rb); - } - } - - let (Some(root), Some(store)) = (self.root.as_ref(), self.store.as_ref()) else { - // Gates guarantee both are present; defensive only. - return Ok(()); - }; - let anns_per_edge = { - let reader = AnnotationArenaReader::new(root, store.as_ref()); - reader - .current_annotations_batch(&edges, self.as_of_t) - .await - .map_err(|e| { - QueryError::execution(format!("annotation forward arena probe: {e}")) - })? - }; - - for (i, rb) in saved_rows.into_iter().enumerate() { - let Some(edge_idx) = edge_of_row[i] else { - continue; - }; - let anns = &anns_per_edge[edge_idx]; - match ann_col { - // Fan out: one output row per live reifier. - None => { - for ann in anns { - let mut row = rb.clone(); - row.push(Binding::sid(ann.clone())); - self.result_buffer.push(row); - } - } - // Pre-bound reifier column: a bound value is a join key - // (keep only the matching candidate); an unbound one is - // filled per candidate. - Some(c) => { - let bound = match &rb[c] { - // Poisoned rows join nothing (nested-loop parity); - // a bound non-ref reifier matches no annotation. - Binding::Poisoned => continue, - Binding::Unbound => None, - b => match binding_sid(b, view.as_ref())? { - Some(sid) => Some(sid), - None => continue, - }, - }; - for ann in anns { - if bound.as_ref().is_some_and(|t| t != ann) { - continue; - } - let mut row = rb.clone(); - row[c] = Binding::sid(ann.clone()); - self.result_buffer.push(row); - } - } - } - } - Ok(()) - } - - /// Emit up to [`PROBE_OUTPUT_CHUNK`] buffered output rows as one batch. - fn drain_chunk(&mut self) -> Option { - if self.buffer_pos >= self.result_buffer.len() { - return None; - } - let end = (self.buffer_pos + PROBE_OUTPUT_CHUNK).min(self.result_buffer.len()); - let num_cols = self.schema.len(); - let mut columns: Vec> = (0..num_cols) - .map(|_| Vec::with_capacity(end - self.buffer_pos)) - .collect(); - for row in &self.result_buffer[self.buffer_pos..end] { - for (col, b) in row.iter().enumerate() { - if col < columns.len() { - columns[col].push(b.clone()); - } - } - } - self.buffer_pos = end; - if columns.is_empty() || columns[0].is_empty() { - return None; - } - Batch::new(self.schema.clone(), columns).ok() - } -} - -/// One base-edge row collected by the sweep: the normalized subject key it -/// hash-joins the driving stream on, the concrete Sids the sidecar maps -/// are probed with, and the original bindings for the pattern's VAR -/// positions (`None` for constants — those columns aren't in the schema). -struct SweptEdge { - s_sid: Sid, - p_sid: Sid, - o_key: GroupKeyOwned, - s_b: Option, - p_b: Option, - o_b: Option, -} - -/// Required-lane counterpart to [`AnnotationEdgeProbeOperator`] for ledgers -/// WITHOUT a sealed annotation arena (bulk-imported roots): instead of -/// per-row `f:reifies*` joins — whose planned chain drives from a -/// bound-object probe per driving row — drain the three reifies predicates -/// ONCE into [`AnnotationSidecarMaps`] and answer every base-edge row by -/// hash lookup. Rows with no matching reifier are dropped (the required -/// chain's semantics: an unreified edge matches no `f:reifies*` row). -/// -/// The base edge is NOT joined per driving row either: re-opening a scan -/// per row costs ~ms each (the KB `UNWIND` shape spent ~30 s over 1k rows -/// this way). Instead ONE unseeded planned scan of the base pattern runs -/// after the child drains, keeping only rows whose subject occurs in the -/// driving stream, and the surviving edges hash-join the driving rows. -/// The planner gates this operator on total ledger size (see -/// `build_single_graph_delegate`), so the sweep is bounded. -/// -/// All scans here (sidecar + base) are ordinary planned scans, so overlay -/// novelty and policy filtering apply — unlike the arena path, no -/// empty-overlay or root-policy gate is needed. -pub struct HashAnnotationEdgeProbeOperator { - child: BoxedOperator, - /// The base-edge triple, executed as ONE unseeded planned scan. - base: TriplePattern, - ann_var: VarId, - s_pos: EdgePos, - p_pos: EdgePos, - o_pos: EdgePos, - r_subj: TriplePattern, - r_pred: TriplePattern, - r_obj: TriplePattern, - stats: Option>, - planning: PlanningContext, - schema: Arc<[VarId]>, - state: OperatorState, - probed: bool, - result_buffer: Vec>, - buffer_pos: usize, -} - -impl HashAnnotationEdgeProbeOperator { - #[allow(clippy::too_many_arguments)] - pub(crate) fn new( - child: BoxedOperator, - base: TriplePattern, - ann_var: VarId, - s_pos: EdgePos, - p_pos: EdgePos, - o_pos: EdgePos, - r_subj: TriplePattern, - r_pred: TriplePattern, - r_obj: TriplePattern, - stats: Option>, - planning: PlanningContext, - ) -> Self { - // Child columns, then the base pattern's var positions the sweep - // produces, then the reifier. - let mut schema_vec: Vec = child.schema().to_vec(); - for pos in [&s_pos, &p_pos, &o_pos] { - if let EdgePos::Var(v) = pos { - if !schema_vec.contains(v) { - schema_vec.push(*v); - } - } - } - if !schema_vec.contains(&ann_var) { - schema_vec.push(ann_var); - } - let schema = Arc::from(schema_vec.into_boxed_slice()); - Self { - child, - base, - ann_var, - s_pos, - p_pos, - o_pos, - r_subj, - r_pred, - r_obj, - stats, - planning, - schema, - state: OperatorState::Created, - probed: false, - result_buffer: Vec::new(), - buffer_pos: 0, - } - } - - /// One pass over the base pattern via an unseeded planned scan, - /// keeping rows whose subject key occurs in `driving` (or every row - /// when `keep_all` — some driving row leaves the subject unbound). - async fn sweep_base_edges( - &self, - ctx: &ExecutionContext<'_>, - view: Option<&fluree_db_binary_index::BinaryGraphView>, - driving: &std::collections::HashSet, - keep_all: bool, - ) -> Result>> { - let store = view.map(fluree_db_binary_index::BinaryGraphView::store); - let mut op = crate::execute::build_where_operators_seeded( - None, - std::slice::from_ref(&Pattern::Triple(self.base.clone())), - self.stats.clone(), - None, - &self.planning, - )?; - let mut edges: HashMap> = HashMap::new(); - op.open(ctx).await?; - while let Some(batch) = op.next_batch(ctx).await? { - ctx.check_cancelled()?; - for row in 0..batch.len() { - let (s_key, s_b) = match &self.s_pos { - EdgePos::Const(sid) => ( - binding_to_group_key_normalized(&Binding::sid(sid.clone()), store, view), - None, - ), - EdgePos::Var(v) => { - let b = batch.get(row, *v).cloned().unwrap_or(Binding::Unbound); - (binding_to_group_key_normalized(&b, store, view), Some(b)) - } - }; - if matches!(s_key, GroupKeyOwned::Absent) - || (!keep_all && !driving.contains(&s_key)) - { - continue; - } - let Some(s_sid) = resolve_pos_ref(&batch, row, &self.s_pos, view)? else { - continue; - }; - let Some(p_sid) = resolve_pos_pred(&batch, row, &self.p_pos, view)? else { - continue; - }; - let o_key = row_obj_key(&batch, row, &self.o_pos, view); - if matches!(o_key, GroupKeyOwned::Absent) { - continue; - } - let var_binding = |pos: &EdgePos| match pos { - EdgePos::Const(_) => None, - EdgePos::Var(v) => { - Some(batch.get(row, *v).cloned().unwrap_or(Binding::Unbound)) - } - }; - let edge = SweptEdge { - s_sid, - p_sid, - o_key, - s_b, - p_b: var_binding(&self.p_pos), - o_b: var_binding(&self.o_pos), - }; - edges.entry(s_key).or_default().push(edge); - } - } - op.close(); - Ok(edges) - } - - /// Emit one output row: child bindings, then the base edge's var - /// positions, then the reifier. - fn emit_row(&mut self, child_batch: &Batch, child_row: usize, edge: &SweptEdge, ann: &Sid) { - let mut rb = Vec::with_capacity(self.schema.len()); - for var in self.schema.iter() { - let binding = if child_batch.schema().contains(var) { - let b = child_batch - .get(child_row, *var) - .cloned() - .unwrap_or(Binding::Unbound); - // An unbound driving position takes the join's value — - // the matched reifier for the ann var, the edge's value - // for a base position — like any join would. - if matches!(b, Binding::Unbound) { - if *var == self.ann_var { - Binding::sid(ann.clone()) - } else { - self.pos_binding_for_var(*var, edge) - .unwrap_or(Binding::Unbound) - } - } else { - b - } - } else if *var == self.ann_var { - Binding::sid(ann.clone()) - } else { - self.pos_binding_for_var(*var, edge) - .unwrap_or(Binding::Unbound) - }; - rb.push(binding); - } - self.result_buffer.push(rb); - } - - /// The swept edge's binding for a base-pattern var, if `var` is one of - /// its positions. - fn pos_binding_for_var(&self, var: VarId, edge: &SweptEdge) -> Option { - for (pos, b) in [ - (&self.s_pos, &edge.s_b), - (&self.p_pos, &edge.p_b), - (&self.o_pos, &edge.o_b), - ] { - if matches!(pos, EdgePos::Var(v) if *v == var) { - return b.clone(); - } - } - None - } - - /// Drain the whole child stream, build the sidecar maps and the - /// base-edge sweep once, and fill `result_buffer` with each driving - /// row hash-joined to its reified edges (rows with none are dropped). - /// The child drains first so an empty driving stream never pays for - /// either sweep. - async fn probe_all(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { - let view = ctx.graph_view(); - let view = view.as_ref(); - let store = view.map(fluree_db_binary_index::BinaryGraphView::store); - - let mut child_batches: Vec = Vec::new(); - let mut driving: std::collections::HashSet = - std::collections::HashSet::new(); - let mut keep_all = false; - while let Some(batch) = self.child.next_batch(ctx).await? { - ctx.check_cancelled()?; - if batch.is_empty() { - continue; - } - for row in 0..batch.len() { - match &self.s_pos { - EdgePos::Const(sid) => { - driving.insert(binding_to_group_key_normalized( - &Binding::sid(sid.clone()), - store, - view, - )); - } - EdgePos::Var(v) => match batch.get(row, *v) { - None | Some(Binding::Unbound) => keep_all = true, - Some(b) => { - driving.insert(binding_to_group_key_normalized(b, store, view)); - } - }, - } - } - child_batches.push(batch); - } - if child_batches.is_empty() { - return Ok(()); - } - - // Runtime lane choice. The plan-time gate admits this lane on an - // estimate that is often unknown; with the whole driving stream in - // hand, decide on what it actually holds. The three drains and the - // base sweep pay off only against MANY bound driving subjects, each - // of which would otherwise cost three scattered point probes. A - // handful of rows — a seed, or the matches of a selective sibling — - // runs the generic chain over the buffered rows instead: StarBench - // P1 swept the whole graph for 9 s to enrich 3 rows. - let child_rows: usize = child_batches.iter().map(Batch::len).sum(); - if child_rows < HASH_ANNOTATION_MIN_DRIVING_ROWS - || (!keep_all && driving.len() < HASH_ANNOTATION_MIN_DRIVING_ROWS) - { - tracing::debug!( - rows = child_rows, - driving = driving.len(), - keep_all, - "annotation required-lane: driving stream below the sweep threshold; running the generic chain over the buffered rows" - ); - return self.probe_generic(ctx, child_batches).await; - } - - let started = std::time::Instant::now(); - // Through the run-scoped memo, so a query that mixes the required - // lane with the value-only OPTIONAL lane (or with a bounded - // variable-length range) drains the sidecar once between them. - let maps = AnnotationSidecarMaps::shared( - &self.r_subj, - &self.r_pred, - &self.r_obj, - self.stats.clone(), - &self.planning, - ctx, - view, - ) - .await?; - let maps_ms = started.elapsed().as_millis(); - let edges = self.sweep_base_edges(ctx, view, &driving, keep_all).await?; - let swept: usize = edges.values().map(Vec::len).sum(); - tracing::debug!( - maps_ms, - sweep_ms = started.elapsed().as_millis() - maps_ms, - swept, - keep_all, - "annotation required-lane hash probe: maps built, base edges swept" - ); - - for batch in std::mem::take(&mut child_batches) { - ctx.check_cancelled()?; - for row in 0..batch.len() { - // A child-bound var position constrains the join like the - // per-row substitution it replaces: the edge must carry the - // SAME value there. Unbound positions are free — the edge - // binds them. - let row_key = |pos: &EdgePos| -> Option { - match pos { - EdgePos::Const(_) => None, // constrained by the scan pattern itself - EdgePos::Var(v) => match batch.get(row, *v) { - None | Some(Binding::Unbound) => None, - Some(b) => Some(binding_to_group_key_normalized(b, store, view)), - }, - } - }; - let s_key = row_key(&self.s_pos); - // Predicate compares in Sid space (a normalized key would - // straddle the predicate/subject id spaces). - let p_bound_sid = match &self.p_pos { - EdgePos::Const(_) => None, - EdgePos::Var(v) => match batch.get(row, *v) { - Some(Binding::Poisoned) => continue, - None | Some(Binding::Unbound) => None, - Some(b) => match binding_sid(b, view)? { - Some(sid) => Some(sid), - None => continue, - }, - }, - }; - let o_key = row_key(&self.o_pos); - let row_edges: Vec<&SweptEdge> = match &s_key { - // Unbound driving subject: every swept edge is a candidate. - None => edges.values().flatten().collect(), - Some(k) => edges.get(k).map(|v| v.iter().collect()).unwrap_or_default(), - }; - // A pre-bound reifier var (nested reifier patterns bind - // `?t` upstream: `?t :p :z . << ?s ?p ?o ~ ?t >>`) is a - // JOIN key like any other position — candidates must - // unify with it, not be overwritten by it. Without this, - // every candidate emitted a row stamped with the child's - // binding (W3C sparql12 eval-triple-terms pattern-8). - // Poisoned = a failed upstream OPTIONAL: join semantics - // drop the row (the nested-loop join skips poisoned rows) - // — never treat it as a wildcard. A bound non-ref reifier - // can match no annotation, so it too joins nothing. - let bound_ann: Option = match batch.get(row, self.ann_var) { - Some(Binding::Poisoned) => continue, - None | Some(Binding::Unbound) => None, - Some(b) => match binding_sid(b, view)? { - Some(sid) => Some(sid), - None => continue, - }, - }; - for edge in row_edges { - if p_bound_sid.as_ref().is_some_and(|ps| *ps != edge.p_sid) { - continue; - } - if o_key.as_ref().is_some_and(|k| *k != edge.o_key) { - continue; - } - for ann in maps.anns_for(&edge.s_sid, &edge.p_sid, &edge.o_key) { - if bound_ann.as_ref().is_some_and(|t| t != ann) { - continue; - } - self.emit_row(&batch, row, edge, ann); - } - } - } - } - tracing::debug!( - driving = driving.len(), - rows = self.result_buffer.len(), - total_ms = started.elapsed().as_millis(), - "annotation required-lane hash probe complete" - ); - Ok(()) - } - - /// Generic-chain fallback over the buffered driving rows: the base edge - /// and the three `f:reifies*` lookups plan as ordinary joins seeded by - /// the replayed child stream (same results as the sweep, per-row - /// probes instead), and each output row is re-shaped to this - /// operator's schema. - async fn probe_generic( - &mut self, - ctx: &ExecutionContext<'_>, - child_batches: Vec, - ) -> Result<()> { - let replay = Box::new(crate::seed::BatchReplayOperator::new( - Arc::from(self.child.schema().to_vec().into_boxed_slice()), - child_batches, - )); - let chain = [ - Pattern::Triple(self.base.clone()), - Pattern::Triple(self.r_subj.clone()), - Pattern::Triple(self.r_pred.clone()), - Pattern::Triple(self.r_obj.clone()), - ]; - let mut op = crate::execute::build_where_operators_seeded( - Some(replay), - &chain, - self.stats.clone(), - None, - &self.planning, - )?; - op.open(ctx).await?; - while let Some(batch) = op.next_batch(ctx).await? { - ctx.check_cancelled()?; - for row in 0..batch.len() { - let rb: Vec = self - .schema - .iter() - .map(|v| batch.get(row, *v).cloned().unwrap_or(Binding::Unbound)) - .collect(); - self.result_buffer.push(rb); - } - } - op.close(); - Ok(()) - } - - /// Emit up to [`PROBE_OUTPUT_CHUNK`] buffered output rows as one batch. - fn drain_chunk(&mut self) -> Option { - if self.buffer_pos >= self.result_buffer.len() { - return None; - } - let end = (self.buffer_pos + PROBE_OUTPUT_CHUNK).min(self.result_buffer.len()); - let num_cols = self.schema.len(); - let mut columns: Vec> = (0..num_cols) - .map(|_| Vec::with_capacity(end - self.buffer_pos)) - .collect(); - for row in &self.result_buffer[self.buffer_pos..end] { - for (col, b) in row.iter().enumerate() { - if col < columns.len() { - columns[col].push(b.clone()); - } - } - } - self.buffer_pos = end; - if columns.is_empty() || columns[0].is_empty() { - return None; - } - Batch::new(self.schema.clone(), columns).ok() - } -} - -#[async_trait] -impl Operator for HashAnnotationEdgeProbeOperator { - fn schema(&self) -> &[VarId] { - &self.schema - } - - async fn open(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { - self.child.open(ctx).await?; - self.state = OperatorState::Open; - self.probed = false; - self.result_buffer.clear(); - self.buffer_pos = 0; - Ok(()) - } - - async fn next_batch(&mut self, ctx: &ExecutionContext<'_>) -> Result> { - if self.state != OperatorState::Open { - return Ok(None); - } - if !self.probed { - self.probe_all(ctx).await?; - self.probed = true; - } - let out = self.drain_chunk(); - if out.is_none() { - self.state = OperatorState::Exhausted; - } - Ok(out) - } - - fn close(&mut self) { - self.child.close(); - self.result_buffer.clear(); - self.state = OperatorState::Closed; - } - - fn estimated_rows(&self) -> Option { - // ~1 reifier per base edge. - self.child.estimated_rows() - } -} - -/// Annotation-first enumeration: stream every live attachment out of the -/// forward arena and bind the reified edge's positions plus the reifier, -/// without scanning the base edge at all. -/// -/// The physical answer to `<< ?s ?p ?o >> :q ?x` (and to `<< ?s :P ?o >>` -/// when `:P` is wider than the arena). The edge-first lanes scan the base -/// edge and probe per row, which for a wildcard base edge means every flake -/// in the graph; the generic chain instead re-probes the base edge once per -/// reifier — 33.6M fully-bound point lookups on full-scale StarBench P2 -/// (757 s). Here the arena's reverse leaves are read once, in reifier order, one -/// leaf at a time, and each live `(edge, ann)` pair becomes a row per -/// driving row, encoded through the dictionaries like scan output. -/// -/// No base-edge visibility check is needed under this lane's gates (root -/// policy, current state, empty overlay): every write surface asserts the -/// base triple it annotates — Turtle-star ingest, JSON-LD `@annotation`, -/// SPARQL UPDATE — `@reifies`-rooted inserts are rejected, and a base-edge -/// retract cascades to its attachments, so a live arena row implies a live -/// default-graph edge. Named-graph attachments (`EdgeKey.g = Some`) are -/// skipped, as the arena probe lane skips them. -/// The dictionaries the enumeration lane encodes through. -struct Dicts<'a> { - store: Option<&'a fluree_db_binary_index::BinaryIndexStore>, - dict_novelty: Option<&'a fluree_db_core::dict_novelty::DictNovelty>, -} - -pub struct AnnotationEnumerateOperator { - child: BoxedOperator, - /// Constant relationship predicate to filter the walk by, or the - /// variable the walk binds; the schema records where it lands. - p_pos: EdgePos, - schema: Arc<[VarId]>, - state: OperatorState, - root: Option, - store: Option>, - as_of_t: i64, - /// Buffered driving rows (normally the one seed row); every live pair is - /// emitted once per driving row. `None` until the child is drained. - child_rows: Option>>, - /// Reverse-arena leaves still to walk, in reifier order. `None` until - /// the branch is read. - leaves: Option>, - /// Predicate `Sid → p_id` for this snapshot, so the walk emits - /// `EncodedPid` like a scan would (the table has a few hundred entries). - pred_ids: HashMap, - /// Subject-dictionary ids already resolved for edge subjects and - /// objects, which recur across the walk far more than they are distinct. - ref_ids: FxHashMap>, - result_buffer: Vec>, - buffer_pos: usize, - estimated: Option, -} - -impl AnnotationEnumerateOperator { - pub(crate) fn new( - child: BoxedOperator, - ann_var: VarId, - s_var: VarId, - p_pos: EdgePos, - o_var: VarId, - estimated: Option, - ) -> Self { - // Child columns, then the edge positions this lane binds, then the - // reifier — the order `emit` fills. - let mut schema_vec: Vec = child.schema().to_vec(); - schema_vec.push(s_var); - if let EdgePos::Var(v) = &p_pos { - schema_vec.push(*v); - } - schema_vec.push(o_var); - schema_vec.push(ann_var); - Self { - child, - p_pos, - schema: Arc::from(schema_vec.into_boxed_slice()), - state: OperatorState::Created, - root: None, - store: None, - as_of_t: 0, - child_rows: None, - leaves: None, - pred_ids: HashMap::new(), - ref_ids: FxHashMap::default(), - result_buffer: Vec::new(), - buffer_pos: 0, - estimated, - } - } - - /// A ref position as the scan would bind it: `EncodedSid` through the - /// persisted subject dictionary (or dictionary novelty), falling back to - /// the materialized `Sid` only when the subject is unknown to both — - /// which the lane's empty-overlay gate makes unexpected. Encoded rows - /// keep the downstream join and aggregate in id space: the materialized - /// form cost 40% of P2's time in `Arc` clone/drop churn. - fn ref_binding(sid: &Sid, dicts: &Dicts<'_>) -> Binding { - match Self::lookup_ref_id(sid, dicts) { - Some(s_id) => Binding::encoded_sid(s_id), - None => Binding::sid(sid.clone()), - } - } - - /// [`ref_binding`](Self::ref_binding) through the operator's memo, for - /// the edge positions whose nodes recur across the walk. Each miss is a - /// dictionary walk; reifiers are unique per edge and take the direct path. - fn memoized_ref_binding( - memo: &mut FxHashMap>, - sid: &Sid, - dicts: &Dicts<'_>, - ) -> Binding { - const MAX_ENTRIES: usize = 1 << 18; - let resolved = match memo.get(sid) { - Some(hit) => *hit, - None => { - let looked_up = Self::lookup_ref_id(sid, dicts); - if memo.len() >= MAX_ENTRIES { - memo.clear(); - } - memo.insert(sid.clone(), looked_up); - looked_up - } - }; - match resolved { - Some(s_id) => Binding::encoded_sid(s_id), - None => Binding::sid(sid.clone()), - } - } - - fn lookup_ref_id(sid: &Sid, dicts: &Dicts<'_>) -> Option { - let persisted = dicts.store.and_then(|st| { - st.find_subject_id_by_parts(sid.namespace_code, &sid.name) - .ok() - .flatten() - }); - persisted.or_else(|| { - dicts - .dict_novelty - .filter(|dn| dn.is_initialized()) - .and_then(|dn| dn.subjects.find_subject(sid.namespace_code, &sid.name)) - }) - } - - /// The reified object as a binding, exactly as the base scan would have - /// bound it: refs encoded, literals with the datatype and language tag - /// the arena stored from the original flake. - fn object_binding( - memo: &mut FxHashMap>, - edge: &EdgeKey, - dicts: &Dicts<'_>, - ) -> Binding { - match (&edge.o, &edge.lang) { - (FlakeValue::Ref(sid), _) => Self::memoized_ref_binding(memo, sid, dicts), - (val, Some(lang)) => Binding::lit_lang(val.clone(), lang.as_str()), - (val, None) => Binding::from_object(val.clone(), edge.dt.clone()), - } - } - - fn emit(&mut self, edge: &EdgeKey, ann: &Sid, dicts: &Dicts<'_>) { - let s_b = Self::memoized_ref_binding(&mut self.ref_ids, &edge.s, dicts); - let p_b = match self.p_pos { - EdgePos::Var(_) => Some(match self.pred_ids.get(&edge.p) { - Some(p_id) => Binding::EncodedPid { p_id: *p_id }, - None => Binding::sid(edge.p.clone()), - }), - EdgePos::Const(_) => None, - }; - let o_b = Self::object_binding(&mut self.ref_ids, edge, dicts); - let ann_b = Self::ref_binding(ann, dicts); - let child_rows = self.child_rows.as_deref().unwrap_or(&[]); - for child_row in child_rows { - let mut row: Vec = Vec::with_capacity(self.schema.len()); - row.extend(child_row.iter().cloned()); - row.push(s_b.clone()); - if let Some(p_b) = &p_b { - row.push(p_b.clone()); - } - row.push(o_b.clone()); - row.push(ann_b.clone()); - self.result_buffer.push(row); - } - } - - /// Emit up to [`PROBE_OUTPUT_CHUNK`] buffered output rows as one batch. - fn drain_chunk(&mut self) -> Option { - if self.buffer_pos >= self.result_buffer.len() { - return None; - } - let end = (self.buffer_pos + PROBE_OUTPUT_CHUNK).min(self.result_buffer.len()); - let mut columns: Vec> = (0..self.schema.len()) - .map(|_| Vec::with_capacity(end - self.buffer_pos)) - .collect(); - for row in &self.result_buffer[self.buffer_pos..end] { - for (col, b) in row.iter().enumerate() { - columns[col].push(b.clone()); - } - } - self.buffer_pos = end; - Batch::new(self.schema.clone(), columns).ok() - } -} - -#[async_trait] -impl Operator for AnnotationEnumerateOperator { - fn schema(&self) -> &[VarId] { - &self.schema - } - - async fn open(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { - self.root = ctx.active_snapshot.annotation_index.clone(); - self.store = ctx.active_snapshot.content_store.clone(); - self.as_of_t = ctx.to_t; - self.pred_ids = ctx - .graph_view() - .map(|view| { - view.store() - .p_sid_table() - .iter() - .enumerate() - .map(|(i, sid)| (sid.clone(), i as u32)) - .collect() - }) - .unwrap_or_default(); - self.child.open(ctx).await?; - self.state = OperatorState::Open; - self.child_rows = None; - self.leaves = None; - self.result_buffer.clear(); - self.buffer_pos = 0; - Ok(()) - } - - async fn next_batch(&mut self, ctx: &ExecutionContext<'_>) -> Result> { - if self.state != OperatorState::Open { - return Ok(None); - } - if self.child_rows.is_none() { - let mut rows: Vec> = Vec::new(); - while let Some(batch) = self.child.next_batch(ctx).await? { - for row in 0..batch.len() { - rows.push( - self.child - .schema() - .iter() - .map(|v| batch.get(row, *v).cloned().unwrap_or(Binding::Unbound)) - .collect(), - ); - } - } - if rows.is_empty() { - self.state = OperatorState::Exhausted; - return Ok(None); - } - self.child_rows = Some(rows); - } - let (Some(root), Some(store)) = (self.root.as_ref(), self.store.as_ref()) else { - // Gates guarantee both are present; defensive only. - self.state = OperatorState::Exhausted; - return Ok(None); - }; - if self.leaves.is_none() { - // Reverse arena: reifier order is dictionary order, so the - // encodes below and the batched probes above walk the subject - // dictionary and PSOT sequentially instead of at random. - let entries = AnnotationArenaReader::new(root, store.as_ref()) - .reverse_leaf_entries() - .await - .map_err(|e| QueryError::execution(format!("annotation arena walk: {e}")))?; - self.leaves = Some(entries.into_iter().map(|e| e.leaf_cid).collect()); - } - let view = ctx.graph_view(); - let dicts = Dicts { - store: view - .as_ref() - .map(fluree_db_binary_index::BinaryGraphView::store), - dict_novelty: ctx.dict_novelty.as_deref(), - }; - loop { - if let Some(batch) = self.drain_chunk() { - return Ok(Some(batch)); - } - ctx.check_cancelled()?; - let Some(cid) = self - .leaves - .as_mut() - .and_then(std::collections::VecDeque::pop_front) - else { - self.state = OperatorState::Exhausted; - return Ok(None); - }; - self.result_buffer.clear(); - self.buffer_pos = 0; - let pairs = { - let (Some(root), Some(store)) = (self.root.as_ref(), self.store.as_ref()) else { - self.state = OperatorState::Exhausted; - return Ok(None); - }; - AnnotationArenaReader::new(root, store.as_ref()) - .live_pairs_in_reverse_leaf(&cid, self.as_of_t) - .await - .map_err(|e| QueryError::execution(format!("annotation arena walk: {e}")))? - }; - for (edge, ann) in &pairs { - if edge.g.is_some() { - continue; - } - if let EdgePos::Const(p) = &self.p_pos { - if edge.p != *p { - continue; - } - } - self.emit(edge, ann, &dicts); - } - } - } - - fn close(&mut self) { - self.child.close(); - self.child_rows = None; - self.leaves = None; - self.result_buffer.clear(); - self.state = OperatorState::Closed; - } - - fn estimated_rows(&self) -> Option { - self.estimated - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::ir::TriplePattern; - use fluree_vocab::db::{REIFIES_OBJECT, REIFIES_PREDICATE, REIFIES_SUBJECT}; - use fluree_vocab::namespaces::FLUREE_DB; - - #[test] - fn enumerate_memo_caches_ref_lookups_and_their_misses() { - let dicts = Dicts { - store: None, - dict_novelty: None, - }; - let mut memo = FxHashMap::default(); - let alice = Sid::new(100, "alice"); - let direct = AnnotationEnumerateOperator::ref_binding(&alice, &dicts); - let first = AnnotationEnumerateOperator::memoized_ref_binding(&mut memo, &alice, &dicts); - let second = AnnotationEnumerateOperator::memoized_ref_binding(&mut memo, &alice, &dicts); - assert_eq!( - memo.len(), - 1, - "one entry per distinct node, misses included" - ); - assert_eq!(memo.get(&alice), Some(&None)); - for b in [&direct, &first, &second] { - assert!( - matches!(b, Binding::Sid { sid, .. } if *sid == alice), - "unknown nodes fall back to the materialized Sid: {b:?}" - ); - } - } - - fn v(n: u16) -> VarId { - VarId(n) - } - fn pred(name: &str) -> Sid { - Sid::new(FLUREE_DB, name) - } - fn user_sid(n: u16, name: &str) -> Sid { - Sid::new(n, name) - } - - /// Canonical expanded chain for `(friend)<-[m:HAS_MEMBER]-(forum)` - /// plus a `m.joinDate` body read, as `expand_edge_annotation_patterns` - /// would emit it: base edge, the three reifies triples, then body. - fn canonical_chain() -> Vec { - let forum = v(1); - let friend = v(2); - let ann = v(3); - let jd = v(4); - let has_member = Ref::Sid(user_sid(15, "HAS_MEMBER")); - let base = TriplePattern { - s: Ref::Var(forum), - p: has_member.clone(), - o: Term::Var(friend), - dtc: None, - }; - let r_subj = TriplePattern { - s: Ref::Var(ann), - p: Ref::Sid(pred(REIFIES_SUBJECT)), - o: Term::Var(forum), - dtc: None, - }; - let r_pred = TriplePattern { - s: Ref::Var(ann), - p: Ref::Sid(pred(REIFIES_PREDICATE)), - o: Term::from(has_member), - dtc: None, - }; - let r_obj = TriplePattern { - s: Ref::Var(ann), - p: Ref::Sid(pred(REIFIES_OBJECT)), - o: Term::Var(friend), - dtc: None, - }; - let body = TriplePattern { - s: Ref::Var(ann), - p: Ref::Sid(user_sid(16, "joinDate")), - o: Term::Var(jd), - dtc: None, - }; - vec![ - Pattern::Triple(base), - Pattern::Triple(r_subj), - Pattern::Triple(r_pred), - Pattern::Triple(r_obj), - Pattern::Triple(body), - ] - } - - #[test] - fn recognizes_canonical_edge_annotation_chain() { - let shape = recognize_annotation_edge(&canonical_chain()).expect("should recognize"); - assert_eq!(shape.ann_var, v(3)); - assert_eq!(shape.p_pred, Ref::Sid(user_sid(15, "HAS_MEMBER"))); - assert!(matches!(shape.s_pos, EdgePos::Var(x) if x == v(1))); - assert!(matches!(shape.o_pos, EdgePos::Var(x) if x == v(2))); - assert_eq!(shape.body.len(), 1, "joinDate read stays in body"); - } - - #[test] - fn recognizes_iri_predicate_the_way_cypher_lowers_it() { - // Cypher lowers a typed relationship to a `Ref::Iri` predicate; - // the reifiesPredicate triple's object is the same IRI. - let mut chain = canonical_chain(); - let iri: Arc = Arc::from("http://ldbc.example/HAS_MEMBER"); - if let Pattern::Triple(t) = &mut chain[0] { - t.p = Ref::Iri(iri.clone()); - } - if let Pattern::Triple(t) = &mut chain[2] { - t.o = Term::Iri(iri.clone()); - } - let shape = recognize_annotation_edge(&chain).expect("should recognize iri pred"); - assert_eq!(shape.p_pred, Ref::Iri(iri)); - } - - #[test] - fn rejects_when_reifies_objects_do_not_match_base_edge() { - let mut chain = canonical_chain(); - // Corrupt reifiesObject to point at the wrong var. - if let Pattern::Triple(t) = &mut chain[3] { - t.o = Term::Var(v(99)); - } - assert!(recognize_annotation_edge(&chain).is_none()); - } - - #[test] - fn rejects_variable_predicate() { - let mut chain = canonical_chain(); - if let Pattern::Triple(t) = &mut chain[0] { - t.p = Ref::Var(v(50)); - } - assert!(recognize_annotation_edge(&chain).is_none()); - } - - #[test] - fn rejects_mismatched_reifier_var() { - let mut chain = canonical_chain(); - // reifiesPredicate uses a different reifier subject var. - if let Pattern::Triple(t) = &mut chain[2] { - t.s = Ref::Var(v(77)); - } - assert!(recognize_annotation_edge(&chain).is_none()); - } - - #[test] - fn rejects_too_short_chain() { - let chain = canonical_chain(); - assert!(recognize_annotation_edge(&chain[..3]).is_none()); - } - - #[test] - fn annotation_sidecar_maps_preserve_multi_target_values() { - let ann = Sid::new(1, "ann"); - let s = Sid::new(3, "s"); - let p1 = Sid::new(2, "p1"); - let p2 = Sid::new(2, "p2"); - let ok = |s: &Sid| binding_to_group_key_normalized(&Binding::sid(s.clone()), None, None); - let o1 = ok(&Sid::new(3, "o1")); - let o2 = ok(&Sid::new(3, "o2")); - let maps = AnnotationSidecarMaps::from_slot_pairs( - vec![(ann.clone(), s.clone())], - vec![(ann.clone(), p1.clone()), (ann.clone(), p2.clone())], - vec![(ann.clone(), o1.clone()), (ann.clone(), o2.clone())], - ); - - assert_eq!(maps.anns_for(&s, &p1, &o1), std::slice::from_ref(&ann)); - assert_eq!(maps.anns_for(&s, &p2, &o2), std::slice::from_ref(&ann)); - // Every pred × obj combination, as the per-slot containment test did. - assert_eq!(maps.anns_for(&s, &p1, &o2), std::slice::from_ref(&ann)); - assert!(maps.anns_for(&Sid::new(3, "other"), &p1, &o1).is_empty()); - } - - #[test] - fn annotation_sidecar_maps_keep_same_subject_edges_apart() { - // Two reifiers on edges that share a subject must not see each - // other: the lookup is by the full (s, p, o) edge. - let (a1, a2) = (Sid::new(1, "a1"), Sid::new(1, "a2")); - let s = Sid::new(3, "hub"); - let p = Sid::new(2, "TREATS"); - let ok = |s: &Sid| binding_to_group_key_normalized(&Binding::sid(s.clone()), None, None); - let (o1, o2) = (ok(&Sid::new(3, "o1")), ok(&Sid::new(3, "o2"))); - let maps = AnnotationSidecarMaps::from_slot_pairs( - vec![(a1.clone(), s.clone()), (a2.clone(), s.clone())], - vec![(a1.clone(), p.clone()), (a2.clone(), p.clone())], - vec![(a1.clone(), o1.clone()), (a2.clone(), o2.clone())], - ); - assert_eq!(maps.anns_for(&s, &p, &o1), [a1]); - assert_eq!(maps.anns_for(&s, &p, &o2), [a2]); - } - - /// The three reifies triples of one hop of a bounded variable-length - /// range, with `hop` selecting the fresh variables the lowering mints per - /// hop and `rel_type` the constant relationship predicate. - fn hop_reifies(hop: u16, rel_type: &str) -> [TriplePattern; 3] { - let ann = v(100 + hop); - let start = v(200 + hop); - let end = v(200 + hop + 1); - [ - TriplePattern { - s: Ref::Var(ann), - p: Ref::Sid(pred(REIFIES_SUBJECT)), - o: Term::Var(start), - dtc: None, - }, - TriplePattern { - s: Ref::Var(ann), - p: Ref::Sid(pred(REIFIES_PREDICATE)), - o: Term::Sid(user_sid(15, rel_type)), - dtc: None, - }, - TriplePattern { - s: Ref::Var(ann), - p: Ref::Sid(pred(REIFIES_OBJECT)), - o: Term::Var(end), - dtc: None, - }, - ] - } - - fn key_for(hop: u16, rel_type: &str, g_id: fluree_db_core::GraphId) -> AnnotationSidecarKey { - let [r_subj, r_pred, r_obj] = hop_reifies(hop, rel_type); - AnnotationSidecarKey::new(&r_subj, &r_pred, &r_obj, &PlanningContext::default(), g_id) - } - - /// Every hop of one bounded range must land on ONE key, or the run-scoped - /// memo degenerates to the per-operator cache it replaced: `*1..5` plans - /// fifteen probe operators and each would re-drain the whole sidecar. - #[test] - fn sidecar_key_is_shared_across_the_hops_of_one_range() { - let first = key_for(0, "KNOWS", 0); - for hop in 1..15 { - assert!( - key_for(hop, "KNOWS", 0) == first, - "hop {hop} must share the first hop's drain" - ); - } - } - - /// ...and must NOT be shared where a position is a real scan filter. A - /// constant `f:reifiesPredicate` narrows the drain to one relationship - /// type, so serving `LIKES` from the `KNOWS` maps would answer a reified - /// edge as unreified. - #[test] - fn sidecar_key_separates_drains_that_filter_differently() { - let knows = key_for(0, "KNOWS", 0); - assert!( - key_for(0, "LIKES", 0) != knows, - "a different constant relationship type filters the predicate drain differently" - ); - assert!( - key_for(0, "KNOWS", 1) != knows, - "a different graph scans different flakes" - ); - - let [r_subj, r_pred, mut r_obj] = hop_reifies(0, "KNOWS"); - r_obj.dtc = Some(fluree_db_core::DatatypeConstraint::LangTag("en".into())); - let tagged = - AnnotationSidecarKey::new(&r_subj, &r_pred, &r_obj, &PlanningContext::default(), 0); - assert!( - tagged != knows, - "a language-tag constraint narrows the object drain to one language" - ); - } -} diff --git a/fluree-db-query/src/context.rs b/fluree-db-query/src/context.rs index f1e9263745..e486a4abc9 100644 --- a/fluree-db-query/src/context.rs +++ b/fluree-db-query/src/context.rs @@ -3,7 +3,6 @@ //! The `ExecutionContext` provides access to database state and configuration //! needed by operators during execution. -use crate::annotation_edge_probe::AnnotationSidecarCache; use crate::bm25::{Bm25IndexProvider, Bm25SearchProvider}; use crate::dataset::{ActiveGraph, ActiveGraphs, DataSet}; use crate::error::QueryError; @@ -412,17 +411,6 @@ pub struct ExecutionContext<'a> { /// as the multi-ledger dataset lane does. `None` everywhere else: the /// common paths pay one `Option` check per scan open. pub scan_provenance_ledger: Option>, - /// Per-query memo: drained `f:reifies*` sidecar maps, shared by every - /// edge-annotation probe operator in the run whose drain would return the - /// same thing (see - /// [`AnnotationSidecarCache`]). - /// - /// A drain is O(#annotations in the ledger) and independent of the result - /// size, and a bounded variable-length Cypher range emits one probe - /// operator per hop of per chain — `*1..3` six, `*1..5` fifteen. Caching - /// per operator therefore multiplies the whole sidecar by the hop count; - /// caching per run costs what one hop costs. - pub annotation_sidecar_cache: AnnotationSidecarCache, /// Per-query memo: constant filter operands → internal subject id, so a /// ` != ?var` FILTER resolves the constant once, not per row. pub const_sid_cache: ConstSidCache, @@ -518,7 +506,6 @@ impl<'a> ExecutionContext<'a> { reasoning_active: false, original_snapshot: snapshot, scan_provenance_ledger: None, - annotation_sidecar_cache: AnnotationSidecarCache::default(), const_sid_cache: ConstSidCache::default(), lang_tag_cache: LangTagCache::default(), subject_iri_cache: SubjectIriCache::default(), @@ -578,7 +565,6 @@ impl<'a> ExecutionContext<'a> { reasoning_active: false, original_snapshot: db.snapshot, scan_provenance_ledger: None, - annotation_sidecar_cache: AnnotationSidecarCache::default(), const_sid_cache: ConstSidCache::default(), lang_tag_cache: LangTagCache::default(), subject_iri_cache: SubjectIriCache::default(), @@ -642,7 +628,6 @@ impl<'a> ExecutionContext<'a> { reasoning_active: false, original_snapshot: db.snapshot, scan_provenance_ledger: None, - annotation_sidecar_cache: AnnotationSidecarCache::default(), const_sid_cache: ConstSidCache::default(), lang_tag_cache: LangTagCache::default(), subject_iri_cache: SubjectIriCache::default(), @@ -695,7 +680,6 @@ impl<'a> ExecutionContext<'a> { reasoning_active: false, original_snapshot: snapshot, scan_provenance_ledger: None, - annotation_sidecar_cache: AnnotationSidecarCache::default(), const_sid_cache: ConstSidCache::default(), lang_tag_cache: LangTagCache::default(), subject_iri_cache: SubjectIriCache::default(), @@ -747,7 +731,6 @@ impl<'a> ExecutionContext<'a> { reasoning_active: false, original_snapshot: snapshot, scan_provenance_ledger: None, - annotation_sidecar_cache: AnnotationSidecarCache::default(), const_sid_cache: ConstSidCache::default(), lang_tag_cache: LangTagCache::default(), subject_iri_cache: SubjectIriCache::default(), @@ -801,7 +784,6 @@ impl<'a> ExecutionContext<'a> { reasoning_active: false, original_snapshot: snapshot, scan_provenance_ledger: None, - annotation_sidecar_cache: AnnotationSidecarCache::default(), const_sid_cache: ConstSidCache::default(), lang_tag_cache: LangTagCache::default(), subject_iri_cache: SubjectIriCache::default(), @@ -1438,7 +1420,6 @@ impl<'a> ExecutionContext<'a> { reasoning_active: self.reasoning_active, original_snapshot: self.original_snapshot, scan_provenance_ledger: self.scan_provenance_ledger.clone(), - annotation_sidecar_cache: self.annotation_sidecar_cache.clone(), const_sid_cache: self.const_sid_cache.clone(), lang_tag_cache: self.lang_tag_cache.clone(), subject_iri_cache: self.subject_iri_cache.clone(), @@ -1501,7 +1482,6 @@ impl<'a> ExecutionContext<'a> { reasoning_active: self.reasoning_active, original_snapshot: self.original_snapshot, scan_provenance_ledger: self.scan_provenance_ledger.clone(), - annotation_sidecar_cache: self.annotation_sidecar_cache.clone(), const_sid_cache: self.const_sid_cache.clone(), lang_tag_cache: self.lang_tag_cache.clone(), subject_iri_cache: self.subject_iri_cache.clone(), @@ -1571,7 +1551,6 @@ impl<'a> ExecutionContext<'a> { // `const_sid_cache` is mandatory: its key is the const IRI ALONE // (store-implicit), so sharing the parent's would alias an s_id // resolved in one graph/store into another. - annotation_sidecar_cache: AnnotationSidecarCache::default(), const_sid_cache: ConstSidCache::default(), lang_tag_cache: LangTagCache::default(), subject_iri_cache: SubjectIriCache::default(), diff --git a/fluree-db-query/src/default_graph_source.rs b/fluree-db-query/src/default_graph_source.rs index c0391b0a44..7abfe92c45 100644 --- a/fluree-db-query/src/default_graph_source.rs +++ b/fluree-db-query/src/default_graph_source.rs @@ -1,14 +1,13 @@ //! Default-graph-source operator — iterate the dataset's default graph //! sources and run an inner subplan once per source. //! -//! This is a planner-internal construct. The -//! `expand_edge_annotation_patterns` pass synthesizes -//! [`Pattern::DefaultGraphSource`] around each expanded edge-annotation -//! triple chain so that under multi-source default-graph queries -//! (`from: [g1, g2]`), each source's base edge correlates only with -//! its own annotation flakes — without this wrapper the f:reifies* -//! lookups fan across all sources via `DatasetOperator` and produce -//! an N×M cross-product against each base-edge match. +//! This is a planner-internal construct. The edge-annotation expansion +//! synthesizes [`Pattern::DefaultGraphSource`] around an annotated edge's +//! chain (its body and the reifier's `rdf:reifies` link) when the default +//! graph is a union of two or more graphs (`from: [g1, g2]`), so each source's +//! link correlates only with its own body — without the wrapper the lookups +//! fan across all sources via `DatasetOperator` and pair a link from one +//! source with an annotation body from another. //! //! Distinct from [`crate::graph::GraphOperator`]: //! - `GraphOperator` implements SPARQL `GRAPH ?g { ... }` semantics — @@ -21,252 +20,23 @@ //! In single-source default-graph mode (no dataset attached) the //! wrapper is a no-op: [`DefaultGraphSourceOperator::open`] builds the //! inner subplan **once**, seeded by the whole child stream, and -//! streams it directly. This lets the base edge hash-join the child -//! instead of replanning + re-executing the inner subplan per parent -//! row — the latter made an annotated object-join O(parent rows) and -//! was the cause of IC5's timeout. The per-row, per-source path is -//! used only when a multi-source dataset is actually attached. +//! streams it directly. The per-row, per-source path is used only when +//! a multi-source dataset is actually attached. use crate::binding::{Batch, Binding}; use crate::context::ExecutionContext; use crate::error::Result; use crate::execute::build_where_operators_seeded; -use crate::ir::{Pattern, Ref}; +use crate::ir::Pattern; use crate::operator::{BoxedOperator, Operator, OperatorState}; use crate::seed::{EmptyOperator, SeedOperator}; use crate::temporal_mode::PlanningContext; use crate::var_registry::VarId; use async_trait::async_trait; -use fluree_db_binary_index::annotation_arena::DEFAULT_TARGET_ROWS_PER_LEAF; -use fluree_db_core::{Sid, StatsView}; +use fluree_db_core::StatsView; use std::collections::HashSet; use std::sync::Arc; -use crate::annotation_edge_probe::HASH_ANNOTATION_MIN_DRIVING_ROWS; - -/// Resolve a recognized relationship predicate ref to a concrete `Sid` -/// for this snapshot. Cypher lowers relationship types to `Ref::Iri`, so -/// the common case is an IRI encode; `None` means the predicate isn't -/// present in this ledger's namespace table (no edges → generic -/// fallback, which yields the same empty result more slowly). -fn resolve_pred_sid(p: &Ref, ctx: &ExecutionContext<'_>) -> Option { - match p { - Ref::Sid(sid) => Some(sid.clone()), - Ref::Iri(iri) => ctx.active_snapshot.encode_iri(iri), - Ref::Var(_) => None, - } -} - -/// The physical lane the delegate takes for a recognized chain. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -enum ChainLane { - /// Forward-arena probe: scan the base edge, probe the arena per row. - Arena, - /// Annotation-first: stream the reverse arena, no base scan. - Enumerate, - /// Per-reifier point probes (the hash lane or the generic chain). - Chain, -} - -/// Entry costs of one wrapper with the child's variables bound, in rows -/// walked per driving row (see `DefaultGraphSourceOperator::chain_lane`). -struct LaneInputs { - /// Base-edge rows an edge-first lane scans. - edge_first: f64, - /// Reifier candidates the chain's cheapest `f:reifies*` lookup yields. - probe_first: f64, - /// Live rows in the sealed arena (`f:reifiesSubject` count), if known. - arena_rows: Option, - /// The base-edge subject is a constant or bound by the child. - subject_bound: bool, - /// The child already binds the reifier. - reifier_bound: bool, - /// The chain runs without its redundant checks (see - /// [`elide_redundant_chain`]): one range scan plus sequential batched - /// probes instead of a point scan per reifier. - chain_elided: bool, -} - -/// One whole-leaf decode of the forward arena, in probed-row equivalents. -/// A leaf is a 4,096-row CBOR blob of string-keyed rows; decoding one costs -/// ~2.3 ms against ~6 µs per probed edge row (full StarBench ledger, where -/// 82% of P11's samples were leaf decode). -const LEAF_ROW_EQUIV: f64 = 400.0; - -/// Pick the lane by entry cost. One scattered point probe is worth about -/// `PROBE_ROW_EQUIV` sequential rows (the crossover measured on the -/// StarBench slice); the arena probe is never allowed to buffer more than -/// `BUFFERED_BASE_MAX_ROWS`, and it pays `LEAF_ROW_EQUIV` per leaf it -/// touches. -fn choose_chain_lane(inputs: LaneInputs) -> ChainLane { - const PROBE_ROW_EQUIV: f64 = 16.0; - // An elided chain drives from one `f:reifies*` range scan in reifier - // order and every later probe is a batched sequential walk: ~2 rows per - // reifier (P11 full scale: 1.1M reifiers, ~2 µs each, against ~6 µs per - // arena-probed edge row). - const SEQUENTIAL_ROW_EQUIV: f64 = 2.0; - // Each live pair the enumeration emits costs a CBOR row decode plus two - // dictionary re-encodes of the edge endpoints: ~6× an elided chain's row - // (P2 full scale: 65 s enumerating 21.4M pairs, 10.6 s on the elided - // chain), still far under the generic chain's point scan per reifier. - const ENUMERATE_ROW_EQUIV: f64 = 12.0; - const BUFFERED_BASE_MAX_ROWS: f64 = 20_000_000.0; - let LaneInputs { - edge_first, - probe_first, - arena_rows, - subject_bound, - reifier_bound, - chain_elided, - } = inputs; - // A bound reifier makes the chain one point probe per driving row; no - // sweep beats that (P22 / C7 / C10 regressed 200× on the arena here). - if reifier_bound { - return ChainLane::Chain; - } - let buffered = edge_first <= BUFFERED_BASE_MAX_ROWS; - // A bound subject keys into one arena leaf per driving row, while the - // chain's `f:reifiesSubject ` lookup walks every reifier of that - // subject — a hub tail the estimates cannot see (StarBench P23 went - // from 46 ms to 192 s on the chain). Keep the point probe. - if subject_bound { - return if buffered { - ChainLane::Arena - } else { - ChainLane::Chain - }; - } - // An unbound subject spreads the edges across the (s, p, o)-ordered - // arena, so the probe decodes about one leaf per edge up to the whole - // arena — a predicate-wide probe pays the entire arena whatever the - // predicate's size (P11 / S5 / C8 / C10 all cost the same 13–15 s). - let rows_per_leaf = DEFAULT_TARGET_ROWS_PER_LEAF as f64; - let total_leaves = arena_rows.map_or(edge_first, |rows| (rows / rows_per_leaf).max(1.0)); - let arena_cost = if buffered { - edge_first + edge_first.min(total_leaves) * LEAF_ROW_EQUIV - } else { - f64::INFINITY - }; - let per_reifier = if chain_elided { - SEQUENTIAL_ROW_EQUIV - } else { - PROBE_ROW_EQUIV - }; - let chain_cost = probe_first * per_reifier; - let enumerate_cost = arena_rows.map_or(f64::INFINITY, |rows| rows * ENUMERATE_ROW_EQUIV); - if arena_cost <= chain_cost && arena_cost <= enumerate_cost { - ChainLane::Arena - } else if enumerate_cost < chain_cost { - ChainLane::Enumerate - } else { - ChainLane::Chain - } -} - -/// Drop the chain triples the write invariants make redundant. Every -/// reifier carries exactly one `f:reifiesSubject`, `f:reifiesPredicate` and -/// `f:reifiesObject`, written into the edge's own graph, and its base triple -/// is asserted (`@reifies` without the base is rejected; retracting the base -/// cascades). So once a reifies lookup has bound the reifier, the base edge -/// never removes a row, and a reifies lookup whose position is a variable -/// nobody reads is a cardinality-one no-op. A constant position stays: it is -/// the constraint. A variable in two positions, or naming the reifier, counts -/// as read: its lookups carry the equality the base scan enforced -/// (`<< ?s :p ?s >>` must not match `:a :p :b`). A variable predicate that is -/// read stays too, and keeps -/// the base edge with it — the base scan binds it as a predicate, the -/// reifies lookup as a plain ref. At least one lookup always remains, so a -/// reifier bound by the body still has to be a reifier (P3 has three -/// `:derives_from` rows on plain subjects that P2 must not count). -/// -/// Measured on the full StarBench ledger this turns the predicate-only chain -/// from one point scan per reifier (P11: 1.1M SPOT opens) into one POST -/// range plus the body's batched probe. -pub(crate) fn elide_redundant_chain( - patterns: &[Pattern], - needed_outside: &HashSet, - child_bound: &HashSet, -) -> Option> { - let shape = crate::annotation_edge_probe::recognize_annotation_edge(patterns)?; - let Pattern::Triple(base) = &shape.base else { - return None; - }; - let mut referenced: HashSet = needed_outside.clone(); - referenced.extend(child_bound.iter().copied()); - let mut counts: std::collections::HashMap = std::collections::HashMap::new(); - crate::execute::collect_var_stats(&shape.body, &mut counts, &mut referenced); - // A variable in two base positions, or naming the reifier, is an - // equality the base scan enforced; keeping its lookups keeps it. - let mut seen: HashSet = HashSet::from([shape.ann_var]); - for v in [base.s.as_var(), base.p.as_var(), base.o.as_var()] - .into_iter() - .flatten() - { - if !seen.insert(v) { - referenced.insert(v); - } - } - - if base.p.as_var().is_some_and(|v| referenced.contains(&v)) { - return None; - } - // A constant position is a constraint; a variable position is one only - // when something reads it. - let position_kept = |var: Option| var.is_none_or(|v| referenced.contains(&v)); - let keep = [ - position_kept(base.s.as_var()), - position_kept(base.p.as_var()), - position_kept(base.o.as_var()), - ]; - let mut kept: Vec = patterns[1..=3] - .iter() - .zip(keep) - .filter(|(_, keep)| *keep) - .map(|(p, _)| p.clone()) - .collect(); - if kept.is_empty() { - kept.push(patterns[1].clone()); - } - kept.extend(shape.body.iter().cloned()); - Some(kept) -} - -/// Whether the hash sidecar lane may take a recognized chain. It drains the -/// three `f:reifies*` predicates and sweeps the base edge before answering a -/// row, which beats per-row probes only against a large or unknown driving -/// stream and a bounded sweep. A child that already binds the reifier keeps -/// the chain, as in [`choose_chain_lane`]: the chain is then one point probe -/// per row, while the lane walks every swept edge per row when the edge -/// subject is unbound. A constant annotation value driving -/// `<< ?s :p ?o >> :q "v"` went from 22 s to 55 s on a 20k-edge ledger that way. -fn hash_lane_admits(driving_rows: Option, sweep_bounded: bool, reifier_bound: bool) -> bool { - !reifier_bound - && sweep_bounded - && driving_rows.is_none_or(|n| n >= HASH_ANNOTATION_MIN_DRIVING_ROWS) -} - -/// Diagnostic override: `FLUREE_ANNOTATION_LANE=arena|enumerate|chain` pins -/// the lane regardless of cost so the same query can be timed on the same -/// ledger per lane. The runtime gates still apply — a forced arena or -/// enumeration lane without a sealed arena falls through to the chain. -fn forced_chain_lane() -> Option { - match std::env::var("FLUREE_ANNOTATION_LANE").ok()?.as_str() { - "arena" => Some(ChainLane::Arena), - "enumerate" => Some(ChainLane::Enumerate), - "chain" => Some(ChainLane::Chain), - _ => None, - } -} - -/// What the annotation-first enumeration lane binds (see -/// `DefaultGraphSourceOperator::enumeration_plan`). -struct EnumerationPlan { - s_var: VarId, - p_pos: crate::annotation_edge_probe::EdgePos, - o_var: VarId, - estimated: Option, -} - pub struct DefaultGraphSourceOperator { child: BoxedOperator, inner_patterns: Vec, @@ -275,26 +45,18 @@ pub struct DefaultGraphSourceOperator { result_buffer: Vec>, buffer_pos: usize, planning: PlanningContext, - /// Planner stats for the inner subplan build. Without these the base edge - /// cannot be costed and falls back to a per-driving-row object scan of the - /// whole edge predicate instead of an object→subject hash join — the inner - /// build previously passed `None`, which is what kept the annotated - /// `HAS_MEMBER` join slow even once it was built once. + /// Planner stats for the inner subplan build, so its patterns are costed. stats: Option>, /// Single default-graph fast path: when no dataset is attached the /// per-source correlation is unnecessary, so the inner subplan is built - /// ONCE seeded by the whole child stream (base edge can hash-join) and - /// streamed directly, instead of replanning + re-executing per parent row. + /// ONCE seeded by the whole child stream and streamed directly, instead of + /// replanning + re-executing per parent row. single_graph_delegate: Option, /// Planner estimate of the wrapped chain's output, with the child's /// variables bound — the same figure `reorder_patterns` placed the /// wrapper by. Reported through `estimated_rows` so the join above sees /// a real driving count instead of the `None` that read as one row. estimated: Option, - /// Variables read outside the wrapper (later siblings and the post-WHERE - /// pipeline); an inner position bound to none of them can drop its - /// `f:reifies*` lookup (see [`elide_redundant_chain`]). - needed_outside: HashSet, } impl DefaultGraphSourceOperator { @@ -303,18 +65,14 @@ impl DefaultGraphSourceOperator { inner_patterns: Vec, planning: PlanningContext, stats: Option>, - needed_outside: HashSet, ) -> Self { - let mut seen: std::collections::HashSet = child.schema().iter().copied().collect(); + let mut seen: HashSet = child.schema().iter().copied().collect(); // New vars in deterministic first-occurrence order across the inner // patterns. A `HashSet` here makes the output column order vary per // process (HashSet iteration is seeded randomly), and the inner // subplan emits batches in pattern order — the mismatch silently - // dropped whole result batches ~half the time. Pattern order is what - // the inner subplan (generic chain or the edge-annotation probe) - // actually produces, so the reported schema and the emitted batches - // agree. + // dropped whole result batches ~half the time. let mut new_vars: Vec = Vec::new(); for p in &inner_patterns { for v in p.produced_vars() { @@ -329,18 +87,11 @@ impl DefaultGraphSourceOperator { let schema = Arc::from(schema_vec.into_boxed_slice()); let bound: HashSet = child.schema().iter().copied().collect(); - let estimated = crate::planner::estimate_annotation_chain_cardinality( + let estimated = crate::planner::estimate_branch_cardinality_from( &inner_patterns, &bound, stats.as_deref(), - ) - .unwrap_or_else(|| { - crate::planner::estimate_branch_cardinality_from( - &inner_patterns, - &bound, - stats.as_deref(), - ) - }); + ); let estimated = Some(estimated.round().max(1.0) as usize); Self { @@ -354,359 +105,9 @@ impl DefaultGraphSourceOperator { stats, single_graph_delegate: None, estimated, - needed_outside, } } - /// Build the single-graph inner subplan. When the chain is a - /// recognized edge-annotation shape and every fast-path gate holds, - /// replace the three generic `f:reifies*` joins with a forward-arena - /// probe (the physical counterpart to a Cypher relationship binding); - /// otherwise fall back to the ordinary join chain — same results, - /// just the slower generic path. - fn build_single_graph_delegate( - &self, - child: BoxedOperator, - ctx: &ExecutionContext<'_>, - ) -> Result { - tracing::debug!( - arena = ctx.active_snapshot.annotation_index.is_some(), - store = ctx.active_snapshot.content_store.is_some(), - history = self.planning.is_history(), - overlay_empty = ctx.overlay().is_effectively_empty(), - root_policy = ctx.policy_enforcer.as_ref().is_none_or(|p| p.is_root()), - multi_ledger = ctx.is_multi_ledger(), - driving_est = ?child.estimated_rows(), - recognized = crate::annotation_edge_probe::recognize_annotation_edge( - &self.inner_patterns - ) - .is_some(), - "annotation delegate gates" - ); - let child_bound: HashSet = child.schema().iter().copied().collect(); - let elided = if self.elision_gates_pass(ctx) { - elide_redundant_chain(&self.inner_patterns, &self.needed_outside, &child_bound) - } else { - None - }; - let lane = self.chain_lane(&child, elided.is_some()); - if self.annotation_probe_gates_pass(ctx) && lane == ChainLane::Arena { - if let Some(shape) = - crate::annotation_edge_probe::recognize_annotation_edge(&self.inner_patterns) - { - // Resolve the relationship predicate: a typed relationship - // is a concrete IRI/Sid (encode for this snapshot — if it - // can't be encoded the predicate has no data here, fall - // back to the generic chain rather than guess); an untyped - // `-[p]->` is a variable the base scan binds per row. - let p_pos = match &shape.p_pred { - Ref::Var(v) => Some(crate::annotation_edge_probe::EdgePos::Var(*v)), - pred => resolve_pred_sid(pred, ctx) - .map(crate::annotation_edge_probe::EdgePos::Const), - }; - if let Some(p_pos) = p_pos { - // Base edge plans normally (visibility + policy), seeded - // by the whole child stream. - let base = build_where_operators_seeded( - Some(child), - std::slice::from_ref(&shape.base), - self.stats.clone(), - None, - &self.planning, - )?; - let probe = Box::new( - crate::annotation_edge_probe::AnnotationEdgeProbeOperator::new( - base, - shape.ann_var, - shape.s_pos, - p_pos, - shape.o_pos, - ), - ); - // Body (relationship-property reads, filters) plans - // normally on top, with the reifier var now bound. - tracing::debug!(lane = "arena", "annotation delegate lane"); - return build_where_operators_seeded( - Some(probe), - &shape.body, - self.stats.clone(), - None, - &self.planning, - ); - } - } - } - // Annotation-first enumeration: the arena lane was declined because - // the base edge is the wider entry point (a wildcard, or a predicate - // larger than the arena). Stream the arena instead of scanning the - // base edge — see `AnnotationEnumerateOperator` for why no base-edge - // check is needed under these gates. - if self.annotation_probe_gates_pass(ctx) && lane == ChainLane::Enumerate { - if let Some(shape) = - crate::annotation_edge_probe::recognize_annotation_edge(&self.inner_patterns) - { - if let Some(plan) = self.enumeration_plan(&shape, &child, ctx) { - let op = Box::new( - crate::annotation_edge_probe::AnnotationEnumerateOperator::new( - child, - shape.ann_var, - plan.s_var, - plan.p_pos, - plan.o_var, - plan.estimated, - ), - ); - tracing::debug!(lane = "enumerate", "annotation delegate lane"); - return build_where_operators_seeded( - Some(op), - &shape.body, - self.stats.clone(), - None, - &self.planning, - ); - } - } - } - // Required-lane hash fallback: no sealed arena (bulk-imported and - // reindexed-without-annotations roots) or an arena gate failed. The - // generic chain evaluates the base edge and the three `f:reifies*` - // joins per driving row; for a large driving stream (a 21k-row - // UNWIND) that is tens of thousands of scattered scan re-opens. - // Drain the sidecar and sweep the base pattern ONCE instead and - // answer each row by hash lookup — but only when the driving - // stream is large or unknown (below the threshold the per-row - // probes beat the sweeps) AND the base sweep is bounded (an - // untyped relationship sweeps the whole default graph once, so - // very large ledgers keep the per-row chain). - if self.hash_annotation_gates_pass(ctx) { - if let Some(shape) = - crate::annotation_edge_probe::recognize_annotation_edge(&self.inner_patterns) - { - let p_pos = match &shape.p_pred { - Ref::Var(v) => Some(crate::annotation_edge_probe::EdgePos::Var(*v)), - pred => resolve_pred_sid(pred, ctx) - .map(crate::annotation_edge_probe::EdgePos::Const), - }; - let admitted = hash_lane_admits( - child.estimated_rows(), - self.base_sweep_bounded(&shape), - child_bound.contains(&shape.ann_var), - ); - if let ( - Some(p_pos), - true, - Pattern::Triple(base_tp), - Pattern::Triple(r_subj), - Pattern::Triple(r_pred), - Pattern::Triple(r_obj), - ) = ( - p_pos, - admitted, - &self.inner_patterns[0], - &self.inner_patterns[1], - &self.inner_patterns[2], - &self.inner_patterns[3], - ) { - let probe = Box::new( - crate::annotation_edge_probe::HashAnnotationEdgeProbeOperator::new( - child, - base_tp.clone(), - shape.ann_var, - shape.s_pos, - p_pos, - shape.o_pos, - r_subj.clone(), - r_pred.clone(), - r_obj.clone(), - self.stats.clone(), - self.planning, - ), - ); - tracing::debug!(lane = "hash", "annotation delegate lane"); - return build_where_operators_seeded( - Some(probe), - &shape.body, - self.stats.clone(), - None, - &self.planning, - ); - } - } - } - tracing::debug!( - lane = "generic", - elided = elided.is_some(), - "annotation delegate lane" - ); - build_where_operators_seeded( - Some(child), - elided.as_deref().unwrap_or(&self.inner_patterns), - self.stats.clone(), - None, - &self.planning, - ) - } - - /// The write invariants [`elide_redundant_chain`] relies on describe - /// current state under full visibility: a history walk replays events - /// the cascade later undid, a non-root policy may hide the base triple - /// but not its reifier, and a dataset correlates sources per row. - fn elision_gates_pass(&self, ctx: &ExecutionContext<'_>) -> bool { - !self.planning.is_history() - && !ctx.is_multi_ledger() - && ctx.policy_enforcer.as_ref().is_none_or(|p| p.is_root()) - } - - /// Which physical lane the chain's entry costs favour. The three lanes - /// scale with different things: the forward-arena probe with the base - /// edge (it drains the base scan into memory and merge-probes the - /// arena), the enumeration with the whole arena (every live pair is - /// walked whatever the predicate), and the generic chain with the - /// number of reifiers it must point-probe (three scattered lookups - /// each). The reifier count is capped by the base-edge count — a reifier - /// needs an edge — so the per-predicate NDV cannot put `TREATS` at 31k - /// reifiers when it has 1.1M, and a child that already binds the - /// reifier makes the chain a per-row probe no sweep can beat (P22 / C7 - /// / C10 regressed 200× when the arena lane was taken there). - fn chain_lane(&self, child: &BoxedOperator, chain_elided: bool) -> ChainLane { - if let Some(forced) = forced_chain_lane() { - return forced; - } - let bound: HashSet = child.schema().iter().copied().collect(); - let stats = self.stats.as_deref(); - let Some(shape) = - crate::annotation_edge_probe::recognize_annotation_edge(&self.inner_patterns) - else { - return ChainLane::Arena; - }; - let Some((edge_first, _)) = - crate::planner::annotation_chain_entry_rows(&self.inner_patterns, &bound, stats) - else { - return ChainLane::Arena; - }; - let Some(probe_first) = - crate::planner::annotation_chain_probe_rows(&self.inner_patterns, &bound, stats) - else { - return ChainLane::Arena; - }; - let subject_bound = match &shape.base { - Pattern::Triple(tp) => match &tp.s { - Ref::Var(v) => bound.contains(v), - _ => true, - }, - _ => true, - }; - let arena_rows = stats.and_then(|s| { - s.get_property(&Sid::new( - fluree_vocab::namespaces::FLUREE_DB, - fluree_vocab::db::REIFIES_SUBJECT, - )) - .map(|p| p.count as f64) - }); - choose_chain_lane(LaneInputs { - edge_first, - probe_first, - arena_rows, - subject_bound, - reifier_bound: bound.contains(&shape.ann_var), - chain_elided, - }) - } - - /// Eligibility for the annotation-first enumeration: both base-edge - /// endpoints are variables the child leaves unbound (so an edge-first - /// scan would sweep a whole predicate or the whole graph), the child - /// binds none of the chain's variables (each live pair simply fans out - /// per driving row), and the predicate is a variable or resolves for - /// this snapshot. The row estimate is the reifier-first entry cost the - /// planner already computed for the wrapper. - fn enumeration_plan( - &self, - shape: &crate::annotation_edge_probe::AnnotationEdgeShape, - child: &BoxedOperator, - ctx: &ExecutionContext<'_>, - ) -> Option { - use crate::annotation_edge_probe::EdgePos; - let (EdgePos::Var(s_var), EdgePos::Var(o_var)) = (&shape.s_pos, &shape.o_pos) else { - return None; - }; - let p_pos = match &shape.p_pred { - Ref::Var(v) => EdgePos::Var(*v), - pred => EdgePos::Const(resolve_pred_sid(pred, ctx)?), - }; - let child_vars: HashSet = child.schema().iter().copied().collect(); - let p_var = match &p_pos { - EdgePos::Var(v) => Some(*v), - EdgePos::Const(_) => None, - }; - if child_vars.contains(s_var) - || child_vars.contains(o_var) - || child_vars.contains(&shape.ann_var) - || p_var.is_some_and(|v| child_vars.contains(&v)) - { - return None; - } - let estimated = crate::planner::annotation_chain_entry_rows( - &self.inner_patterns, - &child_vars, - self.stats.as_deref(), - ) - .map(|(_, reifier_first)| reifier_first.round().max(1.0) as usize); - Some(EnumerationPlan { - s_var: *s_var, - p_pos, - o_var: *o_var, - estimated, - }) - } - - /// Eligibility for the required-lane hash sidecar probe. Unlike the - /// arena path, the drained `f:reifies*` scans are ordinary planned - /// scans — overlay novelty and policy filtering apply — so neither an - /// empty overlay nor root policy is required. History timelines change - /// per-row visibility (the maps are one `to_t` snapshot of a planned - /// scan, which history mode plans differently), and multi-ledger - /// contexts have no single sidecar — both keep the generic chain. - fn hash_annotation_gates_pass(&self, ctx: &ExecutionContext<'_>) -> bool { - !self.planning.is_history() && !ctx.is_multi_ledger() - } - - /// Is the probe's one-pass base-edge sweep bounded? A typed - /// relationship sweeps one predicate partition (its stats count); an - /// untyped one sweeps the whole default graph, bounded by the summed - /// property counts. Stats absent → the ledger has no built index - /// (novelty-scale) — trivially bounded. - fn base_sweep_bounded( - &self, - shape: &crate::annotation_edge_probe::AnnotationEdgeShape, - ) -> bool { - // Backstop against pathological graphs, not a tuned optimum: ~100x - // the validated scale (~190k reified edges / 72k nodes). See - // docs/design/edge-annotations.md ("Buffering and the sweep - // ceiling") for the derivation and what to measure to replace it. - const BASE_SWEEP_MAX_ROWS: u64 = 20_000_000; - let Some(stats) = self.stats.as_deref() else { - return true; - }; - let rows = match &shape.p_pred { - Ref::Var(_) => stats.total_property_flakes(), - Ref::Sid(sid) => stats.get_property(sid).map_or(0, |p| p.count), - Ref::Iri(iri) => stats.get_property_by_iri(iri).map_or(0, |p| p.count), - }; - rows <= BASE_SWEEP_MAX_ROWS - } - - /// Eligibility for the forward-arena probe fast path. All checked - /// against the live execution context so no plan-time vouch is - /// needed. See `annotation_edge_probe` for why each matters. - fn annotation_probe_gates_pass(&self, ctx: &ExecutionContext<'_>) -> bool { - ctx.active_snapshot.annotation_index.is_some() - && ctx.active_snapshot.content_store.is_some() - && !self.planning.is_history() - && ctx.overlay().is_effectively_empty() - && ctx.policy_enforcer.as_ref().is_none_or(|p| p.is_root()) - } - /// Run the inner subplan against a single source graph and merge /// each output row with the parent row. async fn execute_in_source( @@ -820,17 +221,21 @@ impl Operator for DefaultGraphSourceOperator { } async fn open(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { - // (see build_single_graph_delegate for the recognition/fallback split) - // Single default-graph (no dataset): the per-source correlation this + // Single default graph (no dataset): the per-source correlation this // wrapper exists for is a no-op, so build the inner subplan ONCE seeded - // by the whole child stream. The base edge + f:reifies* triples then plan - // as one normal join block — the base edge can hash-join the child — and - // stream directly, instead of replanning and re-executing per parent row - // (which made an annotated object-join O(parent rows): IC5's 65s cliff). - // Multi-source datasets keep the per-row, per-source path below. + // by the whole child stream, and stream it, instead of replanning and + // re-executing per parent row (which made an annotated object-join + // O(parent rows): IC5's 65s cliff). Multi-source datasets keep the + // per-row, per-source path below. if ctx.dataset.is_none() { let child = std::mem::replace(&mut self.child, Box::new(EmptyOperator::new())); - let mut delegate = self.build_single_graph_delegate(child, ctx)?; + let mut delegate = build_where_operators_seeded( + Some(child), + &self.inner_patterns, + self.stats.clone(), + None, + &self.planning, + )?; delegate.open(ctx).await?; self.single_graph_delegate = Some(delegate); } else { @@ -849,7 +254,7 @@ impl Operator for DefaultGraphSourceOperator { // Single-graph fast path: stream the once-built inner subplan directly. // The delegate's batch column order comes from the REORDERED inner - // chain (plus whichever probe lane fired), which need not match this + // chain, which need not match this // operator's declared schema — re-project so positional consumers // above (NestedLoopJoin bind instructions read by column index) see // the contract they were planned against. Matching layouts pass @@ -941,390 +346,9 @@ impl Operator for DefaultGraphSourceOperator { vec![crate::plan_node::PlanChild::child(self.child.as_ref())] } - /// Name the chain and the lane its cost model prefers. - /// - /// The inner subplan is built at `open()` (it needs the snapshot to know - /// whether a sealed arena exists), so `describe()` cannot report the lane - /// that *ran* — that is EXPLAIN ANALYZE territory, per the `Operator` - /// contract. Everything [`Self::chain_lane`] consumes is available here - /// though (the child's schema and the planner stats; only the elision - /// gates need the context), so the cost model's *preference* is reportable - /// and is the single most useful fact about an annotated plan. It is - /// labelled as a preference, and `lane-final` says where the real answer - /// lives: the `annotation delegate lane` tracing event at DEBUG. - /// - /// **`lane-preference` is not evidence of the lane that executed, and must - /// not be used as one.** It is [`Self::chain_lane`]'s answer, computed - /// *before* the runtime gates and — when `FLUREE_ANNOTATION_LANE` is set — - /// simply that variable echoed back, so as a check it confirms its own - /// input. Execution then runs the arena or enumeration lane only if all - /// five of [`Self::annotation_probe_gates_pass`] hold (sealed arena, - /// content store, not a history query, **drained overlay**, root-or-no - /// policy); any one of them falls through to the hash or generic lane - /// with no signal here. A single uncommitted flake fails the overlay - /// condition, which is the easiest of the five to trip by accident. - /// - /// For what actually ran, use the tracing events — `annotation delegate - /// gates` prints each condition individually and `annotation delegate - /// lane` names the lane. Both need `RUST_LOG=debug` **and** the CLI's - /// `-v`; `RUST_LOG` alone emits nothing. fn plan_details(&self) -> serde_json::Map { let mut m = serde_json::Map::new(); - let Some(shape) = - crate::annotation_edge_probe::recognize_annotation_edge(&self.inner_patterns) - else { - m.insert("kind".into(), "unrecognized-chain".into()); - m.insert("patterns".into(), self.inner_patterns.len().into()); - return m; - }; - m.insert("kind".into(), "edge-annotation".into()); - m.insert( - "base".into(), - match &shape.base { - Pattern::Triple(tp) => crate::explain::format_pattern(tp).into(), - other => format!("{other:?}").into(), - }, - ); - m.insert("body-patterns".into(), shape.body.len().into()); - m.insert( - "body-filters".into(), - shape - .body - .iter() - .filter(|p| matches!(p, Pattern::Filter(_))) - .count() - .into(), - ); - let child_bound: HashSet = self.child.schema().iter().copied().collect(); - let elided = - elide_redundant_chain(&self.inner_patterns, &self.needed_outside, &child_bound); - m.insert("chain-elided".into(), elided.is_some().into()); - let lane = match self.chain_lane(&self.child, elided.is_some()) { - ChainLane::Arena => "arena", - ChainLane::Enumerate => "enumerate", - ChainLane::Chain => "chain", - }; - m.insert("lane-preference".into(), lane.into()); - m.insert( - "lane-final".into(), - "decided at open; see the `annotation delegate lane` DEBUG event".into(), - ); + m.insert("patterns".into(), self.inner_patterns.len().into()); m } } - -#[cfg(test)] -mod tests { - use super::{ - choose_chain_lane, elide_redundant_chain, hash_lane_admits, ChainLane, LaneInputs, - }; - use crate::ir::{Pattern, Ref, Term, TriplePattern}; - use crate::var_registry::VarId; - use fluree_db_core::Sid; - use fluree_vocab::db::{REIFIES_OBJECT, REIFIES_PREDICATE, REIFIES_SUBJECT}; - use fluree_vocab::namespaces::FLUREE_DB; - use std::collections::HashSet; - - const FULL_ARENA: Option = Some(21_400_294.0); - const SLICE_ARENA: Option = Some(300_000.0); - - fn lane( - edge_first: f64, - probe_first: f64, - arena_rows: Option, - subject_bound: bool, - reifier_bound: bool, - ) -> ChainLane { - choose_chain_lane(LaneInputs { - edge_first, - probe_first, - arena_rows, - subject_bound, - reifier_bound, - chain_elided: false, - }) - } - - fn lane_elided(edge_first: f64, probe_first: f64, arena_rows: Option) -> ChainLane { - choose_chain_lane(LaneInputs { - edge_first, - probe_first, - arena_rows, - subject_bound: false, - reifier_bound: false, - chain_elided: true, - }) - } - - const S: VarId = VarId(1); - const P: VarId = VarId(2); - const O: VarId = VarId(3); - const ANN: VarId = VarId(4); - const X: VarId = VarId(5); - - fn triple(s: Ref, p: Ref, o: Term) -> Pattern { - Pattern::Triple(TriplePattern { s, p, o, dtc: None }) - } - - fn reifies(name: &str) -> Ref { - Ref::Sid(Sid::new(FLUREE_DB, name)) - } - - /// `<< s p o >> :q ?x` as `expand_edge_annotation_patterns` emits it. - fn chain(s: Ref, p: Ref, o: Term) -> Vec { - let p_as_term = match &p { - Ref::Var(v) => Term::Var(*v), - Ref::Sid(sid) => Term::Sid(sid.clone()), - Ref::Iri(iri) => Term::Iri(iri.clone()), - }; - vec![ - triple(s.clone(), p.clone(), o.clone()), - triple(Ref::Var(ANN), reifies(REIFIES_SUBJECT), s.into()), - triple(Ref::Var(ANN), reifies(REIFIES_PREDICATE), p_as_term), - triple(Ref::Var(ANN), reifies(REIFIES_OBJECT), o), - triple(Ref::Var(ANN), Ref::Sid(Sid::new(9, "q")), Term::Var(X)), - ] - } - - fn reifies_names(patterns: &[Pattern]) -> Vec { - patterns - .iter() - .filter_map(|p| match p { - Pattern::Triple(tp) => match &tp.p { - Ref::Sid(sid) if sid.namespace_code == FLUREE_DB => Some(sid.name.to_string()), - _ => None, - }, - _ => None, - }) - .collect() - } - - fn set(vars: &[VarId]) -> HashSet { - vars.iter().copied().collect() - } - - #[test] - fn count_shape_keeps_only_the_constraining_lookup() { - // P11: `<< ?s :P ?o >> :q ?x` with COUNT(*) — nothing reads ?s/?o. - let typed = Sid::new(9, "P"); - let elided = elide_redundant_chain( - &chain(Ref::Var(S), Ref::Sid(typed.clone()), Term::Var(O)), - &set(&[]), - &set(&[]), - ) - .expect("recognized chain"); - assert_eq!(reifies_names(&elided), vec![REIFIES_PREDICATE]); - assert_eq!(elided.len(), 2, "predicate lookup + body: {elided:?}"); - assert!( - !matches!(&elided[0], Pattern::Triple(tp) if tp.p == Ref::Sid(typed)), - "the base edge must be gone: {elided:?}" - ); - } - - #[test] - fn read_positions_keep_their_lookup_and_a_read_predicate_keeps_everything() { - let typed = Ref::Sid(Sid::new(9, "P")); - // ?o projected, ?s read by a later sibling. - let elided = elide_redundant_chain( - &chain(Ref::Var(S), typed.clone(), Term::Var(O)), - &set(&[O]), - &set(&[S]), - ) - .expect("recognized chain"); - assert_eq!( - reifies_names(&elided), - vec![REIFIES_SUBJECT, REIFIES_PREDICATE, REIFIES_OBJECT] - ); - assert_eq!(elided.len(), 4, "three lookups + body, no base edge"); - // A read variable predicate needs the base scan's predicate binding. - assert!(elide_redundant_chain( - &chain(Ref::Var(S), Ref::Var(P), Term::Var(O)), - &set(&[P]), - &set(&[]), - ) - .is_none()); - // The body's own reads count too. - let mut with_body_read = chain(Ref::Var(S), typed, Term::Var(O)); - with_body_read.push(triple( - Ref::Var(O), - Ref::Sid(Sid::new(9, "r")), - Term::Var(VarId(6)), - )); - let elided = - elide_redundant_chain(&with_body_read, &set(&[]), &set(&[])).expect("recognized chain"); - assert_eq!( - reifies_names(&elided), - vec![REIFIES_PREDICATE, REIFIES_OBJECT] - ); - } - - #[test] - fn hash_lane_leaves_a_bound_reifier_to_the_chain() { - // 2,857 body rows binding the reifier: the lane would walk every - // swept edge per row, the chain probes once per row. - assert!(!hash_lane_admits(Some(2_857), true, true)); - assert!(!hash_lane_admits(None, true, true)); - // The same stream binding the edge subject instead is the lane's case. - assert!(hash_lane_admits(Some(2_857), true, false)); - assert!(hash_lane_admits(None, true, false)); - // Small streams and unbounded sweeps stay on the chain. - assert!(!hash_lane_admits(Some(10), true, false)); - assert!(!hash_lane_admits(Some(2_857), false, false)); - } - - #[test] - fn a_repeated_variable_keeps_the_lookups_that_equate_it() { - let typed = Ref::Sid(Sid::new(9, "P")); - // `<< ?s :P ?s >>` with nothing reading ?s: once the base edge is - // gone, the subject and object lookups joining on ?s are the only - // thing that still requires the two positions to be equal. - let elided = elide_redundant_chain( - &chain(Ref::Var(S), typed.clone(), Term::Var(S)), - &set(&[]), - &set(&[]), - ) - .expect("recognized chain"); - assert_eq!( - reifies_names(&elided), - vec![REIFIES_SUBJECT, REIFIES_PREDICATE, REIFIES_OBJECT] - ); - // A reifier that is also the edge's object keeps - // `?ann f:reifiesObject ?ann`. - let elided = elide_redundant_chain( - &chain(Ref::Var(S), typed, Term::Var(ANN)), - &set(&[]), - &set(&[]), - ) - .expect("recognized chain"); - assert_eq!( - reifies_names(&elided), - vec![REIFIES_PREDICATE, REIFIES_OBJECT] - ); - // A variable predicate repeated in another position counts as read, - // which blocks the rewrite. - assert!(elide_redundant_chain( - &chain(Ref::Var(S), Ref::Var(S), Term::Var(O)), - &set(&[]), - &set(&[]), - ) - .is_none()); - } - - #[test] - fn wildcard_count_keeps_one_lookup_so_the_body_alone_cannot_qualify() { - // P2: `<< ?s ?p ?o >> :q ?x` COUNT(*) — every position unread, but a - // plain subject with `:q` must still not count as a reifier. - let elided = elide_redundant_chain( - &chain(Ref::Var(S), Ref::Var(P), Term::Var(O)), - &set(&[]), - &set(&[]), - ) - .expect("recognized chain"); - assert_eq!(reifies_names(&elided), vec![REIFIES_SUBJECT]); - assert_eq!(elided.len(), 2); - } - - #[test] - fn elided_chain_costs_sequential_rows() { - // P11 with its lookups elided: 31k-est reifiers at ~2 rows each - // beats both the whole-arena decode and the 21.4M enumeration. - assert_eq!( - lane_elided(952_406.0, 31_751.0, FULL_ARENA), - ChainLane::Chain - ); - // P2 elided: one `f:reifiesSubject` range over 21.4M reifiers plus - // the body probe ran 10.6 s against 65 s for the enumeration. - assert_eq!( - lane_elided(1e12, 21_400_294.0, FULL_ARENA), - ChainLane::Chain - ); - // P6 / P7 read the endpoints and the predicate, so nothing is - // elided and the enumeration still beats a point scan per reifier. - assert_eq!( - lane(1e12, 21_400_294.0, FULL_ARENA, false, false), - ChainLane::Enumerate - ); - } - - #[test] - fn lane_choice_matches_measured_starbench_shapes() { - // P2 / P7 / C2: wildcard base edge — walk the arena, slice and full. - assert_eq!( - lane(1e12, 300_000.0, SLICE_ARENA, false, false), - ChainLane::Enumerate - ); - assert_eq!( - lane(1e12, 21_400_294.0, FULL_ARENA, false, false), - ChainLane::Enumerate - ); - // P11 full scale: 952k TREATS edges spread over ~5,200 leaves, so the - // probe decodes the whole arena (14.9 s); the chain drives from - // `f:reifiesPredicate TREATS` (31k est.) and ran 5.3 s. - assert_eq!( - lane(952_406.0, 31_751.0, FULL_ARENA, false, false), - ChainLane::Chain - ); - // Slice P11 (102k edges over 73 leaves): 0.81 s chain vs 1.29 s arena. - assert_eq!( - lane(102_555.0, 3_846.0, SLICE_ARENA, false, false), - ChainLane::Chain - ); - // S20 / S21 object-bound `?s PART_OF `: every edge is its own leaf - // (1,208 edges → 3.1 s on the arena, 0.19 s on the chain). The - // estimate says 11 edges and ~100 reifiers with that object. - assert_eq!( - lane(11.0, 100.0, FULL_ARENA, false, false), - ChainLane::Chain - ); - // P9 / P18 subject-bound hub (3,345 edges in a couple of leaves): - // 50 ms arena vs 200 ms chain. P23 fully bound: 46 ms vs 192 s. - assert_eq!( - lane(3_345.0, 214.0, FULL_ARENA, true, false), - ChainLane::Arena - ); - assert_eq!(lane(1.0, 100.0, FULL_ARENA, true, false), ChainLane::Arena); - // P19: subject bound by a 51k-row child — the chain ran out of memory. - assert_eq!(lane(12.0, 214.0, FULL_ARENA, true, false), ChainLane::Arena); - // C7 / P22 / C10 / P1: the child already binds the reifier, so the - // chain is one probe per driving row — never sweep for it. - assert_eq!( - lane(952_406.0, 1.0, FULL_ARENA, false, true), - ChainLane::Chain - ); - assert_eq!(lane(1e12, 1.0, FULL_ARENA, true, true), ChainLane::Chain); - // The same bound reifier behind a bound subject whose edges fit the - // buffer: without the reifier rule the subject rule takes the arena. - assert_eq!(lane(3_345.0, 1.0, FULL_ARENA, true, true), ChainLane::Chain); - // P5 / P13: bound-object wildcard (1000 est.) with ~9 reifiers. - assert_eq!( - lane(1000.0, 9.0, SLICE_ARENA, false, false), - ChainLane::Chain - ); - // No arena statistics at all: never enumerate. - assert_ne!( - lane(1e12, 300_000.0, None, false, false), - ChainLane::Enumerate - ); - // A base edge wider than the buffer ceiling never takes the probe. - assert_ne!( - lane(30_000_000.0, 30_000_000.0, FULL_ARENA, false, false), - ChainLane::Arena - ); - assert_ne!( - lane(30_000_000.0, 30_000_000.0, FULL_ARENA, true, false), - ChainLane::Arena - ); - } - - #[test] - fn small_arenas_keep_the_edge_first_probe() { - // Cypher-scale ledger (190k reified edges ≈ 46 leaves): a typed - // relationship over 3k edges can touch at most the whole arena, so - // the leaf tax stays bounded and the probe beats 9.5k per-reifier - // point checks. - assert_eq!( - lane(3_000.0, 9_500.0, Some(190_000.0), false, false), - ChainLane::Arena - ); - } -} diff --git a/fluree-db-query/src/execute.rs b/fluree-db-query/src/execute.rs index cb78225d32..f85f8bb366 100644 --- a/fluree-db-query/src/execute.rs +++ b/fluree-db-query/src/execute.rs @@ -48,7 +48,7 @@ pub use runner::ExecutableQuery; // Re-export internal helpers for use in lib.rs pub use where_plan::build_where_operators_seeded; -pub(crate) use where_plan::collect_var_stats; + pub(crate) use where_plan::{analyze_property_join_plan, collect_inner_join_block}; pub use where_plan::{expand_edge_annotation_patterns, expand_edge_annotation_patterns_for}; diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index bd4451b91d..f8bff6982b 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -3117,27 +3117,6 @@ pub fn build_where_operators_seeded_with_needed( } } - // Value-only edge-annotation probe (a Cypher relationship - // binding): answer the whole batch with three set-wise - // reifies scans + hash lookups instead of per-row chains. - if let Some(builder) = crate::optional::AnnotationValueOptionalBuilder::try_new( - required_schema.clone(), - inner_patterns.clone(), - stats.clone(), - *planning, - ) { - operator = Some(Box::new( - OptionalOperator::with_builder( - child, - required_schema, - Box::new(builder), - ) - .with_out_schema(augmented_ref), - )); - i += 1; - continue; - } - // General path: use PlanTreeOptionalBuilder for multi-pattern or // non-triple single patterns (VALUES, BIND, subquery, etc.) let builder = PlanTreeOptionalBuilder::new( @@ -3374,29 +3353,16 @@ pub fn build_where_operators_seeded_with_needed( Pattern::DefaultGraphSource { patterns: inner_patterns, } => { - // Internal — synthesized by `expand_edge_annotation_patterns` - // to correlate the f:reifies* triple chain with a single - // default-graph source under multi-source default queries. + // Internal — synthesized by the edge-annotation expansion to + // correlate an annotated edge's chain with a single + // default-graph source under a default-graph union. let child = require_child(operator, "DEFAULT-GRAPH-SOURCE pattern")?; - // Variables anything outside the wrapper still reads: the - // post-WHERE pipeline (projection pushdown set when there is - // one, else every needed var) plus every later pattern in - // this block. Earlier patterns reach the wrapper through the - // child's schema. The wrapper drops the `f:reifies*` lookups - // whose variable appears in neither. - let mut needed_outside: HashSet = match required_where_vars { - Some(required) => required.iter().copied().collect(), - None => needed_vars.clone(), - }; - let mut outside_counts: HashMap = HashMap::new(); - collect_var_stats(&patterns[i + 1..], &mut outside_counts, &mut needed_outside); operator = Some(Box::new( crate::default_graph_source::DefaultGraphSourceOperator::new( child, inner_patterns.clone(), *planning, stats.clone(), - needed_outside, ), )); i += 1; diff --git a/fluree-db-query/src/lib.rs b/fluree-db-query/src/lib.rs index 9507089b2f..bbf0ddcd7c 100644 --- a/fluree-db-query/src/lib.rs +++ b/fluree-db-query/src/lib.rs @@ -15,7 +15,6 @@ pub mod aggregate; pub(crate) mod aggregate_complement_fold; -pub mod annotation_edge_probe; pub mod binary_history; pub mod binary_range; pub mod binary_scan; diff --git a/fluree-db-query/src/optional.rs b/fluree-db-query/src/optional.rs index f6f2bcf606..d3c6f3898e 100644 --- a/fluree-db-query/src/optional.rs +++ b/fluree-db-query/src/optional.rs @@ -1754,31 +1754,27 @@ fn pattern_filters_tolerate_unbound(p: &Pattern) -> bool { /// (`r2rml_star_is_hash_join_safe`, its own sub-switch). type-var / wildcard / /// bound-subject shapes stay EXCLUDED pending their own differential evidence. /// -/// The Cypher edge-annotation expansion wraps its `[base edge + f:reifies*]` -/// chain in `Pattern::DefaultGraphSource`; in single-source mode that wrapper is -/// a build-once no-op whose per-seed evaluation is likewise a pure restriction by -/// the correlation tuple, so it is admitted recursively (multi-source datasets are -/// excluded at the `build_batch` dataset gate). Merged with the R2RML admission -/// above (DEC-004 F2): the two arms are disjoint `Pattern` variants, so both the -/// R2RML batched-OPTIONAL family and the Cypher value-only annotation probe keep -/// their admission unchanged. +/// The edge-annotation expansion (a Cypher relationship binding) yields the +/// annotation's `rdf:reifies` link and its term components: a pure relation +/// between the term and its components, so its per-seed evaluation is a pure +/// restriction by the correlation tuple too. Under a default-graph union the +/// chain is wrapped in `Pattern::DefaultGraphSource`, admitted recursively here +/// and excluded at the `build_batch` dataset gate. fn inner_pattern_is_hash_join_safe(p: &Pattern) -> bool { match p { Pattern::Triple(_) | Pattern::Filter(_) | Pattern::PropertyPath(_) => true, + // Without this arm a value-only Cypher relationship binding + // (`OPTIONAL { EdgeAnnotation }`) fell to the per-row rebuild path: + // ~25ms/row of replanning that turned a 21k-row UNWIND reindex query + // into minutes of CPU. + Pattern::TermComponents(_) => true, Pattern::R2rml(rp) => { batched_optional_r2rml_enabled() && (r2rml_leaf_is_hash_join_safe(rp) || (batched_optional_r2rml_star_enabled() && r2rml_star_is_hash_join_safe(rp))) } - // The edge-annotation expansion wraps its `[base + f:reifies*]` chain in - // `DefaultGraphSource` (per-source correlation). In single-source mode the - // wrapper is a build-once no-op, and its per-seed evaluation is a pure - // restriction by the correlation tuple — the same property the bare-triple - // chain has. Without this arm a value-only Cypher relationship binding - // (`OPTIONAL { EdgeAnnotation }`) fell to the per-row rebuild path: - // ~25ms/row of replanning that turned a 21k-row UNWIND reindex query into - // minutes of CPU. Multi-source datasets are excluded at the `build_batch` - // gate (dataset presence check). + // The expansion's per-source wrapper; multi-source datasets are + // excluded at the `build_batch` gate (dataset presence check). Pattern::DefaultGraphSource { patterns } => { patterns.iter().all(inner_pattern_is_hash_join_safe) } @@ -1867,225 +1863,6 @@ fn r2rml_star_is_hash_join_safe(rp: &crate::ir::adapters::R2rmlPattern) -> bool && rp.subject_constant.is_none() } -/// Batched builder for the value-only Cypher relationship binding: -/// `OPTIONAL { DefaultGraphSource[base edge + 3 f:reifies* triples] }` with -/// every base-edge position bound by the required row (or constant). -/// -/// The generic path evaluates that chain per required row (or per seeded -/// tuple), and with no stats for the system `f:reifies*` predicates the -/// join can drive from `f:reifiesPredicate` — per row it enumerates -/// ~(sidecar / #relationship-types) candidate reifiers and -/// existence-checks each with its own scan. On a reified ledger that -/// turned a 21k-row UNWIND into minutes of CPU and an OOM. -/// -/// This builder instead drains the three `f:reifies*` predicates ONCE per -/// required batch through ordinary planned scans (overlay-merged and -/// policy-filtered like any scan), builds `subject → reifiers` / -/// `reifier → (predicate, object)` maps, and answers every row by hash -/// lookup. Falls back to the generic per-row path (held as `fallback`) -/// for history queries, attached datasets, and multi-ledger contexts. -pub struct AnnotationValueOptionalBuilder { - fallback: PlanTreeOptionalBuilder, - /// The three reifies triples, with their original vars — executed - /// unseeded so each drains its whole (overlay-merged) predicate. - r_subj: TriplePattern, - r_pred: TriplePattern, - r_obj: TriplePattern, - ann_var: VarId, - s_src: crate::annotation_edge_probe::EdgePos, - p_src: crate::annotation_edge_probe::EdgePos, - o_src: crate::annotation_edge_probe::EdgePos, - stats: Option>, - planning: PlanningContext, -} - -impl AnnotationValueOptionalBuilder { - /// Recognize and construct; `None` defers to the general builder. - pub(crate) fn try_new( - required_schema: Arc<[VarId]>, - inner_patterns: Vec, - stats: Option>, - planning: PlanningContext, - ) -> Option { - use crate::annotation_edge_probe::{recognize_annotation_edge, EdgePos}; - - let [Pattern::DefaultGraphSource { patterns: chain }] = inner_patterns.as_slice() else { - return None; - }; - let shape = recognize_annotation_edge(chain)?; - if !shape.body.is_empty() { - return None; - } - let p_src = match &shape.p_pred { - Ref::Var(v) => EdgePos::Var(*v), - Ref::Sid(sid) => EdgePos::Const(sid.clone()), - Ref::Iri(_) => return None, - }; - // Every variable edge position must be bound by the required row — - // that's what makes the per-row evaluation a pure (s, p, o) lookup. - for pos in [&shape.s_pos, &p_src, &shape.o_pos] { - if let EdgePos::Var(v) = pos { - if !required_schema.contains(v) { - return None; - } - } - } - let (Pattern::Triple(r_subj), Pattern::Triple(r_pred), Pattern::Triple(r_obj)) = - (&chain[1], &chain[2], &chain[3]) - else { - return None; - }; - let (r_subj, r_pred, r_obj) = (r_subj.clone(), r_pred.clone(), r_obj.clone()); - - let fallback = - PlanTreeOptionalBuilder::new(required_schema, inner_patterns, stats.clone(), planning); - // The reifier must be the only optional-only variable; anything else - // means the shape produces bindings this lane doesn't reconstruct. - if fallback.optional_only_vars() != [shape.ann_var] { - return None; - } - Some(Self { - fallback, - r_subj, - r_pred, - r_obj, - ann_var: shape.ann_var, - s_src: shape.s_pos, - p_src, - o_src: shape.o_pos, - stats, - planning, - }) - } - - /// The sidecar maps for this execution, drained on first use. - /// - /// The memo lives on the `ExecutionContext`, not on this builder: the - /// drain costs O(#annotations in the ledger) whatever the result size, and - /// a bounded variable-length Cypher range plans one of these builders per - /// hop of per chain (`*1..3` six, `*1..5` fifteen). A per-operator cache - /// answers the repeat within one operator — a 54k-row result re-drained - /// the whole sidecar ~55 times, once per required batch — but leaves the - /// repeat *across* operators, which is the larger multiple and the one a - /// user can grow just by widening the range. - async fn sidecar_maps( - &self, - ctx: &ExecutionContext<'_>, - view: Option<&fluree_db_binary_index::BinaryGraphView>, - ) -> Result> { - crate::annotation_edge_probe::AnnotationSidecarMaps::shared( - &self.r_subj, - &self.r_pred, - &self.r_obj, - self.stats.clone(), - &self.planning, - ctx, - view, - ) - .await - } - - fn row_sid( - &self, - pos: &crate::annotation_edge_probe::EdgePos, - batch: &Batch, - row: usize, - view: Option<&fluree_db_binary_index::BinaryGraphView>, - ) -> Result> { - use crate::annotation_edge_probe::EdgePos; - match pos { - EdgePos::Const(sid) => Ok(Some(sid.clone())), - EdgePos::Var(v) => match batch.get(row, *v) { - Some(b) => crate::annotation_edge_probe::binding_sid(b, view), - None => Ok(None), - }, - } - } -} - -#[async_trait] -impl OptionalBuilder for AnnotationValueOptionalBuilder { - fn build( - &self, - required_batch: &Batch, - row: usize, - ctx: &ExecutionContext<'_>, - ) -> Result> { - self.fallback.build(required_batch, row, ctx) - } - - async fn build_batch( - &self, - required_batch: &Batch, - start_row: usize, - ctx: &ExecutionContext<'_>, - ) -> Result>> { - if start_row >= required_batch.len() - || self.planning.is_history() - || ctx.dataset.is_some() - || ctx.is_multi_ledger() - { - return Ok(None); - } - let view = ctx.graph_view(); - let view = view.as_ref(); - - // One pass over each reifies predicate (overlay-merged, policy- - // filtered planned scans) — cached across required batches — then - // pure hash lookups per row. - let maps = self.sidecar_maps(ctx, view).await?; - let opt_schema: Arc<[VarId]> = Arc::from(vec![self.ann_var].into_boxed_slice()); - let mut pending = Vec::with_capacity(required_batch.len() - start_row); - for row in start_row..required_batch.len() { - let key = ( - self.row_sid(&self.s_src, required_batch, row, view)?, - self.row_sid(&self.p_src, required_batch, row, view)?, - ); - let o = - crate::annotation_edge_probe::row_obj_key(required_batch, row, &self.o_src, view); - let ((Some(s), Some(p)), false) = ( - key, - matches!(o, crate::group_aggregate::GroupKeyOwned::Absent), - ) else { - pending.push((row, Vec::new())); - continue; - }; - let anns: Vec = maps - .anns_for(&s, &p, &o) - .iter() - .map(|ann| Binding::sid(ann.clone())) - .collect(); - if anns.is_empty() { - pending.push((row, Vec::new())); - } else { - let batch = Batch::new(opt_schema.clone(), vec![anns])?; - pending.push((row, vec![batch])); - } - } - tracing::debug!( - rows = pending.len(), - "annotation value-only optional batched probe complete" - ); - Ok(Some(pending)) - } - - fn schema(&self) -> &[VarId] { - self.fallback.schema() - } - - fn optional_only_vars(&self) -> &[VarId] { - self.fallback.optional_only_vars() - } - - fn unify_instructions(&self) -> &[UnifyInstruction] { - self.fallback.unify_instructions() - } - - fn unmatched_optional(&self) -> UnmatchedOptional { - self.planning.unmatched_optional - } -} - /// True iff `v` occurs in the inner patterns ONLY as the object of one or more /// Triples (never a subject/predicate, never inside a filter/path/other /// pattern). Such a correlation var can be left unbound in the seeded inner so @@ -3152,6 +2929,19 @@ mod tests { assert_eq!(builder.unify_instructions()[0].left_col, 0); // ?s in required } + #[test] + fn annotation_term_components_are_hash_join_safe() { + use crate::ir::{Component, TermComponentsPattern}; + assert!(inner_pattern_is_hash_join_safe(&Pattern::TermComponents( + TermComponentsPattern { + term: VarId(0), + subject: Component::Var(VarId(1)), + predicate: Component::Any, + object: Component::Var(VarId(2)), + } + ))); + } + // PR-4b: the batched-OPTIONAL admission for R2RML inners is NARROW — only a // subject-driven single-object leaf (scalar POM / single-valued ref). Every // richer shape stays on the per-row path pending differential evidence. diff --git a/fluree-db-query/src/planner.rs b/fluree-db-query/src/planner.rs index a1414ffac0..c479dcdae9 100644 --- a/fluree-db-query/src/planner.rs +++ b/fluree-db-query/src/planner.rs @@ -1237,95 +1237,13 @@ pub fn estimate_pattern( }, // DefaultGraphSource wraps an expanded edge-annotation chain and - // runs it once per default-graph source. The chain has its own - // cardinality model (one reifier per matching base edge); anything - // the chain recognizer rejects falls back to the branch model. + // runs it once per default-graph source. Pattern::DefaultGraphSource { patterns, .. } => PatternEstimate::Source { - row_count: estimate_annotation_chain_cardinality(patterns, bound_vars, stats) - .unwrap_or_else(|| estimate_branch_cardinality_from(patterns, bound_vars, stats)), + row_count: estimate_branch_cardinality_from(patterns, bound_vars, stats), }, } } -/// The two entry points of an expanded edge-annotation chain, costed with -/// `bound_vars` already bound: rows the base-edge scan yields (edge-first) -/// and rows the cheaper of the `f:reifiesSubject` / `f:reifiesObject` -/// lookups yields (reifier-first). `f:reifiesPredicate` is left out on -/// purpose: its bound-object estimate divides the arena evenly across -/// predicates, which put `TREATS` at 31k reifiers when it has 1.1M and -/// made the lane choice sweep 952k edges per reifier probe the wrong way -/// round; the endpoint lookups only get small when a binding makes them -/// small. `None` when `patterns` is not a recognized chain. Shared by the -/// wrapper's cardinality estimate and by the delegate's lane choice. -pub(crate) fn annotation_chain_entry_rows( - patterns: &[Pattern], - bound_vars: &HashSet, - stats: Option<&StatsView>, -) -> Option<(f64, f64)> { - let shape = crate::annotation_edge_probe::recognize_annotation_edge(patterns)?; - let triple_rows = |p: &Pattern| match p { - Pattern::Triple(tp) => Some(estimate_triple_row_count(tp, bound_vars, stats)), - _ => None, - }; - let edge_first = triple_rows(&shape.base)?; - let reifier_first = [&patterns[1], &patterns[3]] - .into_iter() - .filter_map(triple_rows) - .fold(f64::INFINITY, f64::min); - Some((edge_first, reifier_first)) -} - -/// Reifier candidates the chain's cheapest `f:reifies*` lookup yields with -/// the child's variables bound — the rows a per-reifier chain must point-check -/// before the body runs. Unlike [`annotation_chain_entry_rows`] this includes -/// `f:reifiesPredicate`: its uniform per-predicate estimate is too coarse to -/// size the wrapper's OUTPUT by, but as a lane entry it is exactly the POST -/// range a predicate-only chain drives from, and it is never larger than the -/// truth by more than the endpoint lookups already are. -pub(crate) fn annotation_chain_probe_rows( - patterns: &[Pattern], - bound_vars: &HashSet, - stats: Option<&StatsView>, -) -> Option { - crate::annotation_edge_probe::recognize_annotation_edge(patterns)?; - let rows = patterns[1..=3] - .iter() - .filter_map(|p| match p { - Pattern::Triple(tp) => Some(estimate_triple_row_count(tp, bound_vars, stats)), - _ => None, - }) - .fold(f64::INFINITY, f64::min); - rows.is_finite().then_some(rows) -} - -/// Cardinality of an expanded edge-annotation chain — `[base edge, three -/// `f:reifies*` triples, body…]`, the only shape `Pattern::DefaultGraphSource` -/// wraps. The generic branch model multiplies the chain's triples in -/// standalone-selectivity order with no regard for connectivity, so -/// `<< ?s :P ?o >>` came out as reifiesPredicate × base edge (3,846 × 102,555 -/// ≈ 4e8 on StarBench P11) and the wrapper sorted behind its own 6.5M-row -/// body triple, which then drove the chain once per row. The chain binds one -/// reifier per matching base edge, so its cardinality is the cheaper of its -/// two entry points — the base edge, or the most selective `f:reifies*` -/// lookup — times the body's expansion with the edge and reifier bound. -pub(crate) fn estimate_annotation_chain_cardinality( - patterns: &[Pattern], - bound_vars: &HashSet, - stats: Option<&StatsView>, -) -> Option { - let shape = crate::annotation_edge_probe::recognize_annotation_edge(patterns)?; - let (edge_first, reifier_first) = annotation_chain_entry_rows(patterns, bound_vars, stats)?; - let reifiers = edge_first.min(reifier_first).max(HIGHLY_SELECTIVE); - if shape.body.is_empty() { - return Some(reifiers); - } - let mut bound = bound_vars.clone(); - bound.extend(shape.base.produced_vars()); - bound.insert(shape.ann_var); - let body = estimate_branch_cardinality_from(&shape.body, &bound, stats); - Some((reifiers * body).max(HIGHLY_SELECTIVE)) -} - /// Estimate cardinality for a sequence of patterns (UNION branch or subquery body). /// /// Uses a context-aware multiplicative model: tracks which variables become bound as @@ -6989,91 +6907,6 @@ mod tests { ); } - #[test] - fn annotation_wrapper_estimate_is_one_reifier_per_edge() { - // StarBench P11: the wrapper's cardinality must not multiply the - // reifiesPredicate lookup by the base edge (3,846 × 102,555 ≈ 4e8), - // which sorted the wrapper behind its own 6.5M-row body triple so - // that triple drove the chain once per row. - use fluree_vocab::db::{REIFIES_OBJECT, REIFIES_PREDICATE, REIFIES_SUBJECT}; - let (s, o, ann, x) = (VarId(0), VarId(1), VarId(2), VarId(3)); - let fsid = |name| Ref::Sid(Sid::new(fluree_vocab::namespaces::FLUREE_DB, name)); - let chain = vec![ - Pattern::Triple(make_pattern(s, "TREATS", o)), - Pattern::Triple(TriplePattern::new( - Ref::Var(ann), - fsid(REIFIES_SUBJECT), - Term::Var(s), - )), - Pattern::Triple(TriplePattern::new( - Ref::Var(ann), - fsid(REIFIES_PREDICATE), - Term::Sid(Sid::new(100, "TREATS")), - )), - Pattern::Triple(TriplePattern::new( - Ref::Var(ann), - fsid(REIFIES_OBJECT), - Term::Var(o), - )), - ]; - let body = Pattern::Triple(make_pattern(ann, "derives_from", x)); - let mut stats = stats_with(&[ - ("TREATS", 102_555, 6_000), - ("derives_from", 6_500_000, 300_000), - ]); - for (name, ndv_values) in [ - (REIFIES_SUBJECT, 34_000), - (REIFIES_PREDICATE, 78), - (REIFIES_OBJECT, 32_000), - ] { - stats.properties.insert( - Sid::new(fluree_vocab::namespaces::FLUREE_DB, name), - PropertyStatData { - count: 300_000, - ndv_values, - ndv_subjects: 300_000, - }, - ); - } - let row_count = |patterns: Vec| match estimate_pattern( - &Pattern::DefaultGraphSource { patterns }, - &HashSet::new(), - Some(&stats), - ) { - PatternEstimate::Source { row_count } => row_count, - other => panic!("wrapper must be a Source: {other:?}"), - }; - - // Bare chain: the base edge (102,555 TREATS rows) bounds the reifier - // count — the endpoint lookups estimate the whole 300k arena and the - // reifiesPredicate lookup is deliberately not consulted (its uniform - // per-predicate split said 3,846 where the slice has 98,641). - let bare = row_count(chain.clone()); - assert!( - (100_000.0..110_000.0).contains(&bare), - "bare chain ≈ 102,555 reifiers, got {bare}" - ); - // With the body nested: reifiers × ~22 derives_from rows each, still - // well under the 6.5M-row body triple on its own. - let mut with_body = chain.clone(); - with_body.push(body.clone()); - let nested = row_count(with_body); - assert!( - (2_000_000.0..2_500_000.0).contains(&nested), - "chain × body ≈ 2.26M, got {nested}" - ); - // And the wrapper drives its body triple, not the other way round. - let ordered = reorder_patterns( - &[Pattern::DefaultGraphSource { patterns: chain }, body], - Some(&stats), - &HashSet::new(), - ); - assert!( - matches!(&ordered[0], Pattern::DefaultGraphSource { .. }), - "the annotation wrapper must seed before its 6.5M-row body triple: {ordered:?}" - ); - } - #[test] fn producer_then_consumer_order_not_regressed() { // IC9 shape: the WITH producer seeds, then its consumer (`HAS_CREATOR`, From 0a25dd4bf9e74949d41bf307b5b7a04a2b044376 Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 21:46:55 -0400 Subject: [PATCH 49/92] fix(import): JSON-LD annotations get their links at bulk import The JSON-LD `@annotation` lowering hands bulk import its f:reifies* bundle as plain triples, and only the reified-triple path spooled a link beside its bundle. A ledger imported from JSON-LD with annotations got no term dictionary, so its link reads were refused as an index built before links (the CLI hid it: `create --from` reindexes). The sink now collects each reifier's subject, predicate and object slots and spools the link once all three are in, the object flake carrying the edge's datatype and tag. --- fluree-db-api/tests/it_triple_term_links.rs | 42 +++++++++++++++++ fluree-db-transact/src/import_sink.rs | 52 +++++++++++++++++++++ 2 files changed, 94 insertions(+) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 038fbb55c0..fd0a030358 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2424,3 +2424,45 @@ GRAPH { .await; assert_eq!(got, strings(&[&["ex:event1"], &["ex:event2"]])); } + +/// Bulk import of JSON-LD `@annotation` writes its bundle as plain triples; +/// the sink spools the link from them, as it does for a reified triple. +#[tokio::test] +async fn jsonld_import_links_annotations() { + const DOC: &str = r#"{"@context": {"ex": "http://example.org/"}, + "@graph": [ + {"@id": "ex:alice", "ex:knows": {"@id": "ex:bob", + "@annotation": {"@id": "ex:claim1", "ex:confidence": 0.9}}}, + {"@id": "ex:doc", "ex:title": {"@value": "chat", "@language": "fr", + "@annotation": {"@id": "ex:claim2", "ex:source": {"@id": "ex:fr"}}}} + ]}"#; + let (fluree, ledger) = import( + &[("data.jsonld", DOC)], + "it/triple-term-links:jsonld-import", + ) + .await; + let got = run_link_query( + &fluree, + &ledger, + "SELECT ?r ?s ?o WHERE { ?r rdf:reifies ?t \ + BIND(SUBJECT(?t) AS ?s) BIND(STR(OBJECT(?t)) AS ?o) } ORDER BY ?r" + .to_string(), + ) + .await; + assert_eq!( + got, + strings(&[ + &["ex:claim1", "ex:alice", "http://example.org/bob"], + &["ex:claim2", "ex:doc", "chat"] + ]) + ); + // The language tag is part of the term, so the tagged edge's quoted + // pattern finds its reifier. + let tagged = run_link_query( + &fluree, + &ledger, + "SELECT ?r WHERE { << ex:doc ex:title \"chat\"@fr ~ ?r >> ex:source ex:fr }".to_string(), + ) + .await; + assert_eq!(tagged, strings(&[&["ex:claim2"]])); +} diff --git a/fluree-db-transact/src/import_sink.rs b/fluree-db-transact/src/import_sink.rs index 3d28f5a56c..a0f5a3c908 100644 --- a/fluree-db-transact/src/import_sink.rs +++ b/fluree-db-transact/src/import_sink.rs @@ -759,6 +759,18 @@ mod inner { prefix_map: HashMap, /// Optional spool context for Tier 2 parallel pipeline. spool_ctx: Option, + /// Each reifier's `f:reifiesSubject` / `f:reifiesPredicate` / + /// `f:reifiesObject` flakes seen so far, when its bundle arrives as + /// plain triples (the JSON-LD `@annotation` lowering). + pending_links: HashMap, + } + + /// A bundle's link-bearing slots, collected until all three are in. + #[derive(Default)] + struct PendingLink { + s: Option, + p: Option, + object: Option, } impl<'a> ImportSink<'a> { @@ -789,6 +801,7 @@ mod inner { encode_error: None, prefix_map: HashMap::new(), spool_ctx: None, + pending_links: HashMap::new(), }) } @@ -812,6 +825,7 @@ mod inner { encode_error: None, prefix_map: HashMap::new(), spool_ctx: None, + pending_links: HashMap::new(), }) } @@ -969,6 +983,44 @@ mod inner { if let Err(e) = written { self.encode_error.get_or_insert(e); } + self.observe_bundle_slot(&flake); + } + } + + /// Spool the RDF 1.2 link of a bundle that arrives as plain triples + /// once its subject, predicate and object slots are in, as + /// `emit_reified_triple` does for a reified triple it parses. + fn observe_bundle_slot(&mut self, flake: &Flake) { + let entry = if fluree_db_core::is_reifies_subject(&flake.p) { + let FlakeValue::Ref(sid) = &flake.o else { + return; + }; + let entry = self.pending_links.entry(flake.s.clone()).or_default(); + entry.s = Some(sid.clone()); + entry + } else if fluree_db_core::is_reifies_predicate(&flake.p) { + let FlakeValue::Ref(sid) = &flake.o else { + return; + }; + let entry = self.pending_links.entry(flake.s.clone()).or_default(); + entry.p = Some(sid.clone()); + entry + } else if fluree_db_core::is_reifies_object(&flake.p) { + let entry = self.pending_links.entry(flake.s.clone()).or_default(); + entry.object = Some(flake.clone()); + entry + } else { + return; + }; + let (Some(s), Some(p), Some(object)) = (&entry.s, &entry.p, &entry.object) else { + return; + }; + let (s, p, object) = (s.clone(), p.clone(), object.clone()); + self.pending_links.remove(&flake.s); + if let Some(ctx) = &mut self.spool_ctx { + if let Err(e) = ctx.write_link_record(&s, &p, &object, self.t) { + self.encode_error.get_or_insert(e); + } } } } From 044c7f8a70c6c540440e2c7633f2f33891da6b92 Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 22:02:44 -0400 Subject: [PATCH 50/92] feat(core): commits carry triple terms The commit codec refused FlakeValue::TripleTerm, so a link could exist only as data derived from an f:reifies* bundle. Commits now encode it under a new object tag: the term's subject, predicate and datatype names in the commit's own dictionaries, an optional language tag, then the object as a nested tag and payload. Both decoders read it. The build resolver turns a committed link's term into a pseudo-record in the chunk's term table, the form bulk import already spools, so full and incremental builds intern it like an imported one; a decimal, big-integer or vector object is keyed by its canonical form, and a nested term is refused, as on every write path. Nothing writes such commits yet: this is the storage the annotation writes move to. Commits that carry triple terms cannot be read by earlier versions. --- fluree-db-api/tests/it_triple_term_links.rs | 105 ++++++++++++++++++ fluree-db-core/src/commit/codec.rs | 2 +- fluree-db-core/src/commit/codec/format.rs | 8 +- fluree-db-core/src/commit/codec/op_codec.rs | 70 +++++++++++- fluree-db-core/src/commit/codec/raw_reader.rs | 46 ++++++++ .../src/run_index/resolve/resolver.rs | 78 ++++++++++++- 6 files changed, 299 insertions(+), 10 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index fd0a030358..9933599037 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2466,3 +2466,108 @@ async fn jsonld_import_links_annotations() { .await; assert_eq!(tagged, strings(&[&["ex:claim2"]])); } + +/// A link written straight into a commit, the form annotation writes move to: +/// it round-trips the commit codec and reads back from novelty, from a full +/// rebuild, and from an incremental build over it. +#[tokio::test] +async fn links_written_into_commits_read_back() { + use fluree_db_core::{Flake, FlakeValue, TripleTermValue}; + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/triple-term-links:committed-links"; + let commit_link = |ledger: LedgerState, o: &'static str, claim: &'static str| { + let fluree = &fluree; + async move { + let mut ns = fluree_db_transact::NamespaceRegistry::from_db(&ledger.snapshot); + let mut ex = |name: &str| ns.sid_for_iri(&format!("http://example.org/{name}")); + let (alice, knows, obj, reifier, confidence) = + (ex("alice"), ex("knows"), ex(o), ex(claim), ex("confidence")); + let t = ledger.t() + 1; + let id_dt = fluree_db_core::edge::id_datatype_sid(); + let term = TripleTermValue { + s: alice.clone(), + p: knows.clone(), + o: FlakeValue::Ref(obj.clone()), + dt: id_dt.clone(), + lang: None, + }; + let flakes = vec![ + Flake::new(alice, knows, FlakeValue::Ref(obj), id_dt, t, true, None), + Flake::new( + reifier.clone(), + fluree_db_core::rdf_reifies_sid().clone(), + FlakeValue::TripleTerm(Box::new(term)), + fluree_db_core::triple_term_datatype_sid().clone(), + t, + true, + None, + ), + Flake::new( + reifier, + confidence, + FlakeValue::Double(0.9), + fluree_db_core::Sid::new(2, "double"), + t, + true, + None, + ), + ]; + let view = fluree_db_transact::stage_flakes( + ledger, + flakes, + fluree_db_transact::StageOptions::new(), + ) + .await + .expect("stage"); + fluree + .commit_staged( + view, + ns, + &fluree_db_ledger::IndexConfig { + reindex_min_bytes: 100_000, + reindex_max_bytes: 1_000_000_000, + }, + fluree_db_transact::CommitOpts::default(), + ) + .await + .expect("commit") + .1 + } + }; + let claims = |ledger: LedgerState| { + let fluree = &fluree; + async move { + run_link_query( + fluree, + &ledger, + "SELECT ?r ?o WHERE { << ex:alice ex:knows ?o ~ ?r >> ex:confidence ?c } ORDER BY ?r" + .to_string(), + ) + .await + } + }; + + let ledger = commit_link(support::genesis_ledger(&fluree, ledger_id), "bob", "claim1").await; + assert_eq!( + claims(ledger).await, + strings(&[&["ex:claim1", "ex:bob"]]), + "novelty" + ); + + support::rebuild_and_publish_index(&fluree, ledger_id).await; + let ledger = fluree.ledger(ledger_id).await.expect("load"); + assert_eq!( + claims(ledger.clone()).await, + strings(&[&["ex:claim1", "ex:bob"]]), + "rebuild" + ); + + commit_link(ledger, "carol", "claim2").await; + support::build_and_publish_index(&fluree, ledger_id).await; + let ledger = fluree.ledger(ledger_id).await.expect("load"); + assert_eq!( + claims(ledger).await, + strings(&[&["ex:claim1", "ex:bob"], &["ex:claim2", "ex:carol"]]), + "incremental" + ); +} diff --git a/fluree-db-core/src/commit/codec.rs b/fluree-db-core/src/commit/codec.rs index 2632d9ba20..2a8314474c 100644 --- a/fluree-db-core/src/commit/codec.rs +++ b/fluree-db-core/src/commit/codec.rs @@ -50,7 +50,7 @@ mod writer; pub use envelope::CodecEnvelope; pub use error::CommitCodecError; pub use format::{CommitSignature, ALGO_ED25519, MAGIC, VERSION, VERSION_V3}; -pub use raw_reader::{CommitOps, RawObject, RawOp}; +pub use raw_reader::{CommitOps, RawObject, RawOp, RawTripleTerm}; #[cfg(feature = "credential")] pub use writer::{write_commit, CommitWriteResult}; diff --git a/fluree-db-core/src/commit/codec/format.rs b/fluree-db-core/src/commit/codec/format.rs index 995e2979cb..dc9ed4ee0d 100644 --- a/fluree-db-core/src/commit/codec/format.rs +++ b/fluree-db-core/src/commit/codec/format.rs @@ -86,6 +86,9 @@ pub enum OTag { Duration = 19, GeoPoint = 20, Vector = 21, + /// An RDF 1.2 triple term: subject, predicate and datatype names, an + /// optional language tag, then the object as a nested tag + payload. + TripleTerm = 22, } impl OTag { @@ -113,6 +116,7 @@ impl OTag { 19 => Ok(OTag::Duration), 20 => Ok(OTag::GeoPoint), 21 => Ok(OTag::Vector), + 22 => Ok(OTag::TripleTerm), _ => Err(CommitCodecError::InvalidOpTag(b)), } } @@ -537,11 +541,11 @@ mod tests { #[test] fn test_otag_round_trip() { - for tag_byte in 0..=21u8 { + for tag_byte in 0..=22u8 { let tag = OTag::from_u8(tag_byte).unwrap(); assert_eq!(tag as u8, tag_byte); } - assert!(OTag::from_u8(22).is_err()); + assert!(OTag::from_u8(23).is_err()); assert!(OTag::from_u8(255).is_err()); } } diff --git a/fluree-db-core/src/commit/codec/op_codec.rs b/fluree-db-core/src/commit/codec/op_codec.rs index 56955deaa5..7325ae8457 100644 --- a/fluree-db-core/src/commit/codec/op_codec.rs +++ b/fluree-db-core/src/commit/codec/op_codec.rs @@ -150,12 +150,22 @@ fn encode_object( let name_id = dicts.object_ref.insert(sid.name.as_ref()); encode_varint(name_id as u64, buf); } - // Commits still carry reifications as `f:reifies*` bundles; the term - // form is interned index-side only until the write path moves over. - FlakeValue::TripleTerm(_) => { - return Err(CommitCodecError::UnsupportedValue( - "triple term objects are not encodable in commits yet".into(), - )); + FlakeValue::TripleTerm(term) => { + buf.push(OTag::TripleTerm as u8); + encode_varint(term.s.namespace_code as u64, buf); + encode_varint(dicts.subject.insert(term.s.name.as_ref()) as u64, buf); + encode_varint(term.p.namespace_code as u64, buf); + encode_varint(dicts.predicate.insert(term.p.name.as_ref()) as u64, buf); + encode_varint(term.dt.namespace_code as u64, buf); + encode_varint(dicts.datatype.insert(term.dt.name.as_ref()) as u64, buf); + match &term.lang { + Some(lang) => { + buf.push(1); + encode_len_prefixed_str(lang, buf); + } + None => buf.push(0), + } + encode_object(&term.o, dicts, buf)?; } FlakeValue::Long(n) => { buf.push(OTag::Long as u8); @@ -464,6 +474,54 @@ mod tests { } } + /// A link `r rdf:reifies <<( s p o )>>` round-trips with each kind of + /// term object: a node, a language-tagged string, and a decimal. + #[test] + fn test_round_trip_triple_term() { + use crate::TripleTermValue; + let ex = |name: &str| Sid::new(101, name); + let objects = [ + (FlakeValue::Ref(ex("bob")), Sid::new(1, "id"), None), + ( + FlakeValue::String("chat".into()), + Sid::new(3, "langString"), + Some("fr".to_string()), + ), + ( + FlakeValue::Decimal(Box::new("1.50".parse().unwrap())), + Sid::new(2, "decimal"), + None, + ), + ]; + for (o, dt, lang) in objects { + let term = TripleTermValue { + s: ex("alice"), + p: ex("knows"), + o, + dt, + lang, + }; + let flake = Flake::new( + ex("claim1"), + crate::rdf_reifies_sid().clone(), + FlakeValue::TripleTerm(Box::new(term.clone())), + crate::triple_term_datatype_sid().clone(), + 7, + true, + None, + ); + let mut dicts = CommitDicts::new(); + let mut buf = Vec::new(); + encode_op(&flake, &mut dicts, &mut buf).unwrap(); + let mut pos = 0; + let decoded = decode_op(&buf, &mut pos, &round_trip_dicts(&dicts), 7).unwrap(); + assert_eq!(pos, buf.len()); + assert_eq!(decoded.o, FlakeValue::TripleTerm(Box::new(term))); + assert_eq!(decoded.s, flake.s); + assert_eq!(decoded.dt, flake.dt); + } + } + #[test] fn test_round_trip_long() { let flake = make_flake_long(101, "Alice", 101, "age", 30, 1); diff --git a/fluree-db-core/src/commit/codec/raw_reader.rs b/fluree-db-core/src/commit/codec/raw_reader.rs index 4749448178..5c900afe32 100644 --- a/fluree-db-core/src/commit/codec/raw_reader.rs +++ b/fluree-db-core/src/commit/codec/raw_reader.rs @@ -208,6 +208,19 @@ pub enum RawObject<'a> { GeoPoint { lat: f64, lng: f64 }, /// Vector of f64 elements (embedding). Vector(Vec), + /// An RDF 1.2 triple term. + TripleTerm(Box>), +} + +/// A triple term's components as the commit stores them: names borrowed from +/// the commit's subject, predicate and datatype dictionaries. +#[derive(Clone)] +pub struct RawTripleTerm<'a> { + pub s: (u16, &'a str), + pub p: (u16, &'a str), + pub dt: (u16, &'a str), + pub lang: Option<&'a str>, + pub o: RawObject<'a>, } impl<'a> TryFrom> for crate::FlakeValue { @@ -227,6 +240,16 @@ impl<'a> TryFrom> for crate::FlakeValue { } match raw { + RawObject::TripleTerm(term) => { + let RawTripleTerm { s, p, dt, lang, o } = *term; + Ok(FlakeValue::TripleTerm(Box::new(crate::TripleTermValue { + s: Sid::new(s.0, s.1), + p: Sid::new(p.0, p.1), + o: FlakeValue::try_from(o)?, + dt: Sid::new(dt.0, dt.1), + lang: lang.map(str::to_string), + }))) + } RawObject::Ref { ns_code, name } => Ok(FlakeValue::Ref(Sid::new(ns_code, name))), RawObject::Long(n) => Ok(FlakeValue::Long(n)), RawObject::Double(n) => Ok(FlakeValue::Double(n)), @@ -613,6 +636,29 @@ pub(super) fn decode_raw_object<'a>( } Ok(RawObject::Vector(vec)) } + OTag::TripleTerm => { + let mut name = |dict: &'a super::string_dict::StringDict| -> Result<(u16, &'a str), CommitCodecError> { + let ns_code = decode_varint(data, pos)? as u16; + let name_id = decode_varint(data, pos)? as u32; + Ok((ns_code, dict.get(name_id)?)) + }; + let s = name(&dicts.subject)?; + let p = name(&dicts.predicate)?; + let dt = name(&dicts.datatype)?; + let lang = match read_u8(data, pos)? { + 0 => None, + _ => Some(decode_inline_str(data, pos)?), + }; + let o_tag = OTag::from_u8(read_u8(data, pos)?)?; + let o = retype_string_literal(decode_raw_object(o_tag, data, pos, dicts)?, dt.0, dt.1); + Ok(RawObject::TripleTerm(Box::new(RawTripleTerm { + s, + p, + dt, + lang, + o, + }))) + } } } diff --git a/fluree-db-indexer/src/run_index/resolve/resolver.rs b/fluree-db-indexer/src/run_index/resolve/resolver.rs index f8a0f9355e..a4f73ed575 100644 --- a/fluree-db-indexer/src/run_index/resolve/resolver.rs +++ b/fluree-db-indexer/src/run_index/resolve/resolver.rs @@ -14,7 +14,7 @@ use bigdecimal::BigDecimal; use chrono; use fluree_db_binary_index::format::run_record::{RunRecord, LIST_INDEX_NONE}; use fluree_db_core::commit::codec::envelope::CodecEnvelope; -use fluree_db_core::commit::codec::raw_reader::{CommitOps, RawObject, RawOp}; +use fluree_db_core::commit::codec::raw_reader::{CommitOps, RawObject, RawOp, RawTripleTerm}; use fluree_db_core::commit::codec::{load_commit_ops, CommitCodecError}; use fluree_db_core::subject_id::SubjectId; use fluree_db_core::temporal::{ @@ -813,6 +813,9 @@ impl CommitResolver { dicts: &mut GlobalDicts, is_assert: bool, ) -> Result, String> { + if let RawObject::TripleTerm(_) = obj { + return Err("triple-term objects are resolved by the chunked builds".into()); + } // Vector handling is unique: assertions allocate + record fact identity; // retractions look up by fact identity (NOT by value, to avoid aliasing // between distinct subjects with the same vector value); unmatched @@ -871,6 +874,7 @@ impl CommitResolver { }; } let result = match obj { + RawObject::TripleTerm(_) => unreachable!("returned above"), RawObject::Long(v) => Ok((ObjKind::NUM_INT, ObjKey::encode_i64(*v))), RawObject::Double(v) => { // NOTE: Do not optimize integral doubles to NUM_INT here. @@ -1593,6 +1597,72 @@ impl SharedResolverState { })) } + /// A link's triple term, written straight into a commit: the term becomes + /// a pseudo-record in the chunk's term table, resolved as the base + /// triple's own record would be, and the link carries its ordinal — the + /// bulk-import sink's form, which the build remaps and interns. + fn resolve_term_chunk( + &mut self, + term: &RawTripleTerm<'_>, + g_id: GraphId, + chunk: &mut RebuildChunk, + is_assert: bool, + ) -> Result<(ObjKind, ObjKey), String> { + let p_id = self.resolve_predicate(term.p.0, term.p.1); + let dt = checked_dt_id(self.resolve_datatype(term.dt.0, term.dt.1))?; + // An arena handle names a value only within one graph and predicate; + // a term keys the object by its canonical form. + let lexical = match &term.o { + RawObject::BigIntStr(_) | RawObject::DecimalStr(_) | RawObject::Vector(_) => { + let value = fluree_db_core::FlakeValue::try_from(term.o.clone()) + .map_err(|e| e.to_string())?; + fluree_db_core::triple_term::lexical_term_object(&value) + } + RawObject::TripleTerm(_) => return Err("nested triple terms are not supported".into()), + _ => None, + }; + let (o_kind, o_key) = match lexical { + Some((o_type, form)) => { + let kind = if o_type == fluree_db_core::o_type::OType::VECTOR { + ObjKind::VECTOR_ID + } else { + ObjKind::NUM_BIG + }; + let id = chunk.strings.get_or_insert(form.as_bytes()); + (kind, ObjKey::encode_u32_id(id)) + } + None => self + .resolve_object_chunk( + &term.o, + g_id, + term.s.0, + term.s.1, + p_id, + LIST_INDEX_NONE, + dt, + chunk, + is_assert, + )? + .ok_or("a triple term's object resolved to nothing")?, + }; + let s_id = self.resolve_subject_chunk(term.s.0, term.s.1, chunk); + let lang_id = self.languages.get_or_insert(term.lang); + let ordinal = chunk.terms.len() as u64; + chunk.terms.push(RunRecord { + g_id, + s_id: SubjectId::from_u64(s_id), + p_id, + dt, + o_kind: o_kind.as_u8(), + op: 1, + o_key: o_key.as_u64(), + t: 0, + lang_id, + i: LIST_INDEX_NONE, + }); + Ok((ObjKind::TRIPLE_TERM, ObjKey::from_u64(ordinal))) + } + /// Resolve subject to a chunk-local sequential u64 ID. fn resolve_subject_chunk(&mut self, ns_code: u16, name: &str, chunk: &mut RebuildChunk) -> u64 { chunk.subjects.get_or_insert(ns_code, name.as_bytes()) @@ -1644,6 +1714,11 @@ impl SharedResolverState { chunk: &mut RebuildChunk, is_assert: bool, ) -> Result, String> { + if let RawObject::TripleTerm(term) = obj { + return self + .resolve_term_chunk(term, g_id, chunk, is_assert) + .map(Some); + } // Vector handling — see `CommitResolver::resolve_object` for details. // Fact-identity `(s_id, p_id, o_i, f32_bits)` → handle is required so: // (a) two distinct subjects with the same vector value don't alias, @@ -1698,6 +1773,7 @@ impl SharedResolverState { }; } let result = match obj { + RawObject::TripleTerm(_) => unreachable!("returned above"), RawObject::Long(v) => Ok((ObjKind::NUM_INT, ObjKey::encode_i64(*v))), RawObject::Double(v) => { // NOTE: Do not optimize integral doubles to NUM_INT here. From 1558138e0d4d736aacb4e8f3c1c696770740d024 Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 22:40:10 -0400 Subject: [PATCH 51/92] feat(transact): write templates take triple terms An object position can be a triple term whose positions resolve per solution, so a writer with variable endpoints can emit the rdf:reifies link directly. --- fluree-db-transact/src/generate/flakes.rs | 108 +++++++++++++++++++++- fluree-db-transact/src/ir.rs | 26 ++++++ fluree-db-transact/src/lib.rs | 4 +- fluree-db-transact/src/parse/jsonld.rs | 19 ++-- fluree-db-transact/src/stage.rs | 13 ++- 5 files changed, 155 insertions(+), 15 deletions(-) diff --git a/fluree-db-transact/src/generate/flakes.rs b/fluree-db-transact/src/generate/flakes.rs index 356712edcc..76310fb95b 100644 --- a/fluree-db-transact/src/generate/flakes.rs +++ b/fluree-db-transact/src/generate/flakes.rs @@ -4,7 +4,7 @@ //! with variable bindings into concrete flakes. use crate::error::{Result, TransactError}; -use crate::ir::{TemplateGraph, TemplateTerm, TripleTemplate}; +use crate::ir::{TemplateGraph, TemplateTerm, TemplateTripleTerm, TripleTemplate}; use crate::namespace::NamespaceRegistry; use fluree_db_core::{Flake, FlakeMeta, FlakeValue, Sid}; use fluree_db_query::{Batch, Binding, VarId}; @@ -382,6 +382,9 @@ impl<'a> FlakeGenerator<'a> { TemplateTerm::Value(_) => Err(TransactError::InvalidTerm( "Subject cannot be a literal value".to_string(), )), + TemplateTerm::TripleTerm(_) => Err(TransactError::InvalidTerm( + "Subject cannot be a triple term".to_string(), + )), } } @@ -435,6 +438,9 @@ impl<'a> FlakeGenerator<'a> { TemplateTerm::Value(_) => Err(TransactError::InvalidTerm( "Predicate cannot be a literal value".to_string(), )), + TemplateTerm::TripleTerm(_) => Err(TransactError::InvalidTerm( + "Predicate cannot be a triple term".to_string(), + )), } } @@ -499,9 +505,57 @@ impl<'a> FlakeGenerator<'a> { let sid = self.skolemize_blank_node(label, solution); Ok((Some(FlakeValue::Ref(sid)), Some(DT_ID.clone()))) } + TemplateTerm::TripleTerm(term) => { + Ok(match self.resolve_triple_term(term, bindings, row)? { + Some(t) => ( + Some(FlakeValue::TripleTerm(Box::new(t))), + Some(fluree_db_core::triple_term_datatype_sid().clone()), + ), + None => (None, None), + }) + } } } + /// A triple term's positions, resolved as a flake's own are; `None` when + /// one is unbound. + fn resolve_triple_term( + &mut self, + term: &TemplateTripleTerm, + bindings: &Batch, + row: usize, + ) -> Result> { + if matches!(term.o, TemplateTerm::TripleTerm(_)) { + return Err(TransactError::InvalidTerm( + "a triple term cannot nest another".to_string(), + )); + } + let s = self.resolve_subject(&term.s, bindings, row)?; + let p = self.resolve_predicate(&term.p, bindings, row)?; + let explicit_dt = term + .dtc + .as_ref() + .map(fluree_db_core::DatatypeConstraint::datatype); + let (o, dt) = self.resolve_object(&term.o, explicit_dt, bindings, row)?; + let lang = match (&term.dtc, &term.o) { + (Some(dtc), _) => dtc.lang_tag().map(str::to_string), + (None, TemplateTerm::Var(v)) => match bindings.get(row, *v) { + Some(Binding::Lit { dtc, .. }) => dtc.lang_tag().map(str::to_string), + _ => None, + }, + _ => None, + }; + let (Some(s), Some(p), Some(o), Some(dt)) = (s, p, o, dt) else { + return Ok(None); + }; + let dt = if lang.is_some() { + DT_LANG_STRING.clone() + } else { + dt + }; + Ok(Some(fluree_db_core::TripleTermValue { s, p, o, dt, lang })) + } + /// Skolemize a blank node to a Sid. /// /// Creates a unique Sid for a blank node label within this transaction and @@ -703,6 +757,58 @@ mod tests { assert!(flakes[0].s.name.contains("b1")); } + #[test] + fn triple_term_template_resolves_per_solution() { + let mut registry = NamespaceRegistry::new(); + let mut generator = FlakeGenerator::new(1, &mut registry, "txn1".to_string()); + let (r, s, o) = (VarId(0), VarId(1), VarId(2)); + let schema: Arc<[VarId]> = Arc::new([r, s, o]); + let sid = |name: &str| Binding::sid(Sid::new(1, name)); + let batch = Batch::new( + schema, + vec![ + vec![sid("ex:r1"), sid("ex:r2")], + vec![sid("ex:alice"), sid("ex:bob")], + vec![ + sid("ex:bob"), + Binding::Lit { + val: FlakeValue::String("salut".into()), + dtc: DatatypeConstraint::LangTag("fr".into()), + t: None, + op: None, + p_id: None, + }, + ], + ], + ) + .unwrap(); + let templates = vec![TripleTemplate::new( + TemplateTerm::Var(r), + TemplateTerm::Sid(Sid::new(2, "reifies")), + TemplateTerm::TripleTerm(Box::new(TemplateTripleTerm { + s: TemplateTerm::Var(s), + p: TemplateTerm::Sid(Sid::new(1, "ex:says")), + o: TemplateTerm::Var(o), + dtc: None, + })), + )]; + + let flakes = generator.generate_assertions(&templates, &batch).unwrap(); + + assert_eq!(flakes.len(), 2); + let term = |i: usize| match &flakes[i].o { + FlakeValue::TripleTerm(t) => t.as_ref().clone(), + other => panic!("not a triple term: {other:?}"), + }; + assert_eq!(flakes[0].dt, *fluree_db_core::triple_term_datatype_sid()); + assert_eq!(term(0).s, Sid::new(1, "ex:alice")); + assert_eq!(term(0).o, FlakeValue::Ref(Sid::new(1, "ex:bob"))); + assert_eq!(term(0).dt, *DT_ID); + assert_eq!(term(1).o, FlakeValue::String("salut".into())); + assert_eq!(term(1).dt, *DT_LANG_STRING); + assert_eq!(term(1).lang.as_deref(), Some("fr")); + } + #[test] fn test_infer_datatype() { assert_eq!( diff --git a/fluree-db-transact/src/ir.rs b/fluree-db-transact/src/ir.rs index be6ff0d546..8cc0c92eed 100644 --- a/fluree-db-transact/src/ir.rs +++ b/fluree-db-transact/src/ir.rs @@ -630,6 +630,20 @@ pub enum TemplateTerm { /// Blank node (will be skolemized to a Sid during flake generation) BlankNode(String), + + /// An RDF 1.2 triple term, its positions resolved per solution; only an + /// object position takes one. + TripleTerm(Box), +} + +/// The positions of a [`TemplateTerm::TripleTerm`], with the object's +/// datatype or language tag when the template declares one. +#[derive(Debug, Clone)] +pub struct TemplateTripleTerm { + pub s: TemplateTerm, + pub p: TemplateTerm, + pub o: TemplateTerm, + pub dtc: Option, } impl TemplateTerm { @@ -647,6 +661,18 @@ impl TemplateTerm { pub fn is_bound(&self) -> bool { !self.is_var() } + + /// This term, or each position of a triple term, recursively. + pub fn for_each_leaf<'a>(&'a self, f: &mut impl FnMut(&'a TemplateTerm)) { + match self { + TemplateTerm::TripleTerm(t) => { + for term in [&t.s, &t.p, &t.o] { + term.for_each_leaf(f); + } + } + leaf => f(leaf), + } + } } /// Inline VALUES bindings for initial solutions diff --git a/fluree-db-transact/src/lib.rs b/fluree-db-transact/src/lib.rs index 9dfc3a492e..6f30a44f27 100644 --- a/fluree-db-transact/src/lib.rs +++ b/fluree-db-transact/src/lib.rs @@ -66,8 +66,8 @@ pub use error::{Result, TransactError}; pub use flake_sink::FlakeSink; pub use generate::{apply_cancellation, FlakeGenerator}; pub use ir::{ - GraphMgmtOp, GraphSel, GraphTarget, InlineValues, TemplateGraph, TemplateTerm, TripleTemplate, - Txn, TxnOpts, TxnType, + GraphMgmtOp, GraphSel, GraphTarget, InlineValues, TemplateGraph, TemplateTerm, + TemplateTripleTerm, TripleTemplate, Txn, TxnOpts, TxnType, }; pub use lower_sparql_update::{ lower_sparql_update, lower_sparql_update_ast, lower_sparql_update_request, LowerError, diff --git a/fluree-db-transact/src/parse/jsonld.rs b/fluree-db-transact/src/parse/jsonld.rs index 145d13cb73..a7eea86ca3 100644 --- a/fluree-db-transact/src/parse/jsonld.rs +++ b/fluree-db-transact/src/parse/jsonld.rs @@ -939,14 +939,17 @@ impl<'a> TemplateParseCtx<'a> { /// stored data). Stable `_:fdb-` ids never appear here — the term parsers /// resolve them to constant `TemplateTerm::Sid`s. fn first_blank_node_in_templates(templates: &[TripleTemplate]) -> Option<&str> { - templates.iter().find_map(|t| { - [&t.subject, &t.predicate, &t.object] - .into_iter() - .find_map(|term| match term { - TemplateTerm::BlankNode(label) => Some(label.as_str()), - _ => None, - }) - }) + let mut found = None; + for t in templates { + for term in [&t.subject, &t.predicate, &t.object] { + term.for_each_leaf(&mut |leaf| { + if let (None, TemplateTerm::BlankNode(label)) = (found, leaf) { + found = Some(label.as_str()); + } + }); + } + } + found } fn parse_update_templates_with_ctx( diff --git a/fluree-db-transact/src/stage.rs b/fluree-db-transact/src/stage.rs index 8b51f22e9b..f9374c3b12 100644 --- a/fluree-db-transact/src/stage.rs +++ b/fluree-db-transact/src/stage.rs @@ -2653,11 +2653,13 @@ fn collect_template_vars(template_groups: &[&[TripleTemplate]]) -> Vec { for group in template_groups { for tmpl in *group { for term in [&tmpl.subject, &tmpl.predicate, &tmpl.object] { - if let TemplateTerm::Var(v) = term { - if seen.insert(*v) { - out.push(*v); + term.for_each_leaf(&mut |leaf| { + if let TemplateTerm::Var(v) = leaf { + if seen.insert(*v) { + out.push(*v); + } } - } + }); } if let TemplateGraph::Var(v) = tmpl.graph { if seen.insert(v) { @@ -3336,6 +3338,9 @@ fn template_term_to_binding(term: &TemplateTerm) -> Result { TemplateTerm::BlankNode(_) => Err(TransactError::InvalidTerm( "Blank nodes not allowed in VALUES data rows".to_string(), )), + TemplateTerm::TripleTerm(_) => Err(TransactError::InvalidTerm( + "Triple terms not allowed in VALUES data rows".to_string(), + )), } } From 0fc2c676b87b5b7ae813a9d6bc190773c1910b50 Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 23:40:03 -0400 Subject: [PATCH 52/92] fix(index): links written into commits index correctly Commits carry triple terms since the commit codec learned them, and the next writer change makes every annotation one. Three build gaps showed: - A full rebuild counted live links from every raw record, so a link restated in a later commit (a bulk import restating an annotation) counted twice. It now counts the deduplicated merge's live winners. - A term keys a decimal, big-integer or vector object by the string id of its canonical form. Neither build path remapped that id for a commit-carried term, so an incremental build decoded another string under the term's type. `remap_term_record` now serves both paths. - An incremental build over an annotated index built before links started a term dictionary for its window alone, whose handles collide with the base's links and lift the reindex refusal. It now declines, and the index build falls back to the full rebuild, which links every annotation in history. --- fluree-db-api/tests/it_triple_term_links.rs | 47 +++++------- fluree-db-indexer/src/build/rebuild.rs | 54 ++++++-------- .../run_index/build/incremental_resolve.rs | 15 +++- .../src/run_index/resolve/resolver.rs | 72 +++++++++++++++++++ 4 files changed, 125 insertions(+), 63 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 9933599037..083f41591d 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2203,8 +2203,9 @@ async fn construct_writes_triple_terms_as_reifications() { /// An annotated ledger whose index predates links (simulated by dropping the /// root's term dictionary) refuses link reads until a full rebuild links its -/// annotations. An incremental build in between does not start a term -/// dictionary, which would cover its window alone and lift the refusal. +/// annotations. A new link declines the incremental build, whose term +/// dictionary would cover its window alone and lift the refusal; the index +/// build falls back to the full rebuild. #[tokio::test] async fn link_reads_refuse_an_index_built_before_links() { use fluree_db_binary_index::format::index_root::IndexRoot; @@ -2234,7 +2235,7 @@ async fn link_reads_refuse_an_index_built_before_links() { (IndexRoot::decode(&bytes).expect("decode"), record.index_t) }; let query = "PREFIX ex: \n\ - SELECT ?r WHERE { << ?s ex:knows ?o ~ ?r >> ex:src ex:x } ORDER BY ?r"; + SELECT ?r ?s WHERE { << ?s ex:knows ?o ~ ?r >> ex:src ex:x } ORDER BY ?r"; let ledger = support::genesis_ledger(&fluree, ledger_id); fluree @@ -2268,43 +2269,27 @@ async fn link_reads_refuse_an_index_built_before_links() { .upsert_turtle(ledger, &claim(2)) .await .expect("claim 2"); - support::build_and_publish_index(&fluree, ledger_id).await; - let (root, _) = current_root().await; - assert!( - root.term_dict.is_none(), - "an incremental build over a pre-link index must not start a term dictionary" - ); let ledger = fluree.ledger(ledger_id).await.expect("load"); assert!(support::query_sparql(&fluree, &ledger, query) .await .is_err()); - // Same `t` as the incremental root, so it publishes with allow-equal. - let record = fluree - .nameservice() - .lookup(ledger_id) - .await - .expect("ns lookup") - .expect("ns record"); - let rebuilt = fluree_db_indexer::rebuild_index_from_commits( - fluree.content_store(ledger_id), - ledger_id, - &record, - fluree_db_indexer::IndexerConfig::default(), - ) - .await - .expect("rebuild"); - fluree - .publisher() - .expect("read-write nameservice") - .publish_index_allow_equal(ledger_id, rebuilt.index_t, &rebuilt.root_id) - .await - .expect("publish rebuild"); + support::build_and_publish_index(&fluree, ledger_id).await; + let (root, _) = current_root().await; + assert!( + root.term_dict.is_some(), + "the index build fell back to a rebuild" + ); let ledger = fluree.ledger(ledger_id).await.expect("load"); let result = support::query_sparql_formatted(&fluree, &ledger, query) .await .expect("a rebuilt index answers"); - assert_eq!(rows(&result), strings(&[&["ex:claim1"], &["ex:claim2"]])); + // Each reifier decodes to its own triple: a dictionary started over the + // window alone would hand claim 2's term the handle claim 1's link holds. + assert_eq!( + rows(&result), + strings(&[&["ex:claim1", "ex:s1"], &["ex:claim2", "ex:s2"]]) + ); } /// The JSON-LD twin of `triple_constructs_the_terms_links_hold`: the diff --git a/fluree-db-indexer/src/build/rebuild.rs b/fluree-db-indexer/src/build/rebuild.rs index ea6c80d234..ba422371c8 100644 --- a/fluree-db-indexer/src/build/rebuild.rs +++ b/fluree-db-indexer/src/build/rebuild.rs @@ -595,34 +595,10 @@ where // their ordinal for the global handle. let terms = &mut chunk_terms[ci]; for term in terms.iter_mut() { - let local_s = term.s_id.as_u64() as usize; - let global_s = *s_remap.get(local_s).ok_or_else(|| { - IndexerError::StorageWrite(format!( - "term subject remap miss: chunk {ci}, local_s={local_s}" - )) - })?; - term.s_id = fluree_db_core::subject_id::SubjectId::from_u64(global_s); - let kind = fluree_db_core::value_id::ObjKind::from_u8(term.o_kind); - if kind == fluree_db_core::value_id::ObjKind::REF_ID { - let local_o = term.o_key as usize; - term.o_key = *s_remap.get(local_o).ok_or_else(|| { - IndexerError::StorageWrite(format!( - "term object remap miss: chunk {ci}, local_o={local_o}" - )) - })?; - } else if kind == fluree_db_core::value_id::ObjKind::LEX_ID - || kind == fluree_db_core::value_id::ObjKind::JSON_ID - { - let local_str = fluree_db_core::value_id::ObjKey::from_u64(term.o_key) - .decode_u32_id() as usize; - let global_str = *str_remap.get(local_str).ok_or_else(|| { - IndexerError::StorageWrite(format!( - "term string remap miss: chunk {ci}, local_str={local_str}" - )) - })?; - term.o_key = - fluree_db_core::value_id::ObjKey::encode_u32_id(global_str).as_u64(); - } + crate::run_index::resolve::resolver::remap_term_record( + term, s_remap, str_remap, + ) + .map_err(|e| IndexerError::StorageWrite(format!("chunk {ci}: {e}")))?; } let mut attachments = std::mem::take(&mut chunk_attachments[ci]); for op in &mut attachments { @@ -1066,7 +1042,7 @@ where crate::run_index::build::ClassMembership::build_from_global_types(&types_paths) .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; - let spot_class_stats = { + let (spot_class_stats, live_term_rows) = { use crate::run_index::build::SpotClassStatsCollector; use crate::run_index::runs::spool::V1SpoolMergeAdapter; use fluree_db_binary_index::format::run_record_v2::cmp_v2_g_spot; @@ -1093,16 +1069,32 @@ where // Iterate with dedup: next_deduped() returns the winning record // per identity group (highest t wins). Feed assertions to collector. + // + // Link counts come from here, not from pass 1: a commit chain can + // re-assert a live link (a bulk import restating an annotation), + // which pass 1 counts once per copy. + let mut live_term_rows: std::collections::BTreeMap<(u32, u32), u64> = + std::collections::BTreeMap::new(); while let Some((winner, op)) = merge .next_deduped() .map_err(|e| IndexerError::StorageWrite(e.to_string()))? { if op == 1 { + if fluree_db_core::o_type::OType::from_u16(winner.o_type) + == fluree_db_core::o_type::OType::TRIPLE_TERM + { + let inner = fluree_db_core::triple_term::term_handle_p_id(winner.o_key); + *live_term_rows.entry((winner.p_id, inner)).or_insert(0) += 1; + } collector.on_record(&winner); } } + let live_term_rows: Vec<(u32, u32, u64)> = live_term_rows + .into_iter() + .map(|((p_id, inner), n)| (p_id, inner, n)) + .collect(); - collector.finish() + (collector.finish(), live_term_rows) }; // ---- Build IndexStats for FIR6 root ---- @@ -1161,7 +1153,7 @@ where let root_classes = fluree_db_core::index_stats::union_per_graph_classes(&final_graphs); let links = crate::stats::link_stat_entries( - &id_stats_result.term_rows, + &live_term_rows, shared.predicates.get(fluree_vocab::rdf::REIFIES), |p_id| { predicate_sids diff --git a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs index c750975b71..0bb324659a 100644 --- a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs +++ b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs @@ -70,6 +70,8 @@ pub enum IncrementalResolveError { Resolve(ResolverError), /// I/O error. Io(io::Error), + /// The window cannot be applied incrementally; a full rebuild can. + NeedsRebuild(String), } impl std::fmt::Display for IncrementalResolveError { @@ -80,6 +82,7 @@ impl std::fmt::Display for IncrementalResolveError { Self::CommitChain(msg) => write!(f, "commit chain: {msg}"), Self::Resolve(e) => write!(f, "resolve: {e}"), Self::Io(e) => write!(f, "I/O: {e}"), + Self::NeedsRebuild(msg) => write!(f, "needs a full rebuild: {msg}"), } } } @@ -676,7 +679,12 @@ pub async fn resolve_incremental_commits_v6( // dictionary, or a fresh one above the base watermarks. let mut chunk_terms = chunk.terms; for term in &mut chunk_terms { - remap_record(term, &reconcile.subject_remap, &reconcile.string_remap)?; + crate::run_index::resolve::resolver::remap_term_record( + term, + &reconcile.subject_remap, + &reconcile.string_remap, + ) + .map_err(|e| IncrementalResolveError::Resolve(ResolverError::Resolve(e)))?; } // Attachment ops replay per reifier from the attachment the base index // holds, so a re-point that touched one slot still moves the link. @@ -843,6 +851,11 @@ pub async fn resolve_incremental_commits_v6( (new_terms, wms.into_iter().collect::>()) }; drop(chunk_terms); + if root.has_annotations && root.term_dict.is_none() && !new_terms.is_empty() { + return Err(IncrementalResolveError::NeedsRebuild( + "the window holds triple terms over an annotated index built before links".to_string(), + )); + } // VECTOR_ID handles are already globally-correct: chunk inserts // appended to the pre-loaded base arena (step 4b) so they return diff --git a/fluree-db-indexer/src/run_index/resolve/resolver.rs b/fluree-db-indexer/src/run_index/resolve/resolver.rs index a4f73ed575..a05d1c64f5 100644 --- a/fluree-db-indexer/src/run_index/resolve/resolver.rs +++ b/fluree-db-indexer/src/run_index/resolve/resolver.rs @@ -2296,6 +2296,40 @@ impl std::error::Error for ResolverError {} /// /// Returns `None` if parsing fails (caller skips emission rather than /// poisoning the index with `0`). +/// Remap a chunk term-table entry ([`SharedResolverState::resolve_term_chunk`]) +/// to global ids: its subject, and a ref or string-keyed object. A term keys +/// a decimal, big-integer or vector object by the string id of its canonical +/// form, so those kinds remap as strings here, though in an ordinary record +/// they hold arena handles. +pub fn remap_term_record( + term: &mut RunRecord, + subject_remap: &[u64], + string_remap: &[u32], +) -> Result<(), String> { + let local_s = term.s_id.as_u64() as usize; + let global_s = *subject_remap + .get(local_s) + .ok_or_else(|| format!("term subject remap miss: local_s={local_s}"))?; + term.s_id = SubjectId::from_u64(global_s); + match ObjKind::from_u8(term.o_kind) { + ObjKind::REF_ID => { + let local_o = term.o_key as usize; + term.o_key = *subject_remap + .get(local_o) + .ok_or_else(|| format!("term object remap miss: local_o={local_o}"))?; + } + ObjKind::LEX_ID | ObjKind::JSON_ID | ObjKind::NUM_BIG | ObjKind::VECTOR_ID => { + let local_str = ObjKey::from_u64(term.o_key).decode_u32_id() as usize; + let global_str = *string_remap + .get(local_str) + .ok_or_else(|| format!("term string remap miss: local_str={local_str}"))?; + term.o_key = ObjKey::encode_u32_id(global_str).as_u64(); + } + _ => {} + } + Ok(()) +} + fn iso_to_epoch_ms(iso: &str) -> Option { chrono::DateTime::parse_from_rfc3339(iso) .ok() @@ -2329,6 +2363,44 @@ fn split_iri_to_value_type_tag( // Tests // ============================================================================ +#[cfg(test)] +mod term_remap_tests { + use super::*; + + fn term(o_kind: ObjKind, o_key: u64) -> RunRecord { + RunRecord { + g_id: 0, + s_id: SubjectId::from_u64(1), + p_id: 0, + dt: 0, + o_kind: o_kind.as_u8(), + op: 1, + o_key, + t: 0, + lang_id: 0, + i: LIST_INDEX_NONE, + } + } + + #[test] + fn lexical_term_objects_remap_as_strings() { + let subjects = [10, 11, 12]; + let strings = [20, 21]; + let remapped = |kind: ObjKind, o_key: u64| { + let mut t = term(kind, o_key); + remap_term_record(&mut t, &subjects, &strings).unwrap(); + (t.s_id.as_u64(), t.o_key) + }; + let str1 = ObjKey::encode_u32_id(1).as_u64(); + let global = ObjKey::encode_u32_id(21).as_u64(); + for kind in [ObjKind::NUM_BIG, ObjKind::VECTOR_ID, ObjKind::LEX_ID] { + assert_eq!(remapped(kind, str1), (11, global), "{kind:?}"); + } + assert_eq!(remapped(ObjKind::REF_ID, 2), (11, 12)); + assert_eq!(remapped(ObjKind::NUM_INT, 7), (11, 7)); + } +} + #[cfg(test)] mod tests { use super::*; From 5e846385e21c806527c297979416b4bf14e488c9 Mon Sep 17 00:00:00 2001 From: bplatz Date: Fri, 2 Oct 2026 23:41:06 -0400 Subject: [PATCH 53/92] feat: annotations store the rdf:reifies link An annotation is now stored as RDF 1.2 defines it: the reifier's link `r rdf:reifies <<( s p o )>>` plus its body, in the edge's graph. The `f:reifies*` bundle is no longer written. Writers: - Turtle, TriG and bulk import write the link flake for every reified form. - SPARQL UPDATE desugars an annotation tail as the spec does, to the base triple plus the link; template lowering takes the triple term, and DELETE WHERE matches it with the reifier pattern queries use. - Cypher CREATE writes the relationship's link. - JSON-LD has no triple-term syntax, so `@annotation` still lowers to `f:reifies*` slot keys, which fold into the link after parsing; a delete-by-selector matches through `@reifies`. Bulk import folds slot triples the same way, which also reads exports written before links. Readers move to the link: the stage-time cascade (a POST probe by term), `@annotation` hydration, export, Cypher relationship rendering and `type()`/`startNode()`/`endNode()`, graph management (links re-home like any flake), and the annotation flags on the index root and novelty. A reifier may now reify several triples, as RDF 1.2 allows, so the single-target check, its replay escape and the ADD collision refusal are gone. Export reads the live links once per export, which resolves annotations in named graphs and retires the base-index scan switch; `--raw-reifies` now writes the links as triples. Commits written before this keep their bundles, and the index build and novelty still derive links from them; a write that retracts such a link cancels the derived assert. There is no migration step. --- docs/cli/export.md | 16 +- docs/cli/server-integration.md | 6 +- docs/concepts/edge-annotations.md | 25 +- docs/design/README.md | 2 +- docs/design/edge-annotations.md | 184 +-- docs/design/rules-engine.md | 2 +- docs/operations/telemetry.md | 11 +- docs/reference/vocabulary.md | 4 +- docs/transactions/turtle.md | 3 +- fluree-db-api/src/admin.rs | 4 +- fluree-db-api/src/commit_transfer.rs | 6 +- fluree-db-api/src/export.rs | 233 ++- fluree-db-api/src/export_annotations.rs | 390 ++--- fluree-db-api/src/export_builder.rs | 19 +- fluree-db-api/src/format/cypher_typed.rs | 109 +- fluree-db-api/src/format/hydration.rs | 359 +---- fluree-db-api/src/import.rs | 5 +- fluree-db-api/src/lib.rs | 2 +- fluree-db-api/src/query/helpers.rs | 8 +- fluree-db-api/src/tx.rs | 64 +- fluree-db-api/src/tx_builder.rs | 2 +- fluree-db-api/tests/it_edge_annotations.rs | 697 ++------- .../tests/it_edge_annotations_indexed.rs | 1265 ++--------------- fluree-db-api/tests/it_import.rs | 80 +- fluree-db-api/tests/it_import_turtle_star.rs | 45 +- .../tests/it_query_sparql_annotations.rs | 74 +- fluree-db-api/tests/it_tracing_spans.rs | 21 +- .../tests/it_turtle_star_write_paths.rs | 60 +- fluree-db-api/tests/support/mod.rs | 103 +- fluree-db-cli/src/cli.rs | 9 +- fluree-db-cli/src/commands/export.rs | 2 +- fluree-db-cli/src/commands/index.rs | 2 +- fluree-db-cli/tests/integration.rs | 118 +- fluree-db-core/src/lib.rs | 20 +- fluree-db-core/src/namespaces.rs | 7 + fluree-db-cypher/src/lower/pattern.rs | 12 +- fluree-db-indexer/src/build/root_assembly.rs | 10 +- .../src/run_index/build/incremental_root.rs | 9 +- fluree-db-novelty/src/lib.rs | 12 + fluree-db-query/src/eval/metadata.rs | 216 +-- fluree-db-server/Cargo.toml | 6 - .../tests/export_omission_headers.rs | 34 +- fluree-db-server/tests/export_scan_source.rs | 198 --- fluree-db-transact/src/flake_sink.rs | 97 +- fluree-db-transact/src/generate/flakes.rs | 43 +- fluree-db-transact/src/import.rs | 45 +- fluree-db-transact/src/import_sink.rs | 223 ++- fluree-db-transact/src/lower_cypher_update.rs | 43 +- fluree-db-transact/src/lower_sparql_update.rs | 349 +++-- .../src/parse/edge_annotations.rs | 172 ++- fluree-db-transact/src/parse/jsonld.rs | 34 +- fluree-db-transact/src/stage.rs | 989 ++----------- fluree-db-transact/tests/it_lower_cypher.rs | 45 +- 53 files changed, 1716 insertions(+), 4778 deletions(-) delete mode 100644 fluree-db-server/tests/export_scan_source.rs diff --git a/docs/cli/export.md b/docs/cli/export.md index e1a94a3776..9b523f8870 100644 --- a/docs/cli/export.md +++ b/docs/cli/export.md @@ -22,7 +22,7 @@ fluree export [LEDGER] [OPTIONS] | `--all-graphs` | Export the default graph plus every named graph (dataset export). Requires `--format trig` or `--format nquads`. The ledger's system graphs are excluded — see `--system-graphs`. | | `--system-graphs` | Also emit the ledger's system graphs (`#txn-meta`, `#config`) under `--all-graphs`. Diagnostic only. | | `--graph ` | Export a specific named graph by IRI. Mutually exclusive with `--all-graphs`. | -| `--raw-reifies` | Emit edge annotations as raw `f:reifies*` system triples instead of RDF 1.2 annotation syntax (pre-4.2 output). | +| `--raw-reifies` | Write each edge annotation as its stored link, `r rdf:reifies <<( s p o )>>`, instead of annotation syntax on the base edge. JSON-LD keeps `@annotation`. | | `--context ` | JSON-LD context for prefix declarations. Overrides the ledger's default context. | | `--context-file ` | Read context from a JSON file. Overrides the ledger's default context. | | `--at

# required, IRI ref -_:ann f:reifiesObject # required, any FlakeValue -_:ann f:reifiesDatatype

# optional (JSON-LD lowering omits; arena recovers from f:reifiesObject's flake-level dt) -_:ann f:reifiesLang "fr" # optional, only for langString objects -_:ann f:reifiesListIndex 3 # v1: always omitted, deferred -``` - -Rules: - -- **Bundle flakes share a graph.** Every `f:reifies*` flake in one bundle has the same flake-level `g`. The decoder rejects mixed-graph bundles outright (`MixedFlakeGraphs`). -- **`f:reifiesGraph` value must match the bundle's flake-level `g`.** Named-graph edges carry `f:reifiesGraph`; default-graph edges omit it. Disagreement is `GraphMismatch`. -- **The base-edge cascade is a Fluree rule, not an entailment.** Retracting a base edge retracts the attachment of every reifier on it, including reifiers the transaction never named. RDF 1.2 defines no entailment from deleting an asserted triple to deleting triples that reify it, and SPARQL 1.2 Update §3.1.2 Example 6 makes the converse point explicitly. The rationale is that an edge's claims should not outlive the edge; the user-facing statement of the divergence, and the spelling table it produces, live in [the concept doc](../concepts/edge-annotations.md#which-spelling-does-what). Note the *entry point* is the opposite case: that `DELETE DATA { s p o ~ :r {| … |} }` retracts the base edge is spec-mandated, since the annotation form asserts the triple. -- **Reserved predicates are firewalled at application write surfaces.** `is_reserved_reifies_predicate(sid)` rejects user-authored `f:reifies*` mention in JSON-LD insert/update/upsert/where+delete+insert and SPARQL UPDATE (all clauses). Bulk import is an administrative bootstrap path and may ingest already-lowered `f:reifies*` bundles; import records `has_annotations=true`, and the indexer builds the arena from those durable facts. -- **Replay validation skips malformed bundles.** A partial bundle (missing required slot), a duplicate slot, or a graph-mismatch is skipped at warmup / arena-build with a `tracing::warn!` and counted in `AttachmentNovelty::observed_malformed_bundle_count()`. The annotation's *non*-`f:reifies` metadata facts remain visible as ordinary RDF; only the attachment binding is lost. -- **Cross-commit multi-target is detected at arena build.** A single annotation SID can reify two different edges across *separate commits* (each bundle is individually well-formed, so the per-bundle decode above can't catch it — the bundles are grouped by `(graph, ann_sid, t, op)`). The transaction path rejects this at stage time (see [Stage-time invariants](#stage-time-invariants)), so it only arises from malformed bulk-import data. `build_arenas_from_flakes` / `build_arenas_from_event_pairs` count annotations whose *net live* state resolves to more than one edge and surface the count on `ArenaBuildOutput.multi_target_annotations` with a `tracing::warn!`. The arena is event-sourced, so the affected reverse lookup returns multiple live edges for those annotations; the count is a data-quality signal, not a rejection (import does not validate input). - -## EdgeKey - -The arena keys on `EdgeKey`, which captures the edge identity from a flake minus the `t/op/m`-bookkeeping that's tracked separately on attachment rows. - -```rust -pub struct EdgeKey { - pub g: Option, // None = default graph - pub s: Sid, - pub p: Sid, - pub o: FlakeValue, // refs, literals, langStrings — full canonical value - pub dt: Sid, - pub lang: Option, - pub list_i: Option, // v1: always None, reserved for list-occurrence annotations -} -``` - -`EdgeKey::from_reifies_facts(&[Flake]) -> Result` decodes a bundle and enforces the rules listed above. `EdgeKey::to_reifies_facts(ann, t, op)` emits the full bundle including `f:reifiesDatatype`; `EdgeKey::to_reifies_facts_jsonld_compatible` omits `f:reifiesDatatype` so the inverse retract bundle matches the assertion shape produced by the JSON-LD lowering (asserting one shape and retracting the other would leave a phantom retract). - -## Sidecar arena layout - -The arena lives in `fluree-db-binary-index/src/annotation_arena/` and mirrors the dictionary CAS trees: - -- **Forward arena** — `EdgeKey -> annotation subjects`. Branches range-route on `EdgeKey`; leaves store sorted `(EdgeKey, ann_sid, t, op)` rows. Used by inline annotation queries and base-edge retract cascade. -- **Reverse arena** — `annotation subject -> EdgeKey`. Branches range-route on `ann_sid`; leaves store sorted `(ann_sid, EdgeKey, t, op)` rows. Used by `@reifies` / `rdf:reifies` annotation-rooted queries and by-id retract. - -Both arenas are content-addressed, immutable, range-routable, lazily loaded. Magic numbers `EAFB1`/`EAFL1` and `EARB1`/`EARL1` mark forward/reverse branch/leaf blobs. - -Per-edge multiplicity is preserved: the forward arena is a multimap on `EdgeKey`. Two `@annotation` blocks against the same base edge with anonymous subjects produce two distinct rows; two with the same explicit `@id` produce one (idempotent). - -## `AnnotationStats` and planner integration - -Each arena seal computes per-slot stats consumed by the planner's cardinality estimator: +An annotation is stored as RDF 1.2 defines it: the reifier's link to the triple it reifies, plus the reifier's own properties. ```text -forward_rows, reverse_rows # total event rows across history -distinct_edges, distinct_annotations # live counts -live_attachment_pairs # number of currently-asserted (edge, ann) pairs -distinct_reified_{subjects,predicates,objects} -reifies_graph_rows, distinct_reified_graphs, distinct_graph_anns -reifies_lang_rows, distinct_reified_langs, distinct_lang_anns +ex:alice ex:knows ex:bob . # the base edge +ex:claim1 rdf:reifies <<( ex:alice ex:knows ex:bob )>> . # the link +ex:claim1 ex:confidence 0.8 . # the body ``` -`stats_view::merge_annotation_stats` overlays these onto the regular `IndexStats.properties` HLL for the seven `f:reifies*` predicates so the planner gets sharp selectivity for `?ann f:reifies* `-shape probes. `f:reifiesDatatype` is intentionally not synthesized from the arena — the arena reconstructs `dt` from the `f:reifiesObject` row's flake-level dt and can't tell whether the on-wire bundle emitted a separate `f:reifiesDatatype` flake. The regular HLL is the source of truth for that slot. +The link is an ordinary flake whose object is a triple term (`FlakeValue::TripleTerm`, datatype `f:tripleTerm`). It rides the same pipeline as every other flake — commits, replay, history, policy, time travel — and lives in the graph of the edge it names. Nothing about it is annotation-specific below the write surfaces and the term dictionary. -Every field is `#[serde(default)]` so older arena roots that predate any given field deserialize cleanly. A zero NDV means "no information"; the planner falls back to `IndexStats.properties`. +A triple term carries its object's datatype and language tag, so `"chat"@fr` and `"chat"@en` are different terms; the term is the edge identity that cascade, hydration and export compare on. -## IndexRoot signals and the sticky bit +## Writers -`IndexRoot` carries three coordinated signals around annotations: +Every write surface produces the link directly: -| Signal | Meaning | -|---|---| -| `annotation_index: Option` | When `Some`, forward + reverse arena CIDs and stats are loaded lazily. | -| `has_annotations: bool` | Sticky flag: true once any `f:reifies*` SID has appeared in the predicate dictionary. Drives the cascade fast-path's zero-cost gate. | -| `had_annotation_arena: bool` | **Sticky bit, never cleared.** Lives in the FIR6 extended-flags byte (low byte of the historically-zero `pad(2)` header field). | +- **Turtle / TriG / N-Quads** (`FlakeSink`, `ImportSink`, TriG import): `reified_triple_link` builds the link flake for each `~ r`, `{| … |}`, `<< s p o ~ r >>` and `r rdf:reifies <<( s p o )>>`; the parser emits the base triple itself. +- **SPARQL UPDATE**: `expand_annotated_triples` desugars an annotation tail as the spec does, to the base triple plus `r rdf:reifies <<( s p o )>>`. Template lowering turns the triple term into `TemplateTerm::TripleTerm`, whose positions resolve per solution; WHERE lowering turns it into the reifier pattern queries use. +- **Cypher**: `CREATE (a)-[r:T]->(b)` writes `a T b` and `r rdf:reifies <<( a T b )>>` with `r` a fresh reifier. +- **JSON-LD**: JSON-LD has no triple-term syntax, so `@annotation` / `@edge` lower (before expansion) to `f:reifies*` slot keys on a sibling node. After parsing, `fold_slots_into_links` turns each annotation's slots into its link template, and bulk import's `ImportSink` does the same with the slot triples it receives. The slots are an intermediate form and are never stored. A delete-by-selector matches the existing link with the query-side `@reifies` form. -The truth table for the indexed-arena guarantee: +The `f:reifies*` predicates are reserved: the write surfaces reject user-authored ones. -| `has_annotations` | `annotation_index` | Meaning | -|---|---|---| -| `false` | `None` | Hard guarantee: zero attachments. Cascade and reads short-circuit. | -| `true` | `Some(_)` | Builder ran. Forward/reverse arenas are authoritative for `t ≤ max_t`; novelty supplies the tail. | -| `true` | `None` | Pre-builder or defensive-drop transitional state. Snapshot may carry `f:reifies*` flakes but no arena yet — readers fall back to scan, cascade still runs. | -| `false` | `Some(_)` | **Invariant violation.** The encoder coerces `has_annotations=true` whenever an arena is present, so this state never reaches the wire. The `encode()` `debug_assert!` catches in-memory regressions in dev/CI. | +## Reading a link: the term dictionary -### Sticky-bit state machine +The index stores a link's object as a triple-term handle, `(inner p_id << 32) | seq`, interned in the term dictionary with its components. The dictionary keeps two reverse trees, subject-first and object-first, so a pattern with a bound subject or object finds its terms without reading every link. Live link counts per inner predicate go in `IndexStats.links` for the planner. -`had_annotation_arena`'s load-bearing role is **"base-index bootstrap is not allowed"**. The provider's one-time PSOT scan of `f:reifies*` flakes (`ApiAttachmentEventsProvider::scan_base_index_for_attachment_events`) is correct only when the base index is the complete history. Any indexer pass that's already touched the annotation history owns it from then on; the provider must not later reconstruct a live-only `Authoritative` arena from such a root. +A decimal, big-integer or vector object is keyed in a term by the string id of its canonical form (`lexical_term_object`), not by an arena handle, which names a value only within one graph and predicate. -The bit is set in three places, all flipping it to `true` and never clearing: +Query lowering of annotation patterns is described in [Annotation patterns read the link](#annotation-patterns-read-the-link). -- `IncrementalRootBuilder::build()` — coerces on any incremental root with `has_annotations=true`, regardless of whether this pass sealed an arena (covers no-provider passes that left `annotation_index=None`). -- `encode_and_write_root_v6` — coerces in the full-rebuild root-assembly path, same condition. -- Decoder coercion — `IndexRoot::decode` and `LedgerSnapshot::from_root_bytes` both coerce the bit from `annotation_index.is_some()` when the wire byte is zero, covering legacy pre-extended-flags roots. +## Transaction-time rules -Only fresh bulk-import roots leave it `false`. Bulk import constructs `IndexRoot` directly in `fluree-db-api/src/import.rs`, bypassing both root-assembly paths. That makes `has_annotations=true, annotation_index=None, had_annotation_arena=false` the unique bootstrap-eligible state. +`cascade_attachment_retracts` keeps links pointing at live edges: -The bootstrap itself runs from `Fluree::reindex` even when the client has no `LedgerManager` (the CLI builds its client `without_ledger_caching()`, so `fluree create --from` and `fluree reindex` have no `AttachmentEventsProvider` to ask). When no provider resolves coverage, `reindex` derives it from the `LedgerState` it already loads (`attachment_events_from_state`): the novelty's attachment events under the same `snapshot.t == 0 → Authoritative, else Augment` rule the provider uses, and the base-index scan when the overlay holds no events and the snapshot is bootstrap-eligible. Before this the indexer received `None`, sealed nothing, and `fluree create` still printed "Annotation arena sealed"; `create` now reports what the reloaded snapshot actually carries. +1. A retracted triple retracts every link naming it (a POST probe on `rdf:reifies` with the term as the object). +2. A transaction that retracts all of a reifier's body retracts its links. +3. A reifier left with no link loses its body when it is a blank node, or in LPG mode (`opts.lpgEdgeLifecycle`, which Cypher `DELETE` sets); an IRI reifier's body otherwise stays as ordinary RDF. -### Provider 4-gate eligibility +The cascade is a Fluree rule, not an entailment: RDF 1.2 does not delete a reifier's statements when the triple it reifies is deleted. Annotation-syntax reads rely on it (see below). A ledger that has never held an annotation pays nothing: `snapshot.has_annotations` and `Novelty::has_annotations` gate the pass. -`ApiAttachmentEventsProvider::attachment_events` boots a one-time base-index scan only when **all four** hold: +A reifier may reify several triples. Re-pointing one is a retract of the old link and an assert of the new; a JSON-LD upsert does that for a reifier it names. In LPG mode, an empty `@annotation: {}` mints a fresh property-less reifier so the relationship keeps an identity; in RDF mode it writes nothing. -1. `try_running_attachment_events` returns an empty event set (the running overlay carries no `f:reifies*` events for this ledger). -2. `snapshot.has_annotations` is true. -3. `snapshot.annotation_index.is_none()` (no arena currently sealed). -4. `!snapshot.had_annotation_arena` (the indexer has never touched this ledger's annotation history). - -When eligible, the provider walks the base index via per-predicate PSOT scans (PSOT-direction sidesteps a SPOT quirk where constant blank-node subjects don't return rows reliably across all backends), groups results by annotation SID, decodes each bundle into an `EdgeKey`, and returns `AttachmentEventCoverage::Authoritative`. The indexer then seals an authoritative arena from the events. Any subsequent indexer pass coerces `had_annotation_arena = true`, so the same ledger is no longer bootstrap-eligible — defensive drops carry the bit forward and stay in the M2a scan-fallback state until the next provider-backed reindex re-seals. - -PSOT scan errors are logged with `tracing::warn!` and surface as `Option::None` (no coverage); the indexer then skips arena seal this pass instead of silently treating the failure as "no annotations." - -## Reads merge arena + novelty under one visibility pass - -`AnnotationArenaReader` exposes two merged lookups: - -```rust -fn current_annotations_merged(edge: &EdgeKey, novelty_events: &[(Sid, i64, bool)], as_of_t: i64) -> Vec -fn current_targets_merged(ann: &Sid, novelty_events: &[(EdgeKey, i64, bool)], as_of_t: i64) -> Vec -``` - -Callers (the hydration injector and the cascade pass) feed the arena's `collect_*_events` output into a single `(t, op)` visibility pass that resolves to currently-live state. Arena and novelty events can interleave arbitrarily — the merge applies one latest-wins pass over the union, so an arena `op = true` followed by a novelty `op = false` (or vice versa) resolves correctly without the caller doing any pre-merging. - -The scan fallback (used when no arena is sealed) runs `db.range(POST, f:reifiesSubject, edge.s)` to find candidate annotation SIDs, then walks each candidate's bundle via `db.range(SPOT, s=ann_sid)` for structural decode. Bundle decode at the SPOT step bypasses view policy: the `f:reifies*` flakes are system-controlled discriminators, not user data, and policy-filtering them at decode would let an incidental policy that hides FLUREE_DB-namespace predicates collapse the bundle and drop the annotation entirely. The annotation **body** continues through the policy-filtered `format_subject` path, so user-data visibility is unchanged. - -## Stage-time invariants - -- **Cascade dedup is graph-aware.** Cascade retracts are deduped against existing retracts in the staged flake set via an explicit key tuple `(Option, Sid, Sid, FlakeValue, Sid, Option)` — `Flake`'s `Eq`/`Hash` ignore `g`, so a plain `HashSet` would collapse two retracts targeting the same `(s, p, o, dt, m)` in different named graphs. -- **Single-target invariant (Fluree v1 storage contract).** RDF 1.2 allows one reifier to be related to several propositions; this implementation deliberately stores one live edge attachment per annotation SID so retract cascade, by-id cleanup, and the forward/reverse arena stay unambiguous. At stage time, for every annotation SID this transaction asserts a `f:reifies*` flake for, the *net* asserted bundle — current snapshot/novelty state, minus this txn's retracts, plus its asserts, deduped as an RDF set keyed on `(g, s, p, o, dt)` — is decoded via `EdgeKey::from_reifies_facts`; a malformed or multi-target result (e.g. a duplicate subject/predicate/object slot) errors with `InvariantViolation`. Validating the *net* bundle rather than counting this txn's `f:reifiesSubject` flakes catches two cases a count misses: re-pointing an `@id` already attached to a different edge in a *prior* transaction (no retract in this txn), and same-subject/different-slot multiplicity within one txn (the subject slot dedupes while predicate/object diverge). The legitimate re-point pattern — retract the old attachment + assert the new — passes because the net bundle resolves to one edge; an idempotent re-assert passes because the set-keyed net bundle is unchanged. -- **Empty `@annotation: {}` is RDF-mode no-op.** No annotation subject is minted; no attachment row is written. In LPG mode (`opts.lpgEdgeLifecycle: true`), an empty block mints a fresh property-less annotation subject so the relationship retains identity. -- **Assertion lowering scopes synthesized siblings per clause.** The standard `@annotation` lowering emits the `f:reifies*` bundle as sibling top-level nodes. For wrapper documents (`{where, insert, delete, …}`) each clause's siblings are attached to *that clause's own value*, not the top-level wrapper. Flushing them onto the wrapper would rewrap `{where, insert}` into `{@graph: […]}`, after which `parse_update` finds no clauses and silently no-ops the whole transaction — the failure mode the per-clause scoping prevents. The blank-node counter stays shared across clauses so anonymous-annotation ids don't collide. - -## Cascade and lifecycle modes - -The cascade gate is mode-independent: when both `snapshot.has_annotations` and `novelty.attachments.has_annotations()` are false, base-edge retracts skip the cascade lookup entirely (zero-cost gate, non-annotation ledgers pay nothing). - -When annotations are possible: - -- **Plain edge retract.** Look up annotations attached to the retracted edge via merged arena + novelty. Retract the complete `f:reifies*` bundle for each. Retract anonymous annotation body metadata always (RDF default). Retract explicit-IRI body metadata only when `opts.lpgEdgeLifecycle: true`. -- **By-id retract.** A delete-clause pre-pass (`lower_delete_annotation_blocks`) rewrites `@annotation: { @id ex:foo }` into explicit `f:reifies*` retract templates before the standard lowering. Threads named-graph context (`f:reifiesGraph` + node-level `@graph` selector) so retracts cancel the correct asserted flake identity. -- **By-selector retract.** A WHERE-bound retract: the pre-pass mints a fresh internal variable, synthesizes a `f:reifies*` WHERE pattern that constrains the variable to live annotations matching the selector body, and emits a by-variable delete template. The mint counter seeds past any user-visible `?_fluree_del_ann_N` occurrence to avoid collision. - -## GC reachability +## Annotation patterns read the link -Annotation forward and reverse branch CIDs are returned by `IndexRoot::all_cas_ids`, so they participate in the same root-chain reachability model as fact indexes and dictionary trees. When a new index root supersedes an old one: +Every annotation pattern reads the link. Reified-triple patterns (`<< s p o ~ ?r >>`, `?r rdf:reifies <<( s p o )>>`, JSON-LD `@reifies`) lower to it directly: `lower_reified_link` (`fluree-db-query/src/ir/term_components.rs`), shared by the SPARQL and JSON-LD lowerings, emits the reifier's `rdf:reifies` link, `?r rdf:reifies ?t`, plus `TermComponents(?t, s, p, o)` relating the term to its components and a `sameTerm` filter per constant component, which the planner turns into the scan's handle interval. A fully constant edge composes to a constant term instead. -- New annotation CIDs are reachable from the new root. -- Old annotation CIDs become garbage only when no retained root references them. -- Leaf-level CIDs behind a branch are walked during the GC-diff pass; `drop.rs` and `gc/collector.rs` both call into the expanded-CAS expansion helpers so a strict GC pass never deletes a still-reachable leaf. +Annotation syntax (`s p o ~ ?r {| ... |}`, JSON-LD `@annotation`, Cypher relationship properties) keeps its `Pattern::EdgeAnnotation` through lowering, and `expand_edge_annotation_patterns_for` (`where_plan.rs`) expands it at planning into the body and the link with its term components (`link_patterns`). The base edge is not joined: every write that attaches a reifier asserts its edge in the same graph and commit, and retracting the edge retracts the link, so a live link names a live edge. A reifier written as `r rdf:reifies <<( s p o )>>` without asserting `s p o` ends that invariant, and the check then has to come back as a join. The chain is wrapped in `Pattern::DefaultGraphSource` only when the default graph is a union of two or more graphs (`PlanningContext::default_graph_union`). -## Annotation patterns read the link +The link names its triple without joining the base edge, so visibility is checked on the term: `QueryPolicyEnforcer` lets a flake whose object is a triple term through only when the triple that term names would be visible, recursively for nested terms. A policy hiding `ex:worksFor` therefore hides the links to `ex:worksFor` edges on every route. -Every annotation pattern reads the link. Reified-triple patterns (`<< s p o ~ ?r >>`, `?r rdf:reifies <<( s p o )>>`, JSON-LD `@reifies`) lower to it directly: `lower_reified_link` (`fluree-db-query/src/ir/term_components.rs`), shared by the SPARQL and JSON-LD lowerings, emits the reifier's `rdf:reifies` link, `?r rdf:reifies ?t`, whose object is a triple-term handle the index's term dictionary resolves, plus `TermComponents(?t, s, p, o)` relating the term to its components and a `sameTerm` filter per constant component, which the planner turns into the scan's handle interval. A fully constant edge composes to a constant term instead. +Wildcard scans (`?s ?p ?o`, wildcard hydration, Cypher property maps) hide `rdf:reifies`; `opts.includeSystemFacts` shows it. -Annotation syntax (`s p o ~ ?r {| ... |}`, JSON-LD `@annotation`, Cypher relationship properties) keeps its `Pattern::EdgeAnnotation` through lowering, with the term variable reserved there, and `expand_edge_annotation_patterns_for` (`where_plan.rs`) expands it at planning into the body and the link with its term components (`link_patterns`). The base edge is not joined: every write that attaches a reifier asserts its edge in the same graph and commit, and retracting the edge retracts the attachment, so a live link names a live edge, and a policy that hides the edge hides the link. Joining it probed the edge once per annotation (an annotation-syntax count over every annotated edge of the StarBench dev slice took 11 s against 0.8 s on the bundle chain, which elided the same check). A reifier that does not assert its triple ends that invariant, and the check then has to come back as a join. The order is the planner's tie-break: the body and the link are probes by reifier and the components decode from the bound term, so a range filter on the body drives the scan when nothing is more selective. The chain joins its enclosing block, wrapped in `Pattern::DefaultGraphSource` only when the default graph is a union of two or more graphs (`PlanningContext::default_graph_union`). The runner encodes the edge's IRIs first, since planning has no ledger to encode against; an IRI still unencoded is compared as the query runs. +Hydration (`@annotation` in subject expansion) and export read links the same way: hydration probes `rdf:reifies` with the rendered edge's term; export reads the ledger's live links once and writes a `~ ` marker on each base edge it reaches. -The link names its triple without joining the base edge, so visibility is checked on the term: `QueryPolicyEnforcer` lets a flake whose object is a triple term through only when the triple that term names would be visible, recursively for nested terms. A policy hiding `ex:worksFor` therefore hides the links to `ex:worksFor` edges on every route, not only on annotation-syntax patterns. +## Ledgers written before links -Links exist only in indexes built since the link form. An annotated ledger (`has_annotations`) whose root has no term dictionary predates them, and its annotations live only in the `f:reifies*` bundles: link reads there (`BinaryScanOperator` on `rdf:reifies`, `TermComponentsOperator`) fail asking for a rebuild rather than answering without those annotations, and an incremental build over such a root leaves link synthesis off, so it never acquires a term dictionary that covers only its window. A full rebuild interns a term for every attachment in history and links them all. +Earlier releases stored an annotation as an `f:reifies*` bundle (`f:reifiesSubject`, `f:reifiesPredicate`, `f:reifiesObject`, …) on the reifier. Those bundles stay in commits, and the link is derived from them wherever it is needed: an index build derives it from each commit's bundle ops (`link_synth`), and novelty derives it for commits the index has not covered (`fluree-db-novelty/src/links.rs`). Readers see only links. A write that retracts a derived link leaves the bundle in place; the retract cancels the derived assert in novelty and at the next build alike. -A rel var whose properties are read anywhere (`r.prop`, `properties(r)`, …) lowers to a REQUIRED annotation pattern — unreified edges do not match, per Cypher's relationship semantics. Value-only rel vars keep the plain base triple + an OPTIONAL annotation pattern (batched as a hash left-join) so unreified edges match with a synthesized relationship value. +An annotated index built before links has no term dictionary. Link reads on it fail asking for a rebuild (`fluree reindex`) rather than answering without its annotations. An incremental build over it declines once its window holds a triple term — a term dictionary covering the window alone would lift that refusal — and the index build falls back to a full rebuild, which links every annotation in history. ## See also - [Edge annotations (concept doc)](../concepts/edge-annotations.md) — the user-facing surface. -- [Index format](index-format.md) — fact indexes, dictionary trees, FIR6 root layout. -- Rustdoc on `fluree_db_core::edge::EdgeKey`, `fluree_db_core::annotation_index::AnnotationIndexRoot`, `fluree_db_novelty::attachments::AttachmentNovelty`, and `fluree_db_binary_index::annotation_arena::AnnotationArenaReader` for type-level contracts. +- [Index format](index-format.md) — fact indexes, dictionary trees, root layout. diff --git a/docs/design/rules-engine.md b/docs/design/rules-engine.md index 3a626b3410..677600fc7a 100644 --- a/docs/design/rules-engine.md +++ b/docs/design/rules-engine.md @@ -174,7 +174,7 @@ cannot carry provenance. |---|---|---|---|---| | A. No minting (status quo) | Provenance modelled as extra derived triples on the endpoints | n/a | none | Cannot attach provenance to the edge itself; a second rule per fact | | B. Deterministic anonymous reifier | Head `@annotation` block mints one reifier per derived triple with id = skolem hash of (rule id, instantiated s p o); body from constants and body vars | Stable: same inputs, same id | Head instantiation emits the `f:reifies*` bundle plus body facts into the overlay; the annotation lanes already read overlays | Bundle size (4–6 flakes per derived edge) multiplies derived-fact counts; budgets must count them | -| C. Reuse a body-bound reifier | `@annotation: {"@id": "?claim"}` re-attaches the supporting claim to the derived edge | Stable | Cheapest | Violates the single-target invariant (one reifier, one live edge); rejected on that ground | +| C. Reuse a body-bound reifier | `@annotation: {"@id": "?claim"}` re-attaches the supporting claim to the derived edge | Stable | Cheapest | Was rejected under the former single-target rule (one reifier, one live edge); RDF 1.2 allows it, so this row is open again | | D. System provenance metadata | Each derived flake carries the rule id in flake metadata; a query-side function exposes it | Stable | Smallest storage | New query surface, invisible to RDF tooling; cannot carry rule-supplied properties like confidence | Recommendation: B, as a follow-up once this engine has landed, with the head diff --git a/docs/operations/telemetry.md b/docs/operations/telemetry.md index b08ca7e433..a54970c653 100644 --- a/docs/operations/telemetry.md +++ b/docs/operations/telemetry.md @@ -305,15 +305,12 @@ query_execute (debug) │ ├── sort_blocking (debug, cross-thread via spawn_blocking) │ └── ... └── format (debug) - └── inject_annotations (debug, edge_in_named_graph, path, annotation_count) - └── annotation_arena_lookup (debug, live_count) ← path = "arena" only + └── inject_annotations (debug, edge_in_named_graph, annotation_count) ``` -`inject_annotations` and `annotation_arena_lookup` fire only on -hydration responses that surface annotation bodies; both are skipped -on non-annotation ledgers via the formatter's zero-cost gate. -`path` is `"arena"` when the cached `AnnotationArenaReader` resolved -the lookup, `"scan"` when the M2a POST-scan fallback ran. +`inject_annotations` fires only on hydration responses that probe an +edge for annotations, reading the reifiers' `rdf:reifies` links; it is +skipped on non-annotation ledgers via the formatter's zero-cost gate. #### Span Tree (Multi-query envelope) diff --git a/docs/reference/vocabulary.md b/docs/reference/vocabulary.md index 4c69bc8c2f..886ff0148a 100644 --- a/docs/reference/vocabulary.md +++ b/docs/reference/vocabulary.md @@ -74,7 +74,7 @@ Commit subjects use the scheme `fluree:commit:` (e.g. `fluree:commit ## Edge-annotation predicates (reserved) -These seven predicates encode the edge that an [edge annotation](../concepts/edge-annotations.md) reifies. They are the durable, system-controlled representation behind the `@annotation` / `@reifies` (JSON-LD) and `{| ... |}` / `~` / `rdf:reifies` (SPARQL 1.2) surfaces. Together they form a *reifies bundle* on the annotation subject. +Earlier releases stored the edge an [edge annotation](../concepts/edge-annotations.md) reifies as these seven predicates, a *reifies bundle* on the annotation subject. An annotation is now stored as its `rdf:reifies` link, `r rdf:reifies <<( s p o )>>`; ledgers written before that still hold bundles, which are read as their links. | Predicate | Full IRI | Datatype | Description | |-----------|----------|----------|-------------| @@ -86,7 +86,7 @@ These seven predicates encode the edge that an [edge annotation](../concepts/edg | `f:reifiesLang` | `https://ns.flur.ee/db#reifiesLang` | `xsd:string` | BCP-47 language tag, present only when the object is a language-tagged string. **Optional.** | | `f:reifiesListIndex` | `https://ns.flur.ee/db#reifiesListIndex` | `xsd:int` | List-occurrence index. **Reserved/deferred** — always omitted in this release. | -**These predicates are reserved.** User-authored mention of any `f:reifies*` IRI (compact or full form) is rejected on every transaction write surface (JSON-LD insert/upsert/update, SPARQL UPDATE, Turtle insert/upsert), and they are filtered out of variable-predicate (`?p`) scans and wildcard (`select: "*"`) hydration so they never surface as ordinary RDF. Mint and manage annotations only through `@annotation` / `@edge` (JSON-LD) or the annotation tail (`{| ... |}` / `~` / `rdf:reifies`) in SPARQL 1.2 and Turtle. Bulk `import` is the one administrative exception: it ingests already-lowered bundles without this firewall. See [Edge annotations](../concepts/edge-annotations.md) for the full surface and the [storage-internals design doc](../design/edge-annotations.md) for the bundle encoding and invariants. +**These predicates are reserved.** User-authored mention of any `f:reifies*` IRI (compact or full form) is rejected on every transaction write surface (JSON-LD insert/upsert/update, SPARQL UPDATE, Turtle insert/upsert), and they are filtered out of variable-predicate (`?p`) scans and wildcard (`select: "*"`) hydration so they never surface as ordinary RDF. Mint and manage annotations only through `@annotation` / `@edge` (JSON-LD) or the annotation tail (`{| ... |}` / `~` / `rdf:reifies`) in SPARQL 1.2 and Turtle. Bulk `import` is the one exception: it reads bundles from an export written before links and stores each annotation's link instead. See [Edge annotations](../concepts/edge-annotations.md) for the full surface and the [storage-internals design doc](../design/edge-annotations.md) for the storage. --- diff --git a/docs/transactions/turtle.md b/docs/transactions/turtle.md index 5ffe8fd3fe..79039b7c92 100644 --- a/docs/transactions/turtle.md +++ b/docs/transactions/turtle.md @@ -457,7 +457,7 @@ ex:dataset-import-2024-01-22 a ex:DatasetImport ; ## Edge annotations (RDF 1.2 / Turtle-star) -The Turtle parser (which also reads N-Triples) accepts the RDF 1.2 *asserting* forms on every Turtle write path — `insert`, `upsert`, bulk `import`, `fluree graph sync`, and the memory importer. All of them produce the same on-disk `f:reifies*` bundle that the JSON-LD `@annotation` and SPARQL 1.2 `{| |}` surfaces write, so cascade retracts, hydration, and the annotation arena treat every surface as one, and the annotations are queryable from every query surface: +The Turtle parser (which also reads N-Triples) accepts the RDF 1.2 *asserting* forms on every Turtle write path — `insert`, `upsert`, bulk `import`, `fluree graph sync`, and the memory importer. All of them store the same `rdf:reifies` link that the JSON-LD `@annotation` and SPARQL 1.2 `{| |}` surfaces write, so cascade retracts and hydration treat every surface as one, and the annotations are queryable from every query surface: ```turtle @prefix ex: . @@ -527,7 +527,6 @@ Rejected with a clear parse or stage error, never silently dropped: - the parenthesized triple term `<<( :s :p :o )>>` anywhere other than the object of `rdf:reifies` (RDF 1.2 triple terms as values are not representable yet), and a triple term nested inside another; - an annotation block nested inside an annotation body (`{| :q :v {| … |} |}`), and an annotation tail on an `rdf:reifies <<( … )>>` statement (it would annotate the reification itself); - an annotation on a collection object (`( :a :b ) {| … |}`); -- one named reifier on two different triples — a reifier denotes exactly one edge (see [the single-target invariant](../concepts/edge-annotations.md#one-annotation-one-edge-single-target-invariant)); - an annotation on an `rdf:type` edge (`:s a :C {| … |}`) on the paths that convert Turtle to JSON-LD first (`upsert`, `graph sync`, memory import) — JSON-LD has no place to hang an annotation on a `@type` value. `insert` and SPARQL UPDATE accept it; - TriG: annotations in a `<#txn-meta>` block — its triples become commit metadata, not edges. diff --git a/fluree-db-api/src/admin.rs b/fluree-db-api/src/admin.rs index 4207b84791..952a2c552d 100644 --- a/fluree-db-api/src/admin.rs +++ b/fluree-db-api/src/admin.rs @@ -1256,7 +1256,7 @@ impl crate::Fluree { pub async fn has_annotations(&self, ledger_id: &str) -> Result { let handle = self.ledger_cached(ledger_id).await?; let ledger = handle.snapshot().await.to_ledger_state(); - Ok(ledger.snapshot.has_annotations || ledger.novelty.attachments.has_annotations()) + Ok(ledger.snapshot.has_annotations || ledger.novelty.has_annotations()) } /// Cancel indexing, delete storage artifacts, purge nameservice record, @@ -2227,7 +2227,7 @@ impl crate::Fluree { if indexer_config.attachment_events.is_none() { let ledger_has_annotations = ledger_state .as_ref() - .map(|st| st.snapshot.has_annotations || st.novelty.attachments.has_annotations()) + .map(|st| st.snapshot.has_annotations || st.novelty.has_annotations()) .unwrap_or(false); if ledger_has_annotations { // Caller-supplied provider wins; fall back to the API's diff --git a/fluree-db-api/src/commit_transfer.rs b/fluree-db-api/src/commit_transfer.rs index 169c512332..1e61b5a6c8 100644 --- a/fluree-db-api/src/commit_transfer.rs +++ b/fluree-db-api/src/commit_transfer.rs @@ -1226,11 +1226,7 @@ async fn stage_commit_flakes( ) -> std::result::Result { let mut options = fluree_db_transact::StageOptions::new() .with_index_config(index_config) - .with_graph_sids(graph_sids) - // Push applies commits that were authored and written elsewhere, so - // authoring invariants are advisory here. See - // `StageOptions::replaying_commit`. - .replaying_commit(); + .with_graph_sids(graph_sids); if let Some(policy_ctx) = policy_ctx.filter(|p| !p.wrapper().is_root()) { options = options.with_policy(policy_ctx); } diff --git a/fluree-db-api/src/export.rs b/fluree-db-api/src/export.rs index efd127a93a..976f472f03 100644 --- a/fluree-db-api/src/export.rs +++ b/fluree-db-api/src/export.rs @@ -38,10 +38,9 @@ pub struct ExportConfig<'a> { pub overlay: Option<&'a dyn OverlayProvider>, /// Dictionary novelty for resolving IDs from committed-but-not-yet-indexed transactions. pub dict_novelty: Option<&'a Arc>, - /// Forward annotation lookup. `None` means this export emits the raw - /// `f:reifies*` system facts as ordinary triples (`--raw-reifies`, or a - /// ledger that has never carried an annotation) and the writers run - /// exactly the loop they ran before RDF 1.2 output existed. + /// Edge → reifier lookup. `None` means this export writes each + /// `rdf:reifies` link as an ordinary triple (`--raw-reifies`, or a ledger + /// that has never carried an annotation) rather than as annotation syntax. pub annotations: Option<&'a AnnotationProbe<'a>>, /// SID of the graph being scanned, as `EdgeKey.g` recorded it. `None` for /// the default graph. Only read when `annotations` is `Some`. @@ -392,9 +391,9 @@ impl crate::export_annotations::ReifierSubject for ExportResolver<'_> { /// annotation syntax, so the common path allocates nothing. struct AnnotationContext<'a> { probe: &'a AnnotationProbe<'a>, - /// `p_id`s of the seven `f:reifies*` predicates in this store's id space, - /// persisted and ephemeral. Hoisted out of the row loop: suppression is - /// then a scan of at most fourteen `u32`s, not an IRI comparison. + /// `p_id`s of the seven legacy `f:reifies*` predicates in this store's id + /// space, persisted and ephemeral. Hoisted out of the row loop: suppression + /// is then a scan of at most fourteen `u32`s, not an IRI comparison. reifies_p_ids: Vec, graph_sid: Option, } @@ -428,9 +427,8 @@ impl<'a> AnnotationContext<'a> { } } -/// An `rdf:reifies` link row. Links are derived from the `f:reifies*` bundles -/// (which export writes as annotation syntax) and never enter commits, so -/// export drops them in every mode, `--raw-reifies` included. +/// An `rdf:reifies` link row: replaced by annotation syntax, and written as +/// the triple it is under `--raw-reifies`. #[inline] fn is_link_row(o_type: u16) -> bool { o_type == OType::TRIPLE_TERM.as_u16() @@ -460,11 +458,8 @@ async fn batch_reifiers( let mut edge_row: Vec = Vec::new(); for row in 0..batch.row_count { let p_id = batch.p_id.get_or(row, 0); - if ann.is_reifies_row(p_id) { - continue; // the bundle itself is never an annotated edge - } let o_type = batch.o_type.get_or(row, 0); - if is_link_row(o_type) { + if ann.is_reifies_row(p_id) || is_link_row(o_type) { continue; } let o_key = batch.o_key.get(row); @@ -505,7 +500,7 @@ async fn batch_reifiers( }); edge_row.push(row); } - let per_edge = ann.probe.live_reifiers(&edges).await?; + let per_edge = ann.probe.live_reifiers(g_id, &edges); let mut out = vec![Vec::new(); batch.row_count]; for (i, row) in edge_row.into_iter().enumerate() { out[row] = per_edge[i].clone(); @@ -563,7 +558,7 @@ pub async fn export_graph_turtle( // to rather than appended after the stream, so a subject never opens twice // (see `UntranslatedBySubject`). let (untranslated, untranslated_reifiers) = - resolve_untranslated(ann.as_ref(), untranslated).await?; + resolve_untranslated(ann.as_ref(), untranslated, config.g_id).await?; let mut untranslated = UntranslatedBySubject::new(store, untranslated); while let Some(batch) = cursor.next_batch()? { @@ -620,27 +615,14 @@ pub async fn export_graph_turtle( Ok(stats) } -/// Split untranslated overlay rows into the bundle rows annotation syntax +/// Split untranslated overlay rows into the link rows annotation syntax /// replaces and the base rows that may carry a marker, resolving every -/// reifier in one probe call. -/// -/// Untranslated rows never pass through `is_reifies_row` — only the -/// translated writers call it — so before this they reached the output raw: -/// a *partial* `f:reifies*` bundle (the rows that did translate were -/// suppressed) and no `~ ` on the edge it described. Round-tripping that -/// file plants a reserved predicate in the target ledger as ordinary data. -/// -/// Filtering them alone would have been worse than the leak. The unresolved -/// counter only moves where the translated path calls `note_bundle_in_scope`, -/// so a silent filter converts a visible wrong answer into an invisible one. -/// Suppression and accounting are the same change. -/// -/// One `live_reifiers` call for the whole untranslated set rather than one -/// per row: `batch_reifiers` is already a per-row probe on annotated ledgers, -/// and stacking a second one is the wrong direction for that cost. +/// reifier in one probe call. Suppressed links are noted in scope, as the +/// translated writers note theirs, so the unresolved count stays honest. async fn resolve_untranslated( ann: Option<&AnnotationContext<'_>>, rows: Vec, + g_id: GraphId, ) -> io::Result<(Vec, HashMap>)> { let Some(ann) = ann else { // `--raw-reifies` and annotation-free ledgers want the rows verbatim. @@ -648,14 +630,17 @@ async fn resolve_untranslated( }; let mut base: Vec = Vec::with_capacity(rows.len()); for f in rows { + if fluree_db_core::is_rdf_reifies(&f.p) { + ann.probe.note_link_sid(f.s.clone()); + continue; + } if fluree_db_core::namespaces::is_reserved_reifies_predicate(&f.p) { - ann.probe.note_bundle_sid(f.s.clone()); continue; } base.push(f); } let keys: Vec = base.iter().map(EdgeKey::from_flake).collect(); - let live = ann.probe.live_reifiers(&keys).await?; + let live = ann.probe.live_reifiers(g_id, &keys); let mut map: HashMap> = HashMap::new(); for (key, reifiers) in keys.into_iter().zip(live) { if !reifiers.is_empty() { @@ -693,17 +678,14 @@ fn write_turtle_batch( let p_id = batch.p_id.get_or(row, 0); let o_type = batch.o_type.get_or(row, 0); let o_key = batch.o_key.get(row); - if is_link_row(o_type) { - continue; - } - - // The `f:reifies*` bundle is the on-disk encoding of an annotation, - // not a triple the ledger was asked to hold. It is replaced by the - // `~ ` markers emitted below, and re-emitting it too would produce - // a file the write path refuses to ingest. + // Annotation syntax replaces each link with the `~ ` marker + // emitted below; a legacy `f:reifies*` bundle is read as its link. if let Some(ann) = ann { + if is_link_row(o_type) { + ann.probe.note_link_in_scope(resolver, s_id); + continue; + } if ann.is_reifies_row(p_id) { - ann.probe.note_bundle_in_scope(resolver, s_id); continue; } } @@ -881,7 +863,7 @@ pub async fn export_graph_jsonld( let mut current_props: Vec<(String, Vec)> = Vec::new(); let mut first_node = true; let (untranslated, untranslated_reifiers) = - resolve_untranslated(ann.as_ref(), untranslated).await?; + resolve_untranslated(ann.as_ref(), untranslated, config.g_id).await?; let mut untranslated = UntranslatedBySubject::new(store, untranslated); while let Some(batch) = cursor.next_batch()? { @@ -891,13 +873,12 @@ pub async fn export_graph_jsonld( let p_id = batch.p_id.get_or(row, 0); let o_type = batch.o_type.get_or(row, 0); let o_key = batch.o_key.get(row); - if is_link_row(o_type) { - continue; - } - if let Some(ann) = ann.as_ref() { + if is_link_row(o_type) { + ann.probe.note_link_in_scope(&resolver, s_id); + continue; + } if ann.is_reifies_row(p_id) { - ann.probe.note_bundle_in_scope(&resolver, s_id); continue; } } @@ -1514,7 +1495,7 @@ pub async fn export_graph_ntriples( let resolver = ExportResolver::new(store, config.dict_novelty, &ephemeral_preds); let ann = AnnotationContext::new(&resolver, config); let (untranslated, untranslated_reifiers) = - resolve_untranslated(ann.as_ref(), untranslated).await?; + resolve_untranslated(ann.as_ref(), untranslated, config.g_id).await?; let mut stats = ExportStats::default(); let graph_term = config.graph_iri.as_deref().map(|iri| { @@ -1577,13 +1558,12 @@ fn write_batch( let p_id = batch.p_id.get_or(row, 0); let o_type = batch.o_type.get_or(row, 0); let o_key = batch.o_key.get(row); - if is_link_row(o_type) { - continue; - } - if let Some(ann) = ann { + if is_link_row(o_type) { + ann.probe.note_link_in_scope(resolver, s_id); + continue; + } if ann.is_reifies_row(p_id) { - ann.probe.note_bundle_in_scope(resolver, s_id); continue; } } @@ -1693,6 +1673,25 @@ fn write_object( value: &FlakeValue, store: &BinaryIndexStore, o_type: u16, +) -> io::Result<()> { + let dt = resolve_datatype_iri(store, o_type); + write_value( + w, + value, + store, + store.resolve_lang_tag(o_type), + dt.as_deref(), + ) +} + +/// Write an object value as an N-Triples term, given its language tag and +/// datatype IRI. +fn write_value( + w: &mut W, + value: &FlakeValue, + store: &BinaryIndexStore, + lang: Option<&str>, + dt: Option<&str>, ) -> io::Result<()> { match value { FlakeValue::Ref(sid) => { @@ -1704,7 +1703,7 @@ fn write_object( FlakeValue::String(s) => { // Check for language tag first (takes precedence over datatype) - if let Some(lang) = store.resolve_lang_tag(o_type) { + if let Some(lang) = lang { w.write_all(b"\"")?; syntax::write_string(w, s)?; write_lang_tag(w, lang)?; @@ -1712,12 +1711,11 @@ fn write_object( } // Resolve datatype; omit ^^ (implicit) - let dt_iri = resolve_datatype_iri(store, o_type); w.write_all(b"\"")?; syntax::write_string(w, s)?; w.write_all(b"\"")?; - if let Some(dt) = &dt_iri { - if *dt != xsd::STRING { + if let Some(dt) = dt { + if dt != xsd::STRING { w.write_all(b"^^<")?; syntax::write_iri(w, dt)?; w.write_all(b">")?; @@ -1732,7 +1730,8 @@ fn write_object( FlakeValue::Long(n) => write_typed_literal( w, &n.to_string(), - &resolve_datatype_iri(store, o_type).unwrap_or_else(|| xsd::LONG.to_string()), + &dt.map(str::to_string) + .unwrap_or_else(|| xsd::LONG.to_string()), ), FlakeValue::Double(f) => write_typed_literal( // W3C canonical xsd:double form (1.0E6; NaN/INF/-INF preserved). Resolve @@ -1742,17 +1741,19 @@ fn write_object( // silently re-typed (CRITICAL-3 #1529 review). w, &canonical_xsd_double(*f), - &resolve_datatype_iri(store, o_type).unwrap_or_else(|| xsd::DOUBLE.to_string()), + &dt.map(str::to_string) + .unwrap_or_else(|| xsd::DOUBLE.to_string()), ), FlakeValue::BigInt(n) => write_typed_literal( w, &n.to_string(), - &resolve_datatype_iri(store, o_type).unwrap_or_else(|| xsd::INTEGER.to_string()), + &dt.map(str::to_string) + .unwrap_or_else(|| xsd::INTEGER.to_string()), ), FlakeValue::Decimal(d) => write_typed_literal( w, &d.to_string(), - &resolve_datatype_iri(store, o_type) + &dt.map(str::to_string) .unwrap_or_else(|| "http://www.w3.org/2001/XMLSchema#decimal".to_string()), ), @@ -1760,84 +1761,62 @@ fn write_object( FlakeValue::DateTime(v) => write_typed_literal_display( w, v.as_ref(), - store, - o_type, + dt, "http://www.w3.org/2001/XMLSchema#dateTime", ), - FlakeValue::Date(v) => write_typed_literal_display( - w, - v.as_ref(), - store, - o_type, - "http://www.w3.org/2001/XMLSchema#date", - ), - FlakeValue::Time(v) => write_typed_literal_display( - w, - v.as_ref(), - store, - o_type, - "http://www.w3.org/2001/XMLSchema#time", - ), - FlakeValue::GYear(v) => write_typed_literal_display( - w, - v.as_ref(), - store, - o_type, - "http://www.w3.org/2001/XMLSchema#gYear", - ), + FlakeValue::Date(v) => { + write_typed_literal_display(w, v.as_ref(), dt, "http://www.w3.org/2001/XMLSchema#date") + } + FlakeValue::Time(v) => { + write_typed_literal_display(w, v.as_ref(), dt, "http://www.w3.org/2001/XMLSchema#time") + } + FlakeValue::GYear(v) => { + write_typed_literal_display(w, v.as_ref(), dt, "http://www.w3.org/2001/XMLSchema#gYear") + } FlakeValue::GYearMonth(v) => write_typed_literal_display( w, v.as_ref(), - store, - o_type, + dt, "http://www.w3.org/2001/XMLSchema#gYearMonth", ), FlakeValue::GMonth(v) => write_typed_literal_display( w, v.as_ref(), - store, - o_type, + dt, "http://www.w3.org/2001/XMLSchema#gMonth", ), - FlakeValue::GDay(v) => write_typed_literal_display( - w, - v.as_ref(), - store, - o_type, - "http://www.w3.org/2001/XMLSchema#gDay", - ), + FlakeValue::GDay(v) => { + write_typed_literal_display(w, v.as_ref(), dt, "http://www.w3.org/2001/XMLSchema#gDay") + } FlakeValue::GMonthDay(v) => write_typed_literal_display( w, v.as_ref(), - store, - o_type, + dt, "http://www.w3.org/2001/XMLSchema#gMonthDay", ), FlakeValue::YearMonthDuration(v) => write_typed_literal_display( w, v.as_ref(), - store, - o_type, + dt, "http://www.w3.org/2001/XMLSchema#yearMonthDuration", ), FlakeValue::DayTimeDuration(v) => write_typed_literal_display( w, v.as_ref(), - store, - o_type, + dt, "http://www.w3.org/2001/XMLSchema#dayTimeDuration", ), FlakeValue::Duration(v) => write_typed_literal_display( w, v.as_ref(), - store, - o_type, + dt, "http://www.w3.org/2001/XMLSchema#duration", ), // Extension types FlakeValue::Json(s) => { - let dt = resolve_datatype_iri(store, o_type) + let dt = dt + .map(str::to_string) .unwrap_or_else(|| "http://www.w3.org/1999/02/22-rdf-syntax-ns#JSON".to_string()); w.write_all(b"\"")?; syntax::write_string(w, s)?; @@ -1846,7 +1825,8 @@ fn write_object( w.write_all(b">") } FlakeValue::Vector(v) => { - let dt = resolve_datatype_iri(store, o_type) + let dt = dt + .map(str::to_string) .unwrap_or_else(|| "https://ns.flur.ee/db#vector".to_string()); // Serialize as JSON array string let json = serde_json::to_string(v).unwrap_or_else(|_| "[]".to_string()); @@ -1857,7 +1837,8 @@ fn write_object( w.write_all(b">") } FlakeValue::GeoPoint(bits) => { - let dt = resolve_datatype_iri(store, o_type) + let dt = dt + .map(str::to_string) .unwrap_or_else(|| "http://www.opengis.net/ont/geosparql#wktLiteral".to_string()); let wkt = bits.to_string(); // "POINT(lng lat)" w.write_all(b"\"")?; @@ -1868,15 +1849,20 @@ fn write_object( } FlakeValue::Null => Ok(()), // should have been filtered above - FlakeValue::TripleTerm(_) => { - let dt = resolve_datatype_iri(store, o_type) - .unwrap_or_else(|| format!("{}tripleTerm", fluree_vocab::fluree::DB)); - let text = value.to_string(); - w.write_all(b"\"")?; - syntax::write_string(w, &text)?; - w.write_all(b"\"^^<")?; - syntax::write_iri(w, &dt)?; - w.write_all(b">") + FlakeValue::TripleTerm(term) => { + let iri = |sid: &Sid| { + store + .sid_to_iri(sid) + .unwrap_or_else(|| format!("_:unknown_{sid}")) + }; + w.write_all(b"<<( ")?; + write_iri_or_bnode(w, &iri(&term.s))?; + w.write_all(b" ")?; + write_iri_or_bnode(w, &iri(&term.p))?; + w.write_all(b" ")?; + let dt = store.sid_to_iri(&term.dt); + write_value(w, &term.o, store, term.lang.as_deref(), dt.as_deref())?; + w.write_all(b" )>>") } } } @@ -1978,6 +1964,10 @@ fn write_raw_object( write_typed_literal(w, &json, &dt)?; Ok(true) } + FlakeValue::TripleTerm(_) => { + write_value(w, &flake.o, store, None, None)?; + Ok(true) + } // Temporal and other types always encode into V3 ops, so they should // never reach the untranslated set. Decline rather than emit a // potentially non-canonical lexical form. @@ -2264,13 +2254,10 @@ fn write_typed_literal(w: &mut W, lexical: &str, datatype_iri: &str) - fn write_typed_literal_display( w: &mut W, value: &T, - store: &BinaryIndexStore, - o_type: u16, + dt: Option<&str>, fallback_dt: &str, ) -> io::Result<()> { - let lexical = value.to_string(); - let dt = resolve_datatype_iri(store, o_type).unwrap_or_else(|| fallback_dt.to_string()); - write_typed_literal(w, &lexical, &dt) + write_typed_literal(w, &value.to_string(), dt.unwrap_or(fallback_dt)) } #[cfg(test)] diff --git a/fluree-db-api/src/export_annotations.rs b/fluree-db-api/src/export_annotations.rs index e9be06bb18..c1c6a82635 100644 --- a/fluree-db-api/src/export_annotations.rs +++ b/fluree-db-api/src/export_annotations.rs @@ -1,84 +1,42 @@ //! Where `fluree export` gets "which reifiers point at this edge". //! //! RDF 1.2 annotation syntax names the reifier at the base edge -//! (`s p o ~ `), so serializing it needs the **forward** direction of the -//! edge-annotation index: `EdgeKey -> ann_sid`. That direction exists on disk -//! as the annotation arena's forward branch, and in memory as the attachment -//! overlay for anything committed since the last index build. This module -//! picks between them once per export and answers a whole `ColumnBatch` at a -//! time. -//! -//! ## Why not read the `f:reifies*` flakes the scan is already passing -//! -//! They are the durable encoding, and export sees every one of them. But they -//! are keyed by *reifier* subject, and the scan is in SPOT order: a reifier -//! whose IRI sorts after its base edge's subject arrives too late to influence -//! the line already written. Reconstructing the mapping from the scan alone -//! therefore means either buffering the whole graph or a second full pass. -//! The arena probe is `O(edges·log edges + covered leaves)` and needs neither. - -use fluree_db_binary_index::annotation_arena::AnnotationArenaReader; -use fluree_db_core::storage::ContentStore; -use fluree_db_core::{AnnotationIndexRoot, EdgeKey, Sid}; -use fluree_db_novelty::AttachmentNovelty; +//! (`s p o ~ `), so serializing it needs the edge → reifier direction of +//! the `rdf:reifies` links. The scan is in SPOT order, and a reifier whose IRI +//! sorts after its base edge's subject arrives too late to mark the line +//! already written, so the links are read once, up front: `O(annotations)`, +//! not `O(dataset)`. + +use fluree_db_core::comparator::IndexType; +use fluree_db_core::range::{range_with_overlay, RangeMatch, RangeOptions, RangeTest}; +use fluree_db_core::{EdgeKey, FlakeValue, GraphId, Sid}; use std::collections::{HashMap, HashSet}; use std::io; -use std::sync::{Arc, Mutex}; +use std::marker::PhantomData; +use std::sync::Mutex; -use crate::{ApiError, LedgerState, Result}; +use crate::{LedgerState, Result}; -/// Forward annotation lookup for the duration of one export. +/// Edge → reifier lookup for the duration of one export. /// -/// Constructed once by [`Self::for_ledger`]; borrowed by every per-graph -/// writer. Holds no state of its own beyond the two sources — the arena -/// reader, which caches branches and hot leaves, is built per probe call -/// because the writers each own their own scan. /// Only `ExportBuilder` constructs one; the type is public because it appears /// on the public `ExportConfig`. An external caller building an `ExportConfig` -/// by hand passes `annotations: None` and gets the pre-RDF-1.2 behaviour. +/// by hand passes `annotations: None` and gets the links as ordinary triples. pub struct AnnotationProbe<'a> { - /// Sealed on-disk arena: authoritative for `t <= its max_t`. - arena: Option<(&'a AnnotationIndexRoot, &'a Arc)>, - /// One reader for the whole export, built from `arena` once. - /// - /// The reader memoises the forward branch and every forward leaf it - /// touches. It used to be constructed inside `live_reifiers`, which runs - /// once per `ColumnBatch`, so those caches were rebuilt and thrown away - /// for every batch of the export and each batch re-fetched the same - /// branch. The source is chosen once per export, so the reader can be. - reader: Option>, - /// Attachment events committed since the last index build. - novelty: Option<&'a AttachmentNovelty>, - /// Bundles recovered by scanning the base index, for a ledger whose arena - /// was never sealed. Resolved once at export start; empty otherwise. - scanned: HashMap>, - as_of_t: i64, + /// Live links at the export's `t`, per graph, by the edge they name + /// (its `g` cleared). + links: HashMap>>, /// Reifiers named by a `~ ` marker somewhere in this export. named: Mutex>, - /// Reifiers whose `f:reifies*` bundle the scan passed — i.e. whose own - /// subject is inside the exported selection, so their properties are in - /// the file. Populated from the rows the writers suppress, which costs - /// nothing extra: they are already being visited and discarded. + /// Reifiers whose link the scan passed — i.e. whose own subject is inside + /// the exported selection, so their properties are in the file. Populated + /// from the rows the writers suppress, which costs nothing extra: they are + /// already being visited and discarded. in_scope: Mutex>, + _ledger: PhantomData<&'a LedgerState>, } impl<'a> AnnotationProbe<'a> { - fn new( - arena: Option<(&'a AnnotationIndexRoot, &'a Arc)>, - novelty: Option<&'a AttachmentNovelty>, - as_of_t: i64, - ) -> Self { - Self { - arena, - reader: arena.map(|(root, store)| AnnotationArenaReader::new(root, store.as_ref())), - novelty, - scanned: HashMap::new(), - as_of_t, - named: Mutex::new(HashSet::new()), - in_scope: Mutex::new(HashSet::new()), - } - } - /// Record that a `~ ` marker (or its per-format equivalent) was /// emitted for `reifier`. pub(crate) fn note_reifier_named(&self, reifier: &Sid) { @@ -89,27 +47,23 @@ impl<'a> AnnotationProbe<'a> { } } - /// Record that the scan passed `s_id`'s `f:reifies*` bundle, so that - /// reifier's own triples are inside this export. - /// - /// Takes the raw subject id and resolves it only on first sight of each - /// reifier — a bundle is up to seven rows, and resolving all of them - /// would allocate seven times per annotation for one set entry. - pub(crate) fn note_bundle_in_scope(&self, resolver: &dyn ReifierSubject, s_id: u64) { + /// Record that the scan passed `s_id`'s link, so that reifier's own + /// triples are inside this export. + pub(crate) fn note_link_in_scope(&self, resolver: &dyn ReifierSubject, s_id: u64) { let Ok(sid) = resolver.reifier_sid(s_id) else { return; }; - self.note_bundle_sid(sid); + self.note_link_sid(sid); } - /// As [`Self::note_bundle_in_scope`], for callers that already hold the + /// As [`Self::note_link_in_scope`], for callers that already hold the /// reifier's `Sid`. /// /// Untranslated overlay rows carry a fully-decoded subject, so they need /// no resolver round-trip. They still have to be *counted*: suppressing a - /// bundle without noting it turns a visible leak into an annotation that - /// vanishes with no marker, no bundle and no number. - pub(crate) fn note_bundle_sid(&self, sid: Sid) { + /// link without noting it turns a visible leak into an annotation that + /// vanishes with no marker, no link and no number. + pub(crate) fn note_link_sid(&self, sid: Sid) { if let Ok(mut in_scope) = self.in_scope.lock() { in_scope.insert(sid); } @@ -117,7 +71,7 @@ impl<'a> AnnotationProbe<'a> { /// Reifiers named in the output whose own description is not in it. /// - /// Read once, after every graph of an export has been written: a bundle + /// Read once, after every graph of an export has been written: a link /// can legitimately be emitted in a later graph than the marker, so the /// answer is only meaningful for the file as a whole. pub(crate) fn out_of_scope_count(&self) -> u64 { @@ -127,18 +81,10 @@ impl<'a> AnnotationProbe<'a> { named.difference(&in_scope).count() as u64 } - /// Bundles the export suppressed whose reifier it never named. - /// - /// The complement of [`Self::out_of_scope_count`], and the one that means - /// data loss: the `f:reifies*` rows were dropped from the output because - /// annotation syntax was going to replace them, and then no marker was - /// emitted. Today this is reachable for annotations inside a named graph — - /// the PSOT scan that seals the arena, and that this module falls back to, - /// returns nothing for a named graph, so neither forward source knows - /// about them. The rows are in the ledger and SPARQL reads them; only - /// these two lookups are blind. Reporting it is what keeps that gap from - /// being a silent truncation, and `--raw-reifies` gets the bundles out - /// verbatim in the meantime. + /// Links the export suppressed whose reifier it never named: the link + /// rows were dropped because annotation syntax was going to replace them, + /// and then no marker was emitted. Reported so that gap can never be a + /// silent truncation. pub(crate) fn unresolved_count(&self) -> u64 { let (Ok(named), Ok(in_scope)) = (self.named.lock(), self.in_scope.lock()) else { return 0; @@ -146,193 +92,79 @@ impl<'a> AnnotationProbe<'a> { in_scope.difference(&named).count() as u64 } - /// Choose an annotation source for `ledger`, or establish that it has none. - /// - /// `Ok(None)` is the hard-guarantee case from the truth table in - /// `fluree_db_core::annotation_index`: no `f:reifies*` flake has ever been - /// observed, on either the indexed side or the overlay. The caller keeps - /// its existing scan untouched, so a ledger without annotations pays one - /// boolean read for all of this. - /// - /// `has_annotations = true` with no arena sealed is the state every - /// `fluree index` pass leaves an annotated ledger in, and a bulk import - /// before its auto-seal pass. There is no forward index to probe, so the - /// bundles are recovered from the base index instead: a PSOT scan of the - /// seven `f:reifies*` predicates, `O(annotations)` rather than - /// `O(dataset)`, reusing the routine and the canonical - /// `EdgeKey::from_reifies_facts` decode the arena's own seal pass uses. - /// - /// Emitting nothing there was never an option — it would turn #1859's - /// loud defect into a quiet one, on the ledgers most likely to carry - /// annotations. Refusing was, until it turned out that `fluree reindex` - /// cannot clear the state (the sticky `had_annotation_arena` bit blocks - /// the re-bootstrap), which would have made the refusal a dead end on - /// ledgers every other reader serves through its own scan fallback. + /// Read `ledger`'s live links as of `as_of_t`, or establish that it has + /// none. `Ok(None)` when neither the index nor novelty has ever held an + /// annotation, so a ledger without annotations pays two boolean reads. pub(crate) async fn for_ledger(ledger: &'a LedgerState, as_of_t: i64) -> Result> { - let snapshot = &ledger.snapshot; - let attachments = &ledger.novelty.attachments; - let novelty = attachments.has_annotations().then_some(attachments); - - if !snapshot.has_annotations { - // Indexed side guarantees zero attachments. The overlay may still - // hold some: a ledger that has never been indexed keeps its entire - // history there, and so does one indexed before its first - // annotation was written. - return Ok(novelty.map(|novelty| Self::new(None, Some(novelty), as_of_t))); - } - - // Kill switch: take the base-index scan even when an arena is - // sealed. The two sources should agree; this is how you find out - // when they do not, without rebuilding an index. - if force_base_index_scan() { - let mut probe = Self::new(None, novelty, as_of_t); - probe.scanned = scan_bundles(ledger, as_of_t).await; - return Ok(Some(probe)); + if !ledger.snapshot.has_annotations && !ledger.novelty.has_annotations() { + return Ok(None); } - match (&snapshot.annotation_index, &snapshot.content_store) { - (Some(root), Some(store)) => Ok(Some(Self::new(Some((root, store)), novelty, as_of_t))), - (Some(_), None) => Err(ApiError::internal( - "ledger has a sealed annotation arena but no content store to read it from", - )), - (None, _) => { - let mut probe = Self::new(None, novelty, as_of_t); - probe.scanned = scan_bundles(ledger, as_of_t).await; - Ok(Some(probe)) - } - } - } - - /// Bundles the base-index scan recovered for `edge`. - #[inline] - fn scanned_for(&self, edge: &EdgeKey) -> Vec { - self.scanned.get(edge).cloned().unwrap_or_default() - } - - /// Live reifiers for each edge, index-aligned with `edges`. - /// - /// Entry `i` is empty when `edges[i]` carries no annotation whose latest - /// event at or before `as_of_t` is an assert. - pub(crate) async fn live_reifiers(&self, edges: &[EdgeKey]) -> io::Result>> { - if edges.is_empty() { - return Ok(Vec::new()); - } - match (self.arena, self.novelty) { - (None, None) => Ok(edges.iter().map(|e| self.scanned_for(e)).collect()), - - // Sealed arena, nothing pending: one sorted merge-scan for the - // whole batch. `current_annotations_batch` is arena-only by - // contract, which is exactly what an empty overlay makes correct. - (Some(_), None) => { - let reader = self - .reader - .as_ref() - .expect("reader is built whenever arena is"); - reader - .current_annotations_batch(edges, self.as_of_t) - .await - .map_err(|e| io::Error::other(format!("annotation arena probe: {e}"))) - } - - // Attachments committed since the arena was sealed. The batched - // read cannot see a novelty *retract* of an indexed attachment, so - // each edge merges its own event stream instead. Slower, and - // confined to ledgers with pending annotation novelty. - (Some(_), Some(novelty)) => { - let reader = self - .reader - .as_ref() - .expect("reader is built whenever arena is"); - let mut out = Vec::with_capacity(edges.len()); - for edge in edges { - let events = novelty.collect_forward_events(edge); - out.push( - reader - .current_annotations_merged(edge, &events, self.as_of_t) - .await - .map_err(|e| { - io::Error::other(format!("annotation arena merge: {e}")) - })?, - ); + let graphs = std::iter::once(0).chain( + ledger + .snapshot + .graph_registry + .iter_entries() + .map(|(g_id, _)| g_id), + ); + let mut links: HashMap>> = HashMap::new(); + for g_id in graphs { + let flakes = range_with_overlay( + &ledger.snapshot, + g_id, + ledger.novelty.as_ref(), + IndexType::Psot, + RangeTest::Eq, + RangeMatch::predicate(fluree_db_core::rdf_reifies_sid().clone()), + RangeOptions::new().with_to_t(as_of_t), + ) + .await?; + for flake in flakes { + let FlakeValue::TripleTerm(term) = flake.o else { + continue; + }; + let edge = EdgeKey { + g: None, + s: term.s, + p: term.p, + o: term.o, + dt: term.dt, + lang: term.lang, + list_i: None, + }; + let reifiers = links.entry(g_id).or_default().entry(edge).or_default(); + if !reifiers.contains(&flake.s) { + reifiers.push(flake.s); } - Ok(out) } - - // No arena: the overlay is the whole history. An in-memory - // `BTreeMap` lookup per edge, no I/O. - (None, Some(novelty)) => Ok(edges - .iter() - .map(|edge| { - let mut out = self.scanned_for(edge); - for ann in novelty.current_annotations_for_at(edge, self.as_of_t) { - if !out.contains(&ann) { - out.push(ann); - } - } - out - }) - .collect()), } - } -} - -/// Whether `FLUREE_EXPORT_ANNOTATION_SCAN` asks for the base-index scan in -/// place of a sealed arena. -/// -/// Reads the *value*, not merely the presence. This flag is positively -/// named, so `=0` has to mean off — it previously tested `.is_ok()`, which -/// made `FLUREE_EXPORT_ANNOTATION_SCAN=0` *enable* the scan and silently -/// trade a correct arena read for the fallback. The `FLUREE_DISABLE_*` flags -/// in this tree are presence-only and named so that presence-only is -/// correct; this one is not one of those. Matches -/// `indexer_attachment_provider::force_annotation_bootstrap`, the other -/// positively-named flag in the crate. -fn force_base_index_scan() -> bool { - env_flag_enabled( - std::env::var("FLUREE_EXPORT_ANNOTATION_SCAN") - .ok() - .as_deref(), - ) -} - -/// The parse behind [`force_base_index_scan`], separated from the read so it -/// can be tested without mutating a process-global. -fn env_flag_enabled(value: Option<&str>) -> bool { - matches!(value, Some(v) if v == "1" || v.eq_ignore_ascii_case("true")) -} - -/// Recover `EdgeKey -> reifiers` from the base index for an unsealed ledger. -/// -/// Best effort by design: the underlying scan reports its own failures as "no -/// coverage", and an export that emits no annotation marker is a better -/// outcome than one that fails outright on a ledger every other reader can -/// serve. -#[cfg(not(target_arch = "wasm32"))] -async fn scan_bundles(ledger: &LedgerState, as_of_t: i64) -> HashMap> { - let events = crate::indexer_attachment_provider::scan_base_index_for_attachment_events_in( - &ledger.snapshot, - ledger.novelty.as_ref(), - as_of_t, - &ledger.snapshot.ledger_id, - ) - .await; - let mut out: HashMap> = HashMap::new(); - for (edge, ann, t, op) in events.unwrap_or_default() { - // The scan already drops retracted rows; `t` still has to be honoured - // so a time-travel export does not show an annotation written later. - if !op || t > as_of_t { - continue; - } - let slot = out.entry(edge).or_default(); - if !slot.contains(&ann) { - slot.push(ann); + for reifiers in links.values_mut().flat_map(HashMap::values_mut) { + reifiers.sort(); } + Ok(Some(Self { + links, + named: Mutex::new(HashSet::new()), + in_scope: Mutex::new(HashSet::new()), + _ledger: PhantomData, + })) } - out -} -#[cfg(target_arch = "wasm32")] -async fn scan_bundles(_ledger: &LedgerState, _as_of_t: i64) -> HashMap> { - HashMap::new() + /// Live reifiers for each edge of graph `g_id`, index-aligned with + /// `edges`; entry `i` is empty when `edges[i]` carries no annotation. + pub(crate) fn live_reifiers(&self, g_id: GraphId, edges: &[EdgeKey]) -> Vec> { + let Some(links) = self.links.get(&g_id) else { + return vec![Vec::new(); edges.len()]; + }; + edges + .iter() + .map(|edge| { + let key = EdgeKey { + g: None, + ..edge.clone() + }; + links.get(&key).cloned().unwrap_or_default() + }) + .collect() + } } /// Resolves a subject id to the `Sid` a reifier was stored under. @@ -342,25 +174,3 @@ async fn scan_bundles(_ledger: &LedgerState, _as_of_t: i64) -> HashMap io::Result; } - -#[cfg(test)] -mod env_flag_tests { - use super::env_flag_enabled; - - /// A positively-named flag has to read its value. `=0` meaning "on" is - /// the defect this guards, and the docs promise `=1` means force. - #[test] - fn only_affirmative_values_enable_the_scan() { - assert!(env_flag_enabled(Some("1"))); - assert!(env_flag_enabled(Some("true"))); - assert!(env_flag_enabled(Some("TRUE"))); - - assert!(!env_flag_enabled(Some("0")), "`=0` must not enable it"); - assert!(!env_flag_enabled(Some("false"))); - assert!( - !env_flag_enabled(Some("")), - "set-but-empty is not an opt-in" - ); - assert!(!env_flag_enabled(None), "unset is off"); - } -} diff --git a/fluree-db-api/src/export_builder.rs b/fluree-db-api/src/export_builder.rs index 6b601b0f6f..4886562706 100644 --- a/fluree-db-api/src/export_builder.rs +++ b/fluree-db-api/src/export_builder.rs @@ -86,13 +86,10 @@ impl<'a> ExportBuilder<'a> { self } - /// Emit edge annotations as the raw `f:reifies*` system facts, the output - /// every release before RDF 1.2 annotation syntax produced. - /// - /// Kept as an escape hatch for consumers pinned to those bytes. Note that - /// Fluree's own JSON-LD and Turtle write surfaces reject hand-written - /// `f:reifies*` triples, so this output is re-ingestible only through the - /// bulk-import path. + /// Write each annotation as its stored `rdf:reifies` link, + /// `r rdf:reifies <<( s p o )>>`, rather than as annotation syntax on the + /// base edge. JSON-LD has no triple-term syntax, so a JSON-LD export + /// keeps `@annotation`. pub fn raw_reifies(mut self) -> Self { self.raw_reifies = true; self @@ -263,10 +260,10 @@ impl<'a> ExportBuilder<'a> { let overlay: &dyn fluree_db_core::OverlayProvider = ledger.novelty.as_ref(); let dict_novelty = &ledger.dict_novelty; - // Forward annotation lookup, chosen once for the whole export. `None` - // on a ledger that has never carried an annotation — and on - // `raw_reifies()`, which keeps the pre-RDF-1.2 output byte for byte. - let annotations = if self.raw_reifies { + // Edge → reifier lookup, read once for the whole export. `None` on a + // ledger that has never carried an annotation — and on + // `raw_reifies()`, which writes the links as triples. + let annotations = if self.raw_reifies && !matches!(self.format, ExportFormat::JsonLd) { None } else { AnnotationProbe::for_ledger(&ledger, to_t).await? diff --git a/fluree-db-api/src/format/cypher_typed.rs b/fluree-db-api/src/format/cypher_typed.rs index ee266a0396..4937f90819 100644 --- a/fluree-db-api/src/format/cypher_typed.rs +++ b/fluree-db-api/src/format/cypher_typed.rs @@ -429,12 +429,6 @@ struct NodeHydrator<'a> { enforcer: Option>, rdf_type: Option, node_marker: Option, - /// The `f:reifies{Subject,Predicate,Object}` predicate Sids (when the - /// dictionary knows them): a subject carrying these is an edge - /// annotation and renders as a Relationship, never a Node. - reifies_subject: Option, - reifies_predicate: Option, - reifies_object: Option, cache: HashMap, /// Rendered top-level cells per subject (Node, or Relationship for /// reifier subjects); what [`Self::subject_cell`] serves. @@ -462,6 +456,17 @@ struct NodeHydrator<'a> { #[cfg(not(target_arch = "wasm32"))] const PREFETCH_CONCURRENCY: usize = 16; +/// The `(start, type, end)` of a link's triple when it relates two nodes. +fn node_edge(link_object: &FlakeValue) -> Option<(Sid, Sid, Sid)> { + match link_object { + FlakeValue::TripleTerm(term) => match &term.o { + FlakeValue::Ref(end) => Some((term.s.clone(), term.p.clone(), end.clone())), + _ => None, + }, + _ => None, + } +} + impl<'a> NodeHydrator<'a> { fn new( view: &'a GraphDb, @@ -481,13 +486,6 @@ impl<'a> NodeHydrator<'a> { enforcer, rdf_type: view.snapshot.encode_iri(fluree_vocab::rdf::TYPE), node_marker: view.snapshot.encode_iri(fluree_vocab::fluree::NODE), - reifies_subject: view - .snapshot - .encode_iri(fluree_vocab::reifies_iris::SUBJECT), - reifies_predicate: view - .snapshot - .encode_iri(fluree_vocab::reifies_iris::PREDICATE), - reifies_object: view.snapshot.encode_iri(fluree_vocab::reifies_iris::OBJECT), cell_cache: HashMap::new(), cache: HashMap::new(), flake_cache: HashMap::new(), @@ -538,7 +536,7 @@ impl<'a> NodeHydrator<'a> { return Ok(hit.clone()); } let p_iri = self.compactor.decode_sid(pred)?; - let name = if p_iri.starts_with(FLUREE_SYSTEM_NS) { + let name = if p_iri.starts_with(FLUREE_SYSTEM_NS) || p_iri == fluree_vocab::rdf::REIFIES { None } else { Some(Arc::from(self.name(&p_iri).as_str())) @@ -649,25 +647,11 @@ impl<'a> NodeHydrator<'a> { FormatError::InvalidBinding(format!("batched subject crawl failed: {e}")) })?; - let reifies_pids: Option<(u32, u32, u32)> = match ( - self.reifies_subject.as_ref(), - self.reifies_predicate.as_ref(), - self.reifies_object.as_ref(), - ) { - (Some(rs), Some(rp), Some(ro)) => match ( - store.sid_to_p_id(rs), - store.sid_to_p_id(rp), - store.sid_to_p_id(ro), - ) { - (Some(a), Some(b), Some(c)) => Some((a, b, c)), - _ => None, - }, - _ => None, - }; + let reifies_pid = store.sid_to_p_id(fluree_db_core::rdf_reifies_sid()); for (s_id, sid) in subjects { let rows = rows_by_subject.remove(&s_id).unwrap_or_default(); let cell = if let Some(rel) = - self.relationship_from_rows(&sid, &rows, store.as_ref(), g_id, reifies_pids)? + self.relationship_from_rows(&sid, &rows, store.as_ref(), g_id, reifies_pid)? { CypherCell::Relationship(Box::new(rel)) } else { @@ -680,45 +664,34 @@ impl<'a> NodeHydrator<'a> { Ok(()) } - /// Detect and render a reifier subject from batched-crawl rows: present - /// `f:reifies{Subject,Predicate,Object}` rows with node-ref objects make - /// it a Relationship. Returns `None` for ordinary nodes. + /// Detect and render a reifier subject from batched-crawl rows: an + /// `rdf:reifies` link to a node→node triple makes it a Relationship. + /// Returns `None` for ordinary nodes. fn relationship_from_rows( &mut self, sid: &Sid, rows: &[(u32, u16, u64)], store: &fluree_db_binary_index::BinaryIndexStore, g_id: u16, - reifies_pids: Option<(u32, u32, u32)>, + reifies_pid: Option, ) -> Result> { - let Some((rs_pid, rp_pid, ro_pid)) = reifies_pids else { + let Some(reifies_pid) = reifies_pid else { return Ok(None); }; - let mut start = None; - let mut pred = None; - let mut end = None; + let mut edge = None; for &(p_id, o_type, o_key) in rows { - if p_id != rs_pid && p_id != rp_pid && p_id != ro_pid { - continue; - } - if !fluree_db_core::o_type::OType::from_u16(o_type).is_node_ref() { + if p_id != reifies_pid { continue; } let decoded = store .decode_value_v3(o_type, o_key, p_id, g_id) - .map_err(|e| FormatError::InvalidBinding(format!("decode reifies ref: {e}")))?; - let FlakeValue::Ref(target) = decoded else { - continue; - }; - if p_id == rs_pid { - start = Some(target); - } else if p_id == rp_pid { - pred = Some(target); - } else { - end = Some(target); + .map_err(|e| FormatError::InvalidBinding(format!("decode link: {e}")))?; + edge = node_edge(&decoded); + if edge.is_some() { + break; } } - let (Some(start), Some(pred), Some(end)) = (start, pred, end) else { + let Some((start, pred, end)) = edge else { return Ok(None); }; // Annotation (user) properties: the scalar rows minus bookkeeping. @@ -804,7 +777,9 @@ impl<'a> NodeHydrator<'a> { return Ok(hit.clone()); } let name = match store.resolve_predicate_iri(p_id) { - Some(iri) if iri.starts_with(FLUREE_SYSTEM_NS) => None, + Some(iri) if iri.starts_with(FLUREE_SYSTEM_NS) || iri == fluree_vocab::rdf::REIFIES => { + None + } Some(iri) => Some(Arc::from(self.name(iri).as_str())), None => None, }; @@ -952,28 +927,14 @@ impl<'a> NodeHydrator<'a> { return Ok(hit.clone()); } let flakes = self.subject_flakes(sid).await?; - let mut start = None; - let mut pred = None; - let mut end = None; - for flake in flakes.iter().filter(|f| f.op) { - if Some(&flake.p) == self.reifies_subject.as_ref() { - if let FlakeValue::Ref(s) = &flake.o { - start = Some(s.clone()); - } - } else if Some(&flake.p) == self.reifies_predicate.as_ref() { - if let FlakeValue::Ref(p) = &flake.o { - pred = Some(p.clone()); - } - } else if Some(&flake.p) == self.reifies_object.as_ref() { - if let FlakeValue::Ref(o) = &flake.o { - end = Some(o.clone()); - } - } - } - let cell = match (start, pred, end) { + let edge = flakes + .iter() + .filter(|f| f.op && fluree_db_core::is_rdf_reifies(&f.p)) + .find_map(|f| node_edge(&f.o)); + let cell = match edge { // A reified node→node edge. (Literal-object annotations keep // the node rendering — Bolt relationships need node endpoints.) - (Some(start), Some(pred), Some(end)) => { + Some((start, pred, end)) => { let start_iri = self.compactor.decode_sid_shared(&start)?; let end_iri = self.compactor.decode_sid_shared(&end)?; let type_iri = self.compactor.decode_sid(&pred)?; diff --git a/fluree-db-api/src/format/hydration.rs b/fluree-db-api/src/format/hydration.rs index 088f1c00b2..b99f57210d 100644 --- a/fluree-db-api/src/format/hydration.rs +++ b/fluree-db-api/src/format/hydration.rs @@ -56,7 +56,7 @@ use fluree_db_query::ir::{ Column, ForwardItem, HydrationSpec, NestedModifiers, NestedOrderKey, NestedSelectSpec, Root, }; use fluree_db_query::QueryPolicyEnforcer; -use fluree_vocab::namespaces::{BLANK_NODE, FLUREE_DB, JSON_LD}; +use fluree_vocab::namespaces::{BLANK_NODE, JSON_LD}; use fluree_vocab::rdf::{self, TYPE as RDF_TYPE_IRI}; use futures::future::BoxFuture; use futures::stream::{self, StreamExt, TryStreamExt}; @@ -865,21 +865,8 @@ impl<'a> DatasetCtx<'a> { /// resolve cross-ledger refs. fn formatter_for(&'a self, idx: usize) -> HydrationFormatter<'a> { let view = &self.views[idx]; - // Mirror `HydrationFormatter::new`: cache one arena reader and the - // overlay's `Novelty` downcast per view so annotation lookups share - // branch/leaf caches and skip repeated dynamic dispatch. - let arena_reader = match ( - view.db.snapshot.annotation_index.as_ref(), - view.db.snapshot.content_store.as_ref(), - ) { - (Some(root), Some(store)) => Some( - fluree_db_binary_index::annotation_arena::AnnotationArenaReader::new( - root, - store.as_ref(), - ), - ), - _ => None, - }; + // Mirror `HydrationFormatter::new`: cache the overlay's `Novelty` + // downcast per view. let novelty = view .db .overlay @@ -892,7 +879,6 @@ impl<'a> DatasetCtx<'a> { normalize_arrays: self.normalize_arrays, policy: view.policy.as_ref(), tracker: self.tracker, - arena_reader, novelty, dataset: Some(self), active_idx: idx, @@ -1118,29 +1104,10 @@ struct HydrationFormatter<'a> { policy: Option<&'a PolicyContext>, /// Optional execution tracker for fuel/policy tracking. tracker: Option<&'a Tracker>, - /// Single arena reader reused across every annotation lookup in - /// this response. Constructed once on `new()` when the snapshot - /// satisfies `has_arena_reader()`. Holds the loaded forward / - /// reverse branches plus a per-CID leaf cache, so successive - /// edge lookups amortize the CAS reads. `None` falls back to the - /// scan path in `inject_annotations`. - arena_reader: Option< - fluree_db_binary_index::annotation_arena::AnnotationArenaReader< - 'a, - dyn fluree_db_core::storage::ContentStore, - >, - >, /// Cached downcast of `db.overlay` to the concrete - /// `fluree_db_novelty::Novelty` type. Computed once on `new()` - /// and reused across every annotation-hydration call site — - /// previously each call did its own - /// `as_any().downcast_ref::()` (three times per - /// ref-valued property: in the gate, in - /// `arena_lookup_annotations`, and twice in - /// `is_live_annotation_subject`). `None` means the overlay - /// isn't the concrete `Novelty` type (test fakes / future - /// overlays); callers fall back to the scan path the same way - /// they did before. + /// `fluree_db_novelty::Novelty` type, for the annotation gate. `None` + /// means the overlay isn't the concrete `Novelty` type (test fakes / + /// future overlays), which never skips an annotation lookup. novelty: Option<&'a fluree_db_novelty::Novelty>, /// Dataset context for cross-ledger reference resolution. /// @@ -1166,22 +1133,6 @@ impl<'a> HydrationFormatter<'a> { policy: Option<&'a PolicyContext>, tracker: Option<&'a Tracker>, ) -> Self { - // Cache one arena reader for the whole response so successive - // edge lookups share branch + leaf caches. Constructed only - // when both `annotation_index` and `content_store` are set on - // the snapshot — otherwise the scan path runs. - let arena_reader = match ( - db.snapshot.annotation_index.as_ref(), - db.snapshot.content_store.as_ref(), - ) { - (Some(root), Some(store)) => Some( - fluree_db_binary_index::annotation_arena::AnnotationArenaReader::new( - root, - store.as_ref(), - ), - ), - _ => None, - }; // Cache the overlay's `Novelty` downcast once. The overlay // pointer is fixed for the lifetime of the formatter (one // hydration response), so doing the dynamic dispatch up @@ -1198,7 +1149,6 @@ impl<'a> HydrationFormatter<'a> { normalize_arrays: config.normalize_arrays, policy, tracker, - arena_reader, novelty, dataset: None, active_idx: 0, @@ -1287,12 +1237,11 @@ impl<'a> HydrationFormatter<'a> { // // The check runs **before** `fetch_subject_properties` // so it sees the unfiltered annotation membership — a - // policy that allows `ex:role` but denies `f:reifies*` - // would strip the discriminator from the rendered flake - // set, and we'd lose the signal. Going through the - // overlay (and arena, when present) sidesteps the - // policy filter entirely; annotation membership is a - // structural property of the snapshot, not user data. + // policy that allows `ex:role` but hides the link would + // strip the discriminator from the rendered flake set, and + // we'd lose the signal. The membership probe sidesteps the + // policy filter; annotation membership is a structural + // property of the snapshot, not user data. // // Only fires at the top of the expansion (`depth.current // == 0`) — recursive ref expansion keeps the existing @@ -1803,13 +1752,18 @@ impl<'a> HydrationFormatter<'a> { } /// Look up the rendered annotation bodies attached to `flake`'s - /// base edge. Returns an empty vec when the ledger has no - /// annotations or when the edge has none. + /// base edge: the bodies of the reifiers whose `rdf:reifies` link + /// names it. Returns an empty vec when the ledger has no annotations + /// or when the edge has none. /// /// Used as the probe step by both the Ref arm (via /// `inject_annotations`) and the literal arm (which needs to /// decide whether to promote a scalar render to a value-object /// shape before injecting). + /// + /// The link probe bypasses view policy, as membership does: the + /// base edge is already visible, and each body renders through + /// `format_subject`, which applies policy. async fn lookup_annotation_bodies<'b>( &'b self, flake: &'b Flake, @@ -1817,34 +1771,20 @@ impl<'a> HydrationFormatter<'a> { visited: &'b mut HashSet, cache: &'b mut HydrationCaches, ) -> Result> { - // Zero-cost gate for non-annotation ledgers — mirrors the - // cascade fast-path in `fluree_db_transact::stage` so a - // hydration query like `select: {"?s": ["*"]}` doesn't pay - // a POST scan per ref value when the ledger has never seen - // an `f:reifies*` flake. Two signals: - // - // - `snapshot.has_annotations`: sticky bit on `IndexRoot`, - // set at indexer time when any of the seven reserved - // `f:reifies*` predicate SIDs first appears in the - // predicate dictionary. Zero historical exposure on - // ledgers that never used annotations. - // - `novelty.attachments.has_annotations()`: in-memory - // overlay sticky bit, flipped on the first observed - // `f:reifies*` bundle. - // - // Both must be false to skip safely. We only consult the - // overlay when it downcasts cleanly to the concrete - // `Novelty` type — for unknown overlay implementations - // (test fakes, future variants), keep the scan fallback so - // we don't silently miss attachments. - if !self.db.snapshot.has_annotations { - let novelty_clean = self.novelty.map(|n| !n.attachments.has_annotations()); - if matches!(novelty_clean, Some(true)) { - return Ok(Vec::new()); - } + if !self.may_hold_annotations() { + return Ok(Vec::new()); } - - let edge_key = fluree_db_core::edge::EdgeKey::from_flake(flake); + // A list element is never reified. + if flake.m.as_ref().is_some_and(|m| m.i.is_some()) { + return Ok(Vec::new()); + } + let term = FlakeValue::TripleTerm(Box::new(fluree_db_core::TripleTermValue { + s: flake.s.clone(), + p: flake.p.clone(), + o: flake.o.clone(), + dt: flake.dt.clone(), + lang: flake.m.as_ref().and_then(|m| m.lang.clone()), + })); // Span name preserved across the refactor (was emitted by the // pre-refactor `inject_annotations`). External tooling @@ -1854,130 +1794,43 @@ impl<'a> HydrationFormatter<'a> { use tracing::Instrument; let span = tracing::debug_span!( "inject_annotations", - edge_in_named_graph = edge_key.g.is_some(), - path = tracing::field::Empty, + edge_in_named_graph = flake.g.is_some(), annotation_count = tracing::field::Empty, ); async { - // Arena-backed fast path. - if self.arena_reader.is_some() { - if let Some(mut ann_sids) = self.arena_lookup_annotations(&edge_key).await? { - tracing::Span::current().record("path", "arena"); - tracing::Span::current().record("annotation_count", ann_sids.len()); - if ann_sids.is_empty() { - return Ok(Vec::new()); - } - // Sort by Sid for stable, path-independent output - // order. `merge_live_annotations` already returns - // BTreeMap-sorted today, but pinning the sort here - // keeps arena/scan parity if that helper ever - // changes its collection order. - ann_sids.sort(); - return self - .render_annotation_bodies(&ann_sids, depth, visited, cache) - .await; - } - } - tracing::Span::current().record("path", "scan"); - - // Scan fallback: POST(f:reifiesSubject, edge.s) → candidate - // annotations whose subject points at our base subject. Each - // candidate's bundle is decoded and compared against the - // base flake's EdgeKey (full structural equality including - // `lang` and `dt`) before its body is formatted. - let f_reifies_subject = Sid::new(FLUREE_DB, fluree_vocab::db::REIFIES_SUBJECT); - let candidate_flakes = self + let links = self .db .range( IndexType::Post, RangeTest::Eq, - RangeMatch::predicate_object( - f_reifies_subject, - FlakeValue::Ref(edge_key.s.clone()), - ), + RangeMatch::predicate_object(fluree_db_core::rdf_reifies_sid().clone(), term), ) .await .map_err(|e| { - FormatError::InvalidBinding(format!( - "annotation lookup (f:reifiesSubject scan) failed: {e}" - )) + FormatError::InvalidBinding(format!("annotation link lookup failed: {e}")) })?; - - if candidate_flakes.is_empty() { - return Ok(Vec::new()); - } - - // First pass: dedupe candidates and filter to those whose - // decoded bundle structurally matches `edge_key`. POST - // iteration is sorted by `s`, but we sort the matched set - // explicitly so the output order is path-independent and - // matches the arena fast path (see #3 in the edge- - // annotations review). - let mut matched_anns: Vec = Vec::new(); - let mut seen: HashSet = HashSet::new(); - for cand in &candidate_flakes { - let ann_sid = &cand.s; - if !seen.insert(ann_sid.clone()) { - continue; - } - - // Structural bundle decode bypasses view policy by - // design — same justification as - // `is_live_annotation_subject`. The `f:reifies*` - // flakes are system-controlled discriminators, not - // user data; running them through the policy filter - // here would let a policy that incidentally hides - // FLUREE_DB-namespace predicates collapse the - // bundle decode and drop the annotation entirely, - // even when the annotation body would have been - // policy-visible. Reaching the body still goes - // through `format_subject` below, which applies - // policy normally — so user-data visibility is - // unchanged. - let bundle: Vec = self - .db - .range( - IndexType::Spot, - RangeTest::Eq, - RangeMatch::subject(ann_sid.clone()), - ) - .await - .map_err(|e| { - FormatError::InvalidBinding(format!( - "annotation bundle scan (SPOT s=ann) failed: {e}" - )) - })? - .into_iter() - .filter(|f| fluree_db_core::is_reserved_reifies_predicate(&f.p)) - .collect(); - if bundle.is_empty() { - continue; - } - - let cand_edge = match fluree_db_core::edge::EdgeKey::from_reifies_facts(&bundle) { - Ok(k) => k, - Err(_) => continue, - }; - if cand_edge != edge_key { - continue; - } - - matched_anns.push(ann_sid.clone()); - } - matched_anns.sort(); - - tracing::Span::current().record("annotation_count", matched_anns.len()); - self.render_annotation_bodies(&matched_anns, depth, visited, cache) + let mut reifiers: Vec = links.into_iter().map(|f| f.s).collect(); + reifiers.sort(); + reifiers.dedup(); + tracing::Span::current().record("annotation_count", reifiers.len()); + self.render_annotation_bodies(&reifiers, depth, visited, cache) .await } .instrument(span) .await } - /// Render the annotation bodies for a list of resolved annotation - /// SIDs (arena-fast-path output). Shared with the scan fallback's - /// body-rendering loop above through the same wildcard select - /// spec. + /// False only when neither the index nor novelty has ever held an + /// annotation, so hydration of an unannotated ledger pays no probe. + fn may_hold_annotations(&self) -> bool { + self.db.snapshot.has_annotations + || self + .novelty + .is_none_or(fluree_db_novelty::Novelty::has_annotations) + } + + /// Render the annotation bodies of `ann_sids` through a wildcard + /// select spec. async fn render_annotation_bodies<'b>( &'b self, ann_sids: &[Sid], @@ -2045,115 +1898,29 @@ impl<'a> HydrationFormatter<'a> { Ok(JsonValue::Object(wrapper)) } - /// Returns `true` iff `sid` is a currently-asserted annotation - /// subject. Consults the arena reader (when present) and the - /// novelty overlay's `AttachmentNovelty` reverse map. Bypasses - /// view policy by design — annotation membership is a structural - /// snapshot property, and the wildcard-hide rule must hold even - /// when policy denies the discriminating `f:reifies*` flakes. - /// - /// Returns `false` for non-blank-node SIDs without a lookup - /// (every caller already gates on `BLANK_NODE`, but the check - /// is cheap and keeps the helper safe to use elsewhere). + /// Returns `true` iff `sid` is a blank node with a live `rdf:reifies` + /// link. Bypasses view policy by design — annotation membership is a + /// structural snapshot property, and the wildcard-hide rule must hold + /// even when policy denies the link. async fn is_live_annotation_subject(&self, sid: &Sid) -> Result { - if sid.namespace_code != BLANK_NODE { + if sid.namespace_code != BLANK_NODE || !self.may_hold_annotations() { return Ok(false); } - // Overlay-side: AttachmentNovelty's reverse map answers - // "does this ann SID have any live target?" without policy. - let novelty_events: Vec<(fluree_db_core::edge::EdgeKey, i64, bool)> = self - .novelty - .map(|n| n.attachments.collect_reverse_events(sid)) - .unwrap_or_default(); - if let Some(reader) = self.arena_reader.as_ref() { - let live = reader - .current_targets_merged(sid, &novelty_events, self.db.t) - .await - .map_err(|e| { - FormatError::InvalidBinding(format!("annotation membership lookup failed: {e}")) - })?; - return Ok(!live.is_empty()); - } - // No arena: try the overlay's current_targets_for first - // (the merge helper's edge-cancellation logic over the live - // events). - let novelty_says_live = self - .novelty - .is_some_and(|n| n.attachments.current_targets_for(sid).next().is_some()); - if novelty_says_live { - return Ok(true); - } - - // Fall back to an indexed-base SPOT probe. Without this, a - // ledger whose annotation flakes have rolled into the base - // index but never had an arena sealed (M2a scan-fallback - // ledgers, or any snapshot opened without a content_store) - // would leak the anonymous annotation SID as a top-level - // wildcard row — the discriminator (`f:reifiesSubject` on - // the SID) is in base storage but neither the arena reader - // nor the overlay has it. SPOT(s = sid) is the same shape - // `fetch_subject_properties` uses and survives the - // blank-node-subject quirk in practice (verified by the - // scan-fallback hydration path at `lookup_annotation_bodies`). - let f_reifies_subject = Sid::new(FLUREE_DB, fluree_vocab::db::REIFIES_SUBJECT); - let flakes = self + let links = self .db .range( IndexType::Spot, RangeTest::Eq, - RangeMatch::subject(sid.clone()), + RangeMatch::subject_predicate( + sid.clone(), + fluree_db_core::rdf_reifies_sid().clone(), + ), ) .await .map_err(|e| { - FormatError::InvalidBinding(format!("annotation membership SPOT probe failed: {e}")) + FormatError::InvalidBinding(format!("annotation membership probe failed: {e}")) })?; - Ok(flakes.iter().any(|f| f.p == f_reifies_subject)) - } - - /// Arena-backed annotation lookup. Returns `Some(sids)` when the - /// arena reader resolved the query, `None` when a precondition - /// failed (no cached reader, or the overlay is not the expected - /// concrete novelty type) — caller falls back to the M2a scan - /// path. - /// - /// Reuses the formatter's cached `arena_reader` so successive - /// edge lookups in the same response amortize branch + leaf - /// loads. - async fn arena_lookup_annotations( - &self, - edge_key: &fluree_db_core::edge::EdgeKey, - ) -> Result>> { - use tracing::Instrument; - let span = tracing::debug_span!( - "annotation_arena_lookup", - live_count = tracing::field::Empty, - ); - async { - let Some(reader) = self.arena_reader.as_ref() else { - return Ok(None); - }; - // The overlay must be the concrete `Novelty` type so we - // can reach `AttachmentNovelty`. If it isn't (test fakes, - // future overlay variants), bail to the scan path — - // proceeding with an empty novelty event slice would let - // the arena report stale indexed attachments while the - // overlay still holds unobserved retracts. Downcast was - // cached at formatter construction; no per-call dispatch. - let Some(novelty) = self.novelty else { - return Ok(None); - }; - let novelty_events = novelty.attachments.collect_forward_events(edge_key); - let live = reader - .current_annotations_merged(edge_key, &novelty_events, self.db.t) - .await - .map_err(|e| { - FormatError::InvalidBinding(format!("annotation arena lookup failed: {e}")) - })?; - tracing::Span::current().record("live_count", live.len()); - Ok(Some(live)) - } - .instrument(span) - .await + Ok(!links.is_empty()) } /// Format reverse property values diff --git a/fluree-db-api/src/import.rs b/fluree-db-api/src/import.rs index a5806a524c..2bc29c024e 100644 --- a/fluree-db-api/src/import.rs +++ b/fluree-db-api/src/import.rs @@ -6951,10 +6951,7 @@ where // Sticky bit (computed before move into the struct literal). let import_has_annotations = predicate_sids_v6.iter().any(|(ns, name)| { - fluree_db_core::is_reserved_reifies_predicate(&fluree_db_core::Sid::new( - *ns, - name.as_str(), - )) + fluree_db_core::is_annotation_predicate(&fluree_db_core::Sid::new(*ns, name.as_str())) }); let root_v6 = IndexRoot { diff --git a/fluree-db-api/src/lib.rs b/fluree-db-api/src/lib.rs index 45796e063b..3ef5694f44 100644 --- a/fluree-db-api/src/lib.rs +++ b/fluree-db-api/src/lib.rs @@ -5015,7 +5015,7 @@ fn cypher_delete_predicate_is_relationship( _ => return Ok(false), }; - Ok(!fluree_db_core::is_rdf_type(&sid) && !fluree_db_core::is_reserved_reifies_predicate(&sid)) + Ok(!fluree_db_core::is_rdf_type(&sid) && !fluree_db_core::is_annotation_predicate(&sid)) } // ============================================================================ diff --git a/fluree-db-api/src/query/helpers.rs b/fluree-db-api/src/query/helpers.rs index b56d00d8bd..dd43e4df1a 100644 --- a/fluree-db-api/src/query/helpers.rs +++ b/fluree-db-api/src/query/helpers.rs @@ -289,17 +289,17 @@ pub(crate) fn lower_cypher_ast_to_ir( /// value-only bound relationship variables' per-hop OPTIONAL annotation /// probe. Three tiers, each conservative (`true`) when it can't decide: /// -/// 1. Dictionary: `f:reifiesSubject` never entered the dictionary — no +/// 1. Dictionary: `rdf:reifies` never entered the dictionary — no /// annotation was ever written; certain `false`. -/// 2. Index stats: per-property counts show `f:reifiesSubject` facts. -/// 3. Overlay: one PSOT walk answering "any `f:reifiesSubject` flake in +/// 2. Index stats: per-property counts show `rdf:reifies` links. +/// 3. Overlay: one PSOT walk answering "any `rdf:reifies` link in /// novelty?", cached process-wide on `content_version` — callers that /// can't supply the overlay stay conservative. fn reified_edges_possible( snapshot: &LedgerSnapshot, overlay: Option<(&dyn OverlayProvider, u16)>, ) -> bool { - let Some(reifies_sid) = snapshot.encode_iri(fluree_vocab::reifies_iris::SUBJECT) else { + let Some(reifies_sid) = snapshot.encode_iri(fluree_vocab::rdf::REIFIES) else { return false; }; if index_has_reified_edges(snapshot, &reifies_sid) { diff --git a/fluree-db-api/src/tx.rs b/fluree-db-api/src/tx.rs index ce174e346e..1e1a047490 100644 --- a/fluree-db-api/src/tx.rs +++ b/fluree-db-api/src/tx.rs @@ -2196,8 +2196,7 @@ fn bounded_read_subjects( return None; } let retracts = !txn.delete_templates.is_empty() || txn.txn_type == TxnType::Upsert; - if retracts && (ledger.snapshot.has_annotations || ledger.novelty.attachments.has_annotations()) - { + if retracts && (ledger.snapshot.has_annotations || ledger.novelty.has_annotations()) { return None; } @@ -2532,52 +2531,21 @@ fn convert_named_graphs_to_templates( } } - // TriG-star: one `f:reifies*` bundle per reifier attachment, in the - // same graph as the edge it reifies — the shape the JSON-LD - // `@annotation` sibling produces (f:reifiesGraph present, no - // f:reifiesDatatype, f:reifiesLang for language-tagged objects). - if !block.reified.is_empty() { - use fluree_db_core::namespaces::{ - reifies_graph_sid, reifies_lang_sid, reifies_object_sid, reifies_predicate_sid, - reifies_subject_sid, - }; - let graph_sid = ns_registry.sid_for_iri(&block.iri); - for r in &block.reified { - let ann = convert_term(&r.reifier, &block.prefixes, ns_registry)?; - let s = convert_term(&r.subject, &block.prefixes, ns_registry)?; - let p = convert_term(&r.predicate, &block.prefixes, ns_registry)?; - let (o, dtc) = convert_object(&r.object, &block.prefixes, ns_registry)?; - let lang = match &dtc { - Some(DatatypeConstraint::LangTag(lang)) => Some(lang.to_string()), - _ => None, - }; - let mut push = |pred: &fluree_db_core::Sid, - obj: TemplateTerm, - dtc: Option| { - let mut t = - TripleTemplate::new(ann.clone(), TemplateTerm::Sid(pred.clone()), obj) - .in_graph(std::sync::Arc::clone(&graph)); - if let Some(d) = dtc { - t = t.with_dtc(d); - } - templates.push(t); - }; - push( - reifies_graph_sid(), - TemplateTerm::Sid(graph_sid.clone()), - None, - ); - push(reifies_subject_sid(), s, None); - push(reifies_predicate_sid(), p, None); - if let Some(lang) = lang { - push( - reifies_lang_sid(), - TemplateTerm::Value(fluree_db_core::FlakeValue::String(lang)), - None, - ); - } - push(reifies_object_sid(), o, dtc); - } + // TriG-star: each reified triple's link, in the triple's graph. + for r in &block.reified { + let ann = convert_term(&r.reifier, &block.prefixes, ns_registry)?; + let s = convert_term(&r.subject, &block.prefixes, ns_registry)?; + let p = convert_term(&r.predicate, &block.prefixes, ns_registry)?; + let (o, dtc) = convert_object(&r.object, &block.prefixes, ns_registry)?; + let term = fluree_db_transact::TemplateTripleTerm { s, p, o, dtc }; + templates.push( + TripleTemplate::new( + ann, + TemplateTerm::Sid(fluree_db_core::rdf_reifies_sid().clone()), + TemplateTerm::TripleTerm(Box::new(term)), + ) + .in_graph(std::sync::Arc::clone(&graph)), + ); } } diff --git a/fluree-db-api/src/tx_builder.rs b/fluree-db-api/src/tx_builder.rs index c9ec4cf26d..1063552b8b 100644 --- a/fluree-db-api/src/tx_builder.rs +++ b/fluree-db-api/src/tx_builder.rs @@ -1706,7 +1706,7 @@ impl Fluree { let changes_dependencies = staged.iter().any(|flake| { use fluree_vocab::namespaces::{RDFS, SHACL}; flake.g.is_some() - || fluree_db_core::is_reserved_reifies_predicate(&flake.p) + || fluree_db_core::is_annotation_predicate(&flake.p) || flake.p.namespace_code == SHACL || matches!(&flake.o, fluree_db_core::FlakeValue::Ref(o) if o.namespace_code == SHACL) || (flake.p.namespace_code == RDFS diff --git a/fluree-db-api/tests/it_edge_annotations.rs b/fluree-db-api/tests/it_edge_annotations.rs index 1dfc3c846b..1e689b7c99 100644 --- a/fluree-db-api/tests/it_edge_annotations.rs +++ b/fluree-db-api/tests/it_edge_annotations.rs @@ -1061,8 +1061,8 @@ async fn subject_expansion_emits_annotation_block_for_annotated_edge() { // System facts must still be filtered. for k in ann_obj.keys() { assert!( - !k.starts_with("https://ns.flur.ee/db#reifies"), - "f:reifies* must not leak into @annotation body: {k}" + !k.contains("reifies"), + "the link must not leak into @annotation body: {k}" ); } } @@ -1904,8 +1904,8 @@ async fn variable_predicate_scan_hides_f_reifies_in_named_graph() { .collect(); for p in &predicates { assert!( - !p.starts_with("https://ns.flur.ee/db#reifies"), - "f:reifies* must not leak from named-graph variable-predicate scan: {p} \ + !p.contains("reifies"), + "the link must not leak from named-graph variable-predicate scan: {p} \ (full bindings: {predicates:?})" ); } @@ -1975,14 +1975,10 @@ async fn variable_predicate_scan_hides_f_reifies() { // No `f:reifies*` predicate may leak. for p in &predicates { assert!( - !p.starts_with("https://ns.flur.ee/db#reifies"), - "f:reifies* must not leak through variable-predicate scan: {p} \ + !p.contains("reifies"), + "the link must not leak through variable-predicate scan: {p} \ (full bindings: {predicates:?})" ); - assert!( - !p.starts_with("f:reifies"), - "compact f:reifies* form must not leak: {p}" - ); } // The user-authored `ex:role` must still be visible. assert!( @@ -2042,8 +2038,8 @@ async fn opts_include_system_facts_does_not_relax_direct_mention_firewall() { async fn opts_include_system_facts_surfaces_f_reifies() { // The `opts.includeSystemFacts: true` escape disables the // variable-predicate filter so debug / inspection callers can see - // the underlying `f:reifies*` system facts. Without the flag, the - // filter hides them (covered by `variable_predicate_scan_hides_f_reifies`). + // the annotation's `rdf:reifies` link. Without the flag, the filter + // hides it (covered by `variable_predicate_scan_hides_f_reifies`). let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "it/edge-annotations:opts-include-system-facts"; let ledger0 = genesis_ledger(&fluree, ledger_id); @@ -2086,15 +2082,9 @@ async fn opts_include_system_facts_surfaces_f_reifies() { }) .collect(); - // With the escape, all three required `f:reifies*` predicates are - // visible from the annotation subject. - let leaked_reifies: Vec<&String> = predicates - .iter() - .filter(|p| p.starts_with("https://ns.flur.ee/db#reifies") || p.starts_with("f:reifies")) - .collect(); assert!( - leaked_reifies.len() >= 3, - "opts.includeSystemFacts: true must surface the f:reifies* bundle \ + predicates.iter().any(|p| p == fluree_vocab::rdf::REIFIES), + "opts.includeSystemFacts: true must surface the rdf:reifies link \ (got predicates: {predicates:?})" ); } @@ -2152,12 +2142,8 @@ async fn opts_include_system_facts_propagates_through_dataset_path() { .or_else(|| v.get("@id").and_then(|i| i.as_str()).map(String::from)) }) .collect(); - let leaked: Vec<&String> = predicates - .iter() - .filter(|p| p.starts_with("https://ns.flur.ee/db#reifies") || p.starts_with("f:reifies")) - .collect(); assert!( - leaked.len() >= 3, + predicates.iter().any(|p| p == fluree_vocab::rdf::REIFIES), "dataset-path query must propagate opts.includeSystemFacts to the scan operator \ (got predicates: {predicates:?})" ); @@ -2195,15 +2181,14 @@ async fn opts_include_system_facts_works_for_ask_queries() { // Ask whether the annotation subject has *any* predicate. With // the filter on (default) this still answers true via the - // ex:role flake. Pin the discriminating shape: ask via a SID- - // bound predicate of `f:reifiesSubject`-via-variable that only - // matches when the f:reifies* row passes the scan filter. + // ex:role flake. Pin the discriminating shape: a variable-predicate + // row whose object is a triple term, which only the link is. let q_default = json!({ "@context": ctx(), - "ask": [{ - "@id": "ex:emp/alice-acme", - "?p": { "@id": "ex:alice" } - }] + "ask": [ + { "@id": "ex:emp/alice-acme", "?p": "?o" }, + ["filter", "(istriple ?o)"] + ] }); let resp_default = fluree .query(&support::graphdb_from_ledger(&ledger), &q_default) @@ -2215,17 +2200,17 @@ async fn opts_include_system_facts_works_for_ask_queries() { assert_eq!( json_default, JsonValue::Bool(false), - "without includeSystemFacts, ASK over a hidden f:reifies* row must answer false: {json_default}" + "without includeSystemFacts, ASK over the hidden link must answer false: {json_default}" ); // With the opt-in, the ASK now returns true because the scan - // filter is bypassed and the f:reifiesSubject row binds. + // filter is bypassed and the link binds. let q_opt = json!({ "@context": ctx(), - "ask": [{ - "@id": "ex:emp/alice-acme", - "?p": { "@id": "ex:alice" } - }], + "ask": [ + { "@id": "ex:emp/alice-acme", "?p": "?o" }, + ["filter", "(istriple ?o)"] + ], "opts": { "includeSystemFacts": true } }); let resp_opt = fluree @@ -2236,7 +2221,7 @@ async fn opts_include_system_facts_works_for_ask_queries() { assert_eq!( json_opt, JsonValue::Bool(true), - "ASK + opts.includeSystemFacts must surface f:reifies* rows: {json_opt}" + "ASK + opts.includeSystemFacts must surface the link: {json_opt}" ); } @@ -2363,19 +2348,12 @@ async fn wildcard_subject_hydration_hides_f_reifies_predicates() { "user-authored ex:role must remain visible under wildcard hydration: {node:#?}" ); - // No `f:reifies*` predicate may appear under any namespace form - // (full IRI or compact alias). The hydration formatter compacts - // through the query's `@context`, but we don't declare an `f:` - // alias in our test ctx, so any leak would surface as the - // expanded IRI. + // The link may not appear under any namespace form (full IRI or + // compact alias). for key in node.keys() { assert!( - !key.starts_with("https://ns.flur.ee/db#reifies"), - "f:reifies* predicate '{key}' must not leak through wildcard hydration" - ); - assert!( - !key.starts_with("f:reifies"), - "compact f:reifies* form '{key}' must not leak" + !key.contains("reifies"), + "the link '{key}' must not leak through wildcard hydration" ); } } @@ -2621,7 +2599,7 @@ async fn delete_by_annotation_id_retracts_only_targeted_occurrence() { // ex:emp/A → role=Engineer // ex:emp/B → role=Manager // A delete with `@annotation: { @id: ex:emp/A }` must retract - // only A's f:reifies* bundle. B and the base edge survive + // only A's `rdf:reifies` link. B and the base edge survive // unchanged. This is the design's "Delete by Annotation Id" // shape — exactly that occurrence, not the base edge. let fluree = FlureeBuilder::memory().build_memory(); @@ -2670,13 +2648,10 @@ async fn delete_by_annotation_id_retracts_only_targeted_occurrence() { ) .await .expect("delete by annotation id"); - assert_eq!( - r2.receipt.flake_count, 3, - "exactly three f:reifies* retracts (subject/predicate/object)" - ); + assert_eq!(r2.receipt.flake_count, 1, "exactly A's link retract"); // Surviving annotations: only B should appear in `@reifies` - // queries because A's bundle is gone. + // queries because A's link is gone. let surviving = json!({ "@context": ctx(), "select": ["?ann", "?role"], @@ -2926,15 +2901,11 @@ async fn delete_by_annotation_id_named_graph_retracts_in_correct_graph() { ) .await .expect("named-graph by-id delete"); - // Three retract flakes for the bundle (subject/predicate/object). - // f:reifiesGraph is also retracted because the synthesized - // template carries it explicitly to match the original - // assertion's identity. So flake_count = 4: subject + predicate - // + object + reifiesGraph. + // The link retract, in the link's graph: flake identity includes `g`, + // so a default-graph retract would cancel nothing. assert_eq!( - r2.receipt.flake_count, 4, - "named-graph by-id retract must cancel all four reifies* flakes \ - emitted at insert time (subject/predicate/object + reifiesGraph)" + r2.receipt.flake_count, 1, + "named-graph by-id retract must cancel the link" ); // The annotation should no longer surface via @reifies. If the @@ -3587,16 +3558,15 @@ async fn graph_wrapped_query_correctly_pairs_annotations_per_graph() { // EdgeKey round-trip gate tests — literal-object annotations // ===================================================================== // -// Contract: every writer of a reifies bundle (insert sibling, in this -// case) must produce flakes whose decoded `EdgeKey::from_reifies_facts` -// equals the `EdgeKey::from_flake` of the base edge. If these two -// disagree, hydration / cascade / by-selector retract all silently fail -// to find each other. +// Contract: every annotation writer (insert sibling, in this case) must +// produce a link whose triple term equals the `EdgeKey::from_flake` of the +// base edge. If these two disagree, hydration / cascade / by-selector +// retract all silently fail to find each other. // // These tests exercise the full JSON-LD expansion → staging → flake -// pipeline and decode the resulting bundle. They are the load-bearing +// pipeline and read the resulting link. They are the load-bearing // contract that protects against future drift between writer and -// decoder. +// reader. async fn edgekey_roundtrip_for_literal( predicate_compact: &str, @@ -3607,8 +3577,6 @@ async fn edgekey_roundtrip_for_literal( use fluree_db_core::comparator::IndexType; use fluree_db_core::edge::EdgeKey; use fluree_db_core::range::{range_with_overlay, RangeMatch, RangeOptions, RangeTest}; - use fluree_db_core::value::FlakeValue; - use fluree_vocab::reifies_iris; let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = format!("it-edge-annotations-edgekey-roundtrip-{test_label}"); @@ -3638,11 +3606,6 @@ async fn edgekey_roundtrip_for_literal( .snapshot .encode_iri(predicate_full) .expect("encode predicate IRI"); - let reifies_subject_pid = ledger - .snapshot - .encode_iri(reifies_iris::SUBJECT) - .expect("encode f:reifiesSubject"); - // 1. Base-edge flake. let base_flakes = range_with_overlay( &ledger.snapshot, @@ -3663,59 +3626,20 @@ async fn edgekey_roundtrip_for_literal( let base_flake = &base_flakes[0]; let base_edge_key = EdgeKey::from_flake(base_flake); - // 2. Locate the annotation subject via POST f:reifiesSubject → alice. - let ann_pointers = range_with_overlay( - &ledger.snapshot, - 0, - ledger.novelty.as_ref(), - IndexType::Post, - RangeTest::Eq, - RangeMatch::predicate_object( - reifies_subject_pid.clone(), - FlakeValue::Ref(alice_sid.clone()), - ), - RangeOptions::new().with_to_t(ledger.t()), - ) - .await - .expect("scan f:reifiesSubject pointers"); + // 2. The annotation's link. + let links = reifies_bundles_for_subject(ledger, "http://example.org/alice").await; assert_eq!( - ann_pointers.len(), + links.len(), 1, - "[{test_label}] expected exactly one annotation; got {ann_pointers:#?}" - ); - let ann_sid = ann_pointers[0].s.clone(); - - // 3. Full bundle for that annotation subject. - let ann_flakes = range_with_overlay( - &ledger.snapshot, - 0, - ledger.novelty.as_ref(), - IndexType::Spot, - RangeTest::Eq, - RangeMatch::subject(ann_sid.clone()), - RangeOptions::new().with_to_t(ledger.t()), - ) - .await - .expect("scan annotation subject flakes"); - let bundle: Vec<_> = ann_flakes - .iter() - .filter(|f| fluree_db_core::is_reserved_reifies_predicate(&f.p)) - .cloned() - .collect(); - assert!( - !bundle.is_empty(), - "[{test_label}] annotation subject must carry an f:reifies* bundle; \ - got flakes: {ann_flakes:#?}" + "[{test_label}] expected exactly one annotation; got {links:#?}" ); - let decoded_key = EdgeKey::from_reifies_facts(&bundle).unwrap_or_else(|e| { - panic!( - "[{test_label}] decode failed: {e:?}; bundle: {bundle:#?}, base flake: {base_flake:?}" - ) - }); + let bundle = &links[0].1; + let decoded_key = edge_of(bundle) + .unwrap_or_else(|e| panic!("[{test_label}] {e}; base flake: {base_flake:?}")); - // 4. The contract — decoded EdgeKey from the synthesized sibling - // must equal the base edge's EdgeKey. Mismatch here is the - // silent-failure mode that loses annotations. + // 3. The contract — the triple the link names must equal the base + // edge. Mismatch here is the silent-failure mode that loses + // annotations. assert_eq!( decoded_key, base_edge_key, "[{test_label}] decoded EdgeKey diverges from base-edge EdgeKey; \ @@ -3765,72 +3689,79 @@ async fn edgekey_roundtrip_language_tagged_literal_annotation() { // ===================================================================== // // Contract (PR-W15): Turtle-star asserting forms (`<< s p o ~ r >>`, -// `{| … |}`) must produce BIT-IDENTICAL `f:reifies*` flakes to the -// JSON-LD `@annotation` lowering — same predicates, same object -// values/datatypes/metadata, same "no f:reifiesDatatype" shape — so -// cascade retracts, hydration, and the annotation arena treat both -// surfaces as one. Both surfaces are inserted into the SAME ledger -// (JSON-LD first, Turtle second) and their bundles compared per -// transaction time, which keeps Sid identity comparable. - -/// All `f:reifies*` bundles for annotations reifying edges whose -/// subject is `base_subject_iri`, grouped by `(ann_sid, t)`. -async fn reifies_bundles_for_subject( +// `{| … |}`) must produce BIT-IDENTICAL `rdf:reifies` links to the +// JSON-LD `@annotation` lowering — same triple term, datatype and tag — +// so cascade retracts and hydration treat both surfaces as one. Both +// surfaces are inserted into the SAME ledger (JSON-LD first, Turtle +// second) and their links compared per transaction time, which keeps Sid +// identity comparable. + +/// Live `rdf:reifies` links in graph `g_id` naming a triple whose subject is +/// `subject`, grouped by `(reifier, t)`. +async fn links_for_subject_in( ledger: &MemoryLedger, - base_subject_iri: &str, + g_id: fluree_db_core::GraphId, + subject: &fluree_db_core::Sid, + to_t: i64, ) -> Vec<((fluree_db_core::Sid, i64), Vec)> { use fluree_db_core::comparator::IndexType; use fluree_db_core::range::{range_with_overlay, RangeMatch, RangeOptions, RangeTest}; use fluree_db_core::value::FlakeValue; - use fluree_vocab::reifies_iris; use std::collections::BTreeMap; - let subject_sid = ledger - .snapshot - .encode_iri(base_subject_iri) - .expect("encode base subject IRI"); - let reifies_subject_pid = ledger - .snapshot - .encode_iri(reifies_iris::SUBJECT) - .expect("encode f:reifiesSubject"); - - let pointers = range_with_overlay( + let links = range_with_overlay( &ledger.snapshot, - 0, + g_id, ledger.novelty.as_ref(), - IndexType::Post, + IndexType::Psot, RangeTest::Eq, - RangeMatch::predicate_object(reifies_subject_pid, FlakeValue::Ref(subject_sid)), - RangeOptions::new().with_to_t(ledger.t()), + RangeMatch::predicate(fluree_db_core::rdf_reifies_sid().clone()), + RangeOptions::new().with_to_t(to_t), ) .await - .expect("scan f:reifiesSubject pointers"); - - let mut bundles: BTreeMap<(fluree_db_core::Sid, i64), Vec> = - BTreeMap::new(); - for ptr in &pointers { - let ann_sid = ptr.s.clone(); - let ann_flakes = range_with_overlay( - &ledger.snapshot, - 0, - ledger.novelty.as_ref(), - IndexType::Spot, - RangeTest::Eq, - RangeMatch::subject(ann_sid.clone()), - RangeOptions::new().with_to_t(ledger.t()), - ) - .await - .expect("scan annotation subject flakes"); - for f in ann_flakes { - if fluree_db_core::is_reserved_reifies_predicate(&f.p) { - bundles.entry((ann_sid.clone(), f.t)).or_default().push(f); - } + .expect("scan rdf:reifies links"); + let mut out: BTreeMap<(fluree_db_core::Sid, i64), Vec> = BTreeMap::new(); + for link in links { + if matches!(&link.o, FlakeValue::TripleTerm(term) if term.s == *subject) { + out.entry((link.s.clone(), link.t)).or_default().push(link); } } - bundles.into_iter().collect() + out.into_iter().collect() +} + +/// Every annotation reifying an edge whose subject is `base_subject_iri`, +/// in the default graph: its `rdf:reifies` links, grouped by `(ann_sid, t)`. +async fn reifies_bundles_for_subject( + ledger: &MemoryLedger, + base_subject_iri: &str, +) -> Vec<((fluree_db_core::Sid, i64), Vec)> { + let subject_sid = ledger + .snapshot + .encode_iri(base_subject_iri) + .expect("encode base subject IRI"); + links_for_subject_in(ledger, 0, &subject_sid, ledger.t()).await +} + +/// The edge a reifier's single link names. +fn edge_of(links: &[fluree_db_core::Flake]) -> Result { + match links { + [link] => match &link.o { + fluree_db_core::FlakeValue::TripleTerm(term) => Ok(fluree_db_core::edge::EdgeKey { + g: None, + s: term.s.clone(), + p: term.p.clone(), + o: term.o.clone(), + dt: term.dt.clone(), + lang: term.lang.clone(), + list_i: None, + }), + other => Err(format!("link object is not a triple term: {other:?}")), + }, + _ => Err(format!("expected one link, got {links:#?}")), + } } -/// Normalize a bundle to its time/subject-agnostic body: +/// Normalize a reifier's links to their time/subject-agnostic body: /// sorted `(predicate, object, datatype, meta)` rows. fn bundle_body( bundle: &[fluree_db_core::Flake], @@ -3849,8 +3780,8 @@ fn bundle_body( } /// Shared driver: insert the JSON-LD form, then the Turtle-star form, -/// into one ledger; assert the two bundles' bodies are identical and -/// both decode to the base edge's EdgeKey. +/// into one ledger; assert the two links are identical and both name the +/// base edge. async fn assert_turtle_star_matches_jsonld( ledger_id: &str, jsonld_txn: serde_json::Value, @@ -3932,8 +3863,7 @@ async fn assert_turtle_star_matches_jsonld( // Every bundle decodes to the base edge's EdgeKey. for ((ann, _), bundle) in &bundles { - let decoded = EdgeKey::from_reifies_facts(bundle) - .unwrap_or_else(|e| panic!("bundle for {ann:?} failed to decode: {e:?}")); + let decoded = edge_of(bundle).unwrap_or_else(|e| panic!("link for {ann:?}: {e}")); assert_eq!( decoded, base_key, "bundle for {ann:?} reifies the base edge" @@ -3972,9 +3902,8 @@ async fn turtle_star_named_reifier_matches_jsonld_named_annotation() { // is an idempotent no-op — the engine drops the duplicate flakes — // so the same pair can't be exercised twice in one ledger. Bundle // bodies are compared with the f:reifiesSubject row normalized. - use fluree_db_core::edge::EdgeKey; + use fluree_db_core::FlakeValue; - use fluree_vocab::reifies_iris; let fluree = FlureeBuilder::memory().build_memory(); let ledger0 = genesis_ledger(&fluree, "it/turtle-star:named-reifier"); @@ -4008,11 +3937,6 @@ async fn turtle_star_named_reifier_matches_jsonld_named_annotation() { .snapshot .encode_iri("http://example.org/r_tt") .expect("encode r_tt"); - let reifies_subject_pid = ledger - .snapshot - .encode_iri(reifies_iris::SUBJECT) - .expect("encode f:reifiesSubject"); - let jl_bundles = reifies_bundles_for_subject(ledger, "http://example.org/alice").await; let tt_bundles = reifies_bundles_for_subject(ledger, "http://example.org/bob").await; assert_eq!(jl_bundles.len(), 1, "{jl_bundles:#?}"); @@ -4027,8 +3951,8 @@ async fn turtle_star_named_reifier_matches_jsonld_named_annotation() { let normalize = |bundle: &[fluree_db_core::Flake]| { let mut body = bundle_body(bundle); for row in &mut body { - if row.0 == reifies_subject_pid { - row.1 = FlakeValue::String("".to_string()); + if let FlakeValue::TripleTerm(term) = &mut row.1 { + term.s = fluree_db_core::Sid::new(0, ""); } } body.sort_by_key(|row| format!("{row:?}")); @@ -4037,7 +3961,7 @@ async fn turtle_star_named_reifier_matches_jsonld_named_annotation() { assert_eq!( normalize(&jl_bundles[0].1), normalize(&tt_bundles[0].1), - "named-reifier bundle bodies must match modulo the base subject" + "named-reifier links must match modulo the base subject" ); // Each decodes to its own base edge. @@ -4045,7 +3969,7 @@ async fn turtle_star_named_reifier_matches_jsonld_named_annotation() { (&jl_bundles[0].1, "http://example.org/alice"), (&tt_bundles[0].1, "http://example.org/bob"), ] { - let decoded = EdgeKey::from_reifies_facts(bundle).expect("bundle decodes"); + let decoded = edge_of(bundle).expect("one link"); let expected_subject = ledger.snapshot.encode_iri(subject_iri).expect("encode"); assert_eq!(decoded.s, expected_subject); } @@ -4104,13 +4028,6 @@ async fn turtle_rdf_reifies_triple_term_matches_jsonld_annotation() { /// and the annotation is visible to a named-graph-scoped query. #[tokio::test] async fn trig_star_in_graph_block_matches_jsonld_named_graph_annotation() { - use fluree_db_core::comparator::IndexType; - use fluree_db_core::edge::EdgeKey; - use fluree_db_core::range::{range_with_overlay, RangeMatch, RangeOptions, RangeTest}; - use fluree_db_core::value::FlakeValue; - use fluree_vocab::reifies_iris; - use std::collections::BTreeMap; - let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "it/trig-star:named-graph"; let ledger0 = genesis_ledger(&fluree, ledger_id); @@ -4152,46 +4069,11 @@ async fn trig_star_in_graph_block_matches_jsonld_named_graph_annotation() { .graph_registry .graph_id_for_iri(graph_iri) .expect("named graph registered"); - let graph_sid = ledger.snapshot.encode_iri(graph_iri).expect("graph sid"); let subject_sid = ledger .snapshot .encode_iri("http://example.org/alice") .expect("subject sid"); - let reifies_subject_pid = ledger - .snapshot - .encode_iri(reifies_iris::SUBJECT) - .expect("f:reifiesSubject sid"); - let pointers = range_with_overlay( - &ledger.snapshot, - g_id, - ledger.novelty.as_ref(), - IndexType::Post, - RangeTest::Eq, - RangeMatch::predicate_object(reifies_subject_pid, FlakeValue::Ref(subject_sid.clone())), - RangeOptions::new().with_to_t(t_trig), - ) - .await - .expect("scan named-graph f:reifiesSubject pointers"); - let mut bundles: BTreeMap<(fluree_db_core::Sid, i64), Vec> = - BTreeMap::new(); - for ptr in &pointers { - let ann_flakes = range_with_overlay( - &ledger.snapshot, - g_id, - ledger.novelty.as_ref(), - IndexType::Spot, - RangeTest::Eq, - RangeMatch::subject(ptr.s.clone()), - RangeOptions::new().with_to_t(t_trig), - ) - .await - .expect("scan annotation subject"); - for f in ann_flakes { - if fluree_db_core::is_reserved_reifies_predicate(&f.p) { - bundles.entry((ptr.s.clone(), f.t)).or_default().push(f); - } - } - } + let bundles = links_for_subject_in(&ledger, g_id, &subject_sid, t_trig).await; let mut jl: Vec<_> = bundles .iter() .filter(|((_, t), _)| *t == t_jsonld) @@ -4218,14 +4100,9 @@ async fn trig_star_in_graph_block_matches_jsonld_named_graph_annotation() { jl, tt, "TriG-star bundle must be bit-identical to the JSON-LD @graph/@annotation bundle" ); + // Found by the named graph's scan, so both links live in it. for ((ann, _), bundle) in &bundles { - let key = EdgeKey::from_reifies_facts(bundle) - .unwrap_or_else(|e| panic!("bundle for {ann:?} failed to decode: {e:?}")); - assert_eq!( - key.g.as_ref(), - Some(&graph_sid), - "bundle for {ann:?} is graph-anchored" - ); + let key = edge_of(bundle).unwrap_or_else(|e| panic!("link for {ann:?}: {e}")); assert_eq!(key.s, subject_sid); } @@ -4307,7 +4184,6 @@ async fn turtle_star_repeated_anonymous_occurrences_mint_fresh_reifiers() { // Turtle must never dedup by EdgeKey), each with a complete bundle // decoding to the same base EdgeKey. W3C `pattern-3-nomatch` depends // on exactly this behavior. - use fluree_db_core::edge::EdgeKey; let fluree = FlureeBuilder::memory().build_memory(); let ledger0 = genesis_ledger(&fluree, "it/turtle-star:repeated-anon"); @@ -4362,8 +4238,7 @@ async fn turtle_star_repeated_anonymous_occurrences_mint_fresh_reifiers() { // multisets match across surfaces. let mut keys = Vec::new(); for (ann, bundle) in jsonld_anns.iter().chain(turtle_anns.iter()) { - let key = EdgeKey::from_reifies_facts(bundle) - .unwrap_or_else(|e| panic!("bundle for {ann:?} failed to decode: {e:?}")); + let key = edge_of(bundle).unwrap_or_else(|e| panic!("link for {ann:?}: {e}")); keys.push(key); } assert!( @@ -4475,16 +4350,11 @@ async fn turtle_star_anonymous_mints_never_collide_with_user_bnode_labels() { // ===================================================================== /// Insert an annotated literal, then retract by annotation @id, and -/// verify the f:reifiesSubject pointer no longer resolves. Mirrors the +/// verify the link no longer resolves. Mirrors the /// "live cancels assert" cascade contract enforced in the M1b cascade /// tests, but for literal-object annotations. #[tokio::test] async fn delete_by_id_retracts_literal_annotation_bundle() { - use fluree_db_core::comparator::IndexType; - use fluree_db_core::range::{range_with_overlay, RangeMatch, RangeOptions, RangeTest}; - use fluree_db_core::value::FlakeValue; - use fluree_vocab::reifies_iris; - let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "it-edge-annotations-literal-delete-by-id"; let ledger0 = genesis_ledger(&fluree, ledger_id); @@ -4499,28 +4369,7 @@ async fn delete_by_id_retracts_literal_annotation_bundle() { }); let r1 = fluree.insert(ledger0, &insert).await.expect("insert"); - let alice_sid = r1 - .ledger - .snapshot - .encode_iri("http://example.org/alice") - .unwrap(); - let reifies_subj_pid = r1 - .ledger - .snapshot - .encode_iri(reifies_iris::SUBJECT) - .unwrap(); - - let pre = range_with_overlay( - &r1.ledger.snapshot, - 0, - r1.ledger.novelty.as_ref(), - IndexType::Post, - RangeTest::Eq, - RangeMatch::predicate_object(reifies_subj_pid.clone(), FlakeValue::Ref(alice_sid.clone())), - RangeOptions::new().with_to_t(r1.ledger.t()), - ) - .await - .expect("pre scan"); + let pre = reifies_bundles_for_subject(&r1.ledger, "http://example.org/alice").await; assert_eq!(pre.len(), 1, "annotation must be present pre-delete"); let delete = json!({ @@ -4535,17 +4384,7 @@ async fn delete_by_id_retracts_literal_annotation_bundle() { }); let r2 = fluree.update(r1.ledger, &delete).await.expect("delete"); - let post = range_with_overlay( - &r2.ledger.snapshot, - 0, - r2.ledger.novelty.as_ref(), - IndexType::Post, - RangeTest::Eq, - RangeMatch::predicate_object(reifies_subj_pid, FlakeValue::Ref(alice_sid)), - RangeOptions::new().with_to_t(r2.ledger.t()), - ) - .await - .expect("post scan"); + let post = reifies_bundles_for_subject(&r2.ledger, "http://example.org/alice").await; assert!( post.is_empty(), "annotation must be retracted post-delete: got {post:#?}" @@ -4736,11 +4575,6 @@ async fn subject_expansion_promotes_annotated_lang_tagged_literal() { /// the matching one. #[tokio::test] async fn delete_by_selector_retracts_literal_annotation_disambiguating_on_body() { - use fluree_db_core::comparator::IndexType; - use fluree_db_core::range::{range_with_overlay, RangeMatch, RangeOptions, RangeTest}; - use fluree_db_core::value::FlakeValue; - use fluree_vocab::reifies_iris; - let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "it-edge-annotations-literal-delete-selector"; let ledger0 = genesis_ledger(&fluree, ledger_id); @@ -4764,28 +4598,7 @@ async fn delete_by_selector_retracts_literal_annotation_disambiguating_on_body() }); let r1 = fluree.insert(ledger0, &insert).await.expect("insert two"); - let alice_sid = r1 - .ledger - .snapshot - .encode_iri("http://example.org/alice") - .unwrap(); - let reifies_subj_pid = r1 - .ledger - .snapshot - .encode_iri(reifies_iris::SUBJECT) - .unwrap(); - - let pre = range_with_overlay( - &r1.ledger.snapshot, - 0, - r1.ledger.novelty.as_ref(), - IndexType::Post, - RangeTest::Eq, - RangeMatch::predicate_object(reifies_subj_pid.clone(), FlakeValue::Ref(alice_sid.clone())), - RangeOptions::new().with_to_t(r1.ledger.t()), - ) - .await - .expect("pre scan"); + let pre = reifies_bundles_for_subject(&r1.ledger, "http://example.org/alice").await; assert_eq!(pre.len(), 2, "two annotations present pre-delete"); // Selector-form delete: retract the annotation whose body says @@ -4805,24 +4618,14 @@ async fn delete_by_selector_retracts_literal_annotation_disambiguating_on_body() let r2 = fluree.update(r1.ledger, &delete).await.expect("delete"); // One annotation must remain (the payroll one). - let post = range_with_overlay( - &r2.ledger.snapshot, - 0, - r2.ledger.novelty.as_ref(), - IndexType::Post, - RangeTest::Eq, - RangeMatch::predicate_object(reifies_subj_pid, FlakeValue::Ref(alice_sid)), - RangeOptions::new().with_to_t(r2.ledger.t()), - ) - .await - .expect("post scan"); + let post = reifies_bundles_for_subject(&r2.ledger, "http://example.org/alice").await; assert_eq!( post.len(), 1, "exactly one annotation must survive selector delete (the payroll one); got {post:#?}" ); // The surviving annotation must be ex:ann-payroll. - let surviving_sid = &post[0].s; + let surviving_sid = &post[0].0 .0; let payroll_sid = r2 .ledger .snapshot @@ -4838,11 +4641,6 @@ async fn delete_by_selector_retracts_literal_annotation_disambiguating_on_body() /// EdgeKey has `lang = Some("fr")`. #[tokio::test] async fn delete_by_id_retracts_lang_tagged_literal_annotation_bundle() { - use fluree_db_core::comparator::IndexType; - use fluree_db_core::range::{range_with_overlay, RangeMatch, RangeOptions, RangeTest}; - use fluree_db_core::value::FlakeValue; - use fluree_vocab::reifies_iris; - let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "it-edge-annotations-literal-delete-lang"; let ledger0 = genesis_ledger(&fluree, ledger_id); @@ -4858,28 +4656,7 @@ async fn delete_by_id_retracts_lang_tagged_literal_annotation_bundle() { }); let r1 = fluree.insert(ledger0, &insert).await.expect("insert"); - let alice_sid = r1 - .ledger - .snapshot - .encode_iri("http://example.org/alice") - .unwrap(); - let reifies_subj_pid = r1 - .ledger - .snapshot - .encode_iri(reifies_iris::SUBJECT) - .unwrap(); - - let pre = range_with_overlay( - &r1.ledger.snapshot, - 0, - r1.ledger.novelty.as_ref(), - IndexType::Post, - RangeTest::Eq, - RangeMatch::predicate_object(reifies_subj_pid.clone(), FlakeValue::Ref(alice_sid.clone())), - RangeOptions::new().with_to_t(r1.ledger.t()), - ) - .await - .expect("pre scan"); + let pre = reifies_bundles_for_subject(&r1.ledger, "http://example.org/alice").await; assert_eq!(pre.len(), 1); let delete = json!({ @@ -4895,17 +4672,7 @@ async fn delete_by_id_retracts_lang_tagged_literal_annotation_bundle() { }); let r2 = fluree.update(r1.ledger, &delete).await.expect("delete"); - let post = range_with_overlay( - &r2.ledger.snapshot, - 0, - r2.ledger.novelty.as_ref(), - IndexType::Post, - RangeTest::Eq, - RangeMatch::predicate_object(reifies_subj_pid, FlakeValue::Ref(alice_sid)), - RangeOptions::new().with_to_t(r2.ledger.t()), - ) - .await - .expect("post scan"); + let post = reifies_bundles_for_subject(&r2.ledger, "http://example.org/alice").await; assert!( post.is_empty(), "lang-tagged annotation must be retracted post-delete: got {post:#?}" @@ -5000,93 +4767,45 @@ async fn cross_language_annotation_does_not_cross_match() { "both annotation bodies must be inserted; got: {bare_rows:#?}" ); - // Confirm both `f:reifiesLang` flakes landed (one per annotation). - // We scan directly because the user-facing query path filters - // system predicates from variable-predicate output. + // Confirm both links landed, each naming its own tagged literal. We scan + // directly because the user-facing query path hides `rdf:reifies` from + // variable-predicate output. The language tag is part of the term: a + // writer that dropped it would make the two links one. { use fluree_db_core::comparator::IndexType; use fluree_db_core::range::{range_with_overlay, RangeMatch, RangeOptions, RangeTest}; - use fluree_vocab::reifies_iris; - let reifies_lang_pid = ledger - .snapshot - .encode_iri(reifies_iris::LANG) - .expect("encode f:reifiesLang"); - let lang_flakes = range_with_overlay( - &ledger.snapshot, - 0, - ledger.novelty.as_ref(), - IndexType::Spot, - RangeTest::Eq, - RangeMatch::new().with_predicate(reifies_lang_pid), - RangeOptions::new().with_to_t(ledger.t()), - ) - .await - .expect("scan f:reifiesLang"); - assert_eq!(lang_flakes.len(), 2, "expected two f:reifiesLang flakes"); - - // ALSO scan the f:reifiesObject flakes to verify whether the - // language tag IS stored on the per-flake `m.lang` (in which - // case the existing `edge.dtc` clone IS the right - // disambiguator and the bug is elsewhere) or NOT stored there - // (in which case the new `f:reifiesLang` constraint triple is - // necessary). This tells us which IR shape is internally - // consistent. - let reifies_obj_pid = ledger - .snapshot - .encode_iri(reifies_iris::OBJECT) - .expect("encode f:reifiesObject"); - let obj_flakes = range_with_overlay( + let links = range_with_overlay( &ledger.snapshot, 0, ledger.novelty.as_ref(), - IndexType::Spot, + IndexType::Psot, RangeTest::Eq, - RangeMatch::new().with_predicate(reifies_obj_pid), + RangeMatch::predicate(fluree_db_core::rdf_reifies_sid().clone()), RangeOptions::new().with_to_t(ledger.t()), ) .await - .expect("scan f:reifiesObject"); - assert_eq!(obj_flakes.len(), 2, "expected two f:reifiesObject flakes"); - - // The where_plan expansion treats `edge.dtc = LangTag(...)` - // on the synthesized f:reifiesObject lookup as the per- - // language disambiguator. That depends on the writer - // storing the language tag on the flake's `m.lang`. Assert - // the exact wire layout the IR relies on so a future writer - // refactor can't silently move the language tag off the - // f:reifiesObject flake without flipping this test red. + .expect("scan rdf:reifies"); + assert_eq!(links.len(), 2, "expected two links: {links:#?}"); let rdf_lang_string_sid = ledger .snapshot .encode_iri("http://www.w3.org/1999/02/22-rdf-syntax-ns#langString") .expect("encode rdf:langString"); - for f in &obj_flakes { - assert_eq!( - f.o, - fluree_db_core::value::FlakeValue::String("chat".to_string()), - "f:reifiesObject object must be the lexical string \"chat\"; got {:?}", - f.o - ); + let mut langs = std::collections::HashSet::new(); + for link in &links { + let fluree_db_core::FlakeValue::TripleTerm(term) = &link.o else { + panic!("link object must be a triple term: {link:?}"); + }; assert_eq!( - f.dt, rdf_lang_string_sid, - "f:reifiesObject dt must be rdf:langString for lang-tagged literal; got {:?}", - f.dt - ); - assert!( - f.m.as_ref() - .and_then(|m| m.lang.as_ref()) - .is_some_and(|l| l == "fr" || l == "en"), - "f:reifiesObject must carry m.lang in {{fr,en}}; got {:?}", - f.m + term.o, + fluree_db_core::value::FlakeValue::String("chat".to_string()) ); + assert_eq!(term.dt, rdf_lang_string_sid); + langs.insert(term.lang.clone().expect("tagged")); } - let langs: std::collections::HashSet = obj_flakes - .iter() - .filter_map(|f| f.m.as_ref().and_then(|m| m.lang.clone())) - .collect(); assert_eq!( langs, ["fr".to_string(), "en".to_string()].into_iter().collect(), - "f:reifiesObject m.lang values must be exactly {{fr, en}}" + "the links' tags must be exactly {{fr, en}}" ); } @@ -5520,23 +5239,6 @@ async fn run_graph_mgmt(fluree: &MemoryFluree, ledger: MemoryLedger, sparql: &st .ledger } -/// Like [`run_graph_mgmt`] but returns the staging `Result` mapped to its -/// error string — for the negative (rejection) tests. -async fn try_run_graph_mgmt( - fluree: &MemoryFluree, - ledger: MemoryLedger, - sparql: &str, -) -> std::result::Result { - let txn = lower_graph_mgmt(&ledger.snapshot, sparql); - fluree - .stage_owned(ledger) - .txn(txn) - .execute() - .await - .map(|r| r.ledger) - .map_err(|e| e.to_string()) -} - /// Pull the annotation body's `ex:role` out of a subject-expansion /// hydration row. Tolerates compact-vs-expanded keys and bare-object /// vs single-element-array shapes (both formatter-legal). @@ -5776,11 +5478,10 @@ async fn transfer_named_to_default_drops_reifies_graph_anchor() { } #[tokio::test] -async fn add_across_graphs_with_reifier_collision_errors() { - // Same explicit reifier @id reifying DIFFERENT edges in two graphs. - // ADD merges the bundles onto one subject → a second f:reifiesSubject → - // from_reifies_facts returns Duplicate → BOTH annotations silently drop. - // The fix detects this and fails loud (COPY/MOVE are immune; see below). +async fn add_across_graphs_merges_a_shared_reifier() { + // Same explicit reifier @id reifying DIFFERENT edges in two graphs. ADD + // merges the source's link into the destination, where ex:r1 then names + // both triples — a reifier may reify several (RDF 1.2). let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "it/edge-annotations:add-reifier-collision"; let g1 = "http://example.org/g1"; @@ -5801,27 +5502,13 @@ async fn add_across_graphs_with_reifier_collision_errors() { ) .await; - let err = try_run_graph_mgmt(&fluree, ledger.clone(), &format!("ADD <{g1}> TO <{g2}>")) - .await - .expect_err("ADD with a cross-graph reifier collision must be rejected"); - assert!( - err.contains("reifier") && err.contains("r1"), - "expected a reifier-collision rejection naming the reifier, got: {err}" - ); + let ledger = run_graph_mgmt(&fluree, ledger, &format!("ADD <{g1}> TO <{g2}>")).await; - // The rejected ADD is atomic: dest g2's pre-existing ex:r1 annotation - // (reifying bob→globex) is untouched and still decodes cleanly. - let g2_ann = support::decode_annotations_for_subject( - &ledger, - graph_id(&ledger, g2), - "http://example.org/bob", - ) - .await; - assert_eq!( - g2_ann.len(), - 1, - "dest's pre-existing annotation must survive the rejected ADD" - ); + let g2_id = graph_id(&ledger, g2); + for subject in ["http://example.org/alice", "http://example.org/bob"] { + let anns = support::decode_annotations_for_subject(&ledger, g2_id, subject).await; + assert_eq!(anns.len(), 1, "g2's ex:r1 must reify {subject}'s edge"); + } } #[tokio::test] @@ -6454,76 +6141,6 @@ async fn one_reifier_on_the_same_edge_in_two_graphs_writes_in_one_transaction() } } -#[tokio::test] -async fn replaying_a_commit_does_not_refuse_a_reifier_an_older_build_wrote() { - // The single-target invariant runs on `stage_flakes`, which put it on the - // push path. `insert_turtle` and the permissive bulk-import sink did not - // enforce it before this work, so a commit written by an older build can - // hold a reifier on two edges. Refusing that on push would strand the - // ledger permanently — there is no way forward short of rewriting history - // — to prevent data that is already written. - // - // Authoring still refuses it, which is the case where refusing changes the - // outcome. This drives both sides of that distinction through the same - // flakes, so the difference is the option and nothing else. - use fluree_db_core::{Flake, FlakeValue, Sid}; - - let fluree = FlureeBuilder::memory().build_memory(); - let ledger_id = "it/edge-annotations:replayed-commit"; - let ledger0 = genesis_ledger(&fluree, ledger_id); - - let sid = |ns: u16, name: &str| Sid::new(ns, name); - let reifies = |local: &str| { - Sid::new( - fluree_vocab::namespaces::FLUREE_DB, - Box::leak(local.to_string().into_boxed_str()) as &'static str, - ) - }; - let ns = 100u16; - // One reifier, two different subjects: a bundle that cannot decode. - let assert_reifies = |object: Sid| { - Flake::new( - sid(ns, "claim1"), - reifies(fluree_vocab::db::REIFIES_SUBJECT), - FlakeValue::Ref(object), - Sid::new(0, "@id"), - 1, - true, - None, - ) - }; - let flakes = vec![ - assert_reifies(sid(ns, "alice")), - assert_reifies(sid(ns, "carol")), - ]; - - let authored = fluree_db_transact::stage_flakes( - ledger0.clone(), - flakes.clone(), - fluree_db_transact::StageOptions::new(), - ) - .await; - let err = match authored { - Err(e) => e, - Ok(_) => panic!("authoring a two-target reifier must be refused"), - }; - assert!( - err.to_string().contains("claim1"), - "the refusal must name the reifier: {err}" - ); - - let replayed = fluree_db_transact::stage_flakes( - ledger0, - flakes, - fluree_db_transact::StageOptions::new().replaying_commit(), - ) - .await; - assert!( - replayed.is_ok(), - "replaying an already-authored commit must apply, not strand the ledger" - ); -} - // =========================================================================== // Annotation-form DELETE: the full spelling matrix (issue #1861) // diff --git a/fluree-db-api/tests/it_edge_annotations_indexed.rs b/fluree-db-api/tests/it_edge_annotations_indexed.rs index 1f214604f2..287b4f44ea 100644 --- a/fluree-db-api/tests/it_edge_annotations_indexed.rs +++ b/fluree-db-api/tests/it_edge_annotations_indexed.rs @@ -1,29 +1,5 @@ -//! Edge annotations — M2b indexed-arena integration tests. -//! -//! Pins the slice 5 validation matrix: -//! -//! - **Incremental seal then arena-backed query** — annotated insert, -//! trigger reindex with attachment-events provider attached, -//! verify `LedgerSnapshot.annotation_index.is_some()` and that -//! queries return the same results they did pre-arena. -//! - **Full rebuild fallback** — when no `Authoritative` events are -//! provided, the new root carries `annotation_index = None` and -//! queries fall back to the M2a scan path. Correctness is -//! identical to the arena path; only the read shape differs. -//! - **Post-defensive-drop sticky bit** — the -//! `(prev_arena=None, has_annotations=true)` cell stays in -//! scan-fallback under `Augment` coverage. We drive a ledger into -//! that state with a no-provider reindex, then verify the next -//! reindex with provider stays scan-only. -//! - **Storage inspection** — when an arena is sealed, the four -//! expected blobs (forward leaf + branch, reverse leaf + branch) -//! exist in CAS at the CIDs the index root advertises. -//! -//! All tests run against the file-backed (non-memory) path so -//! storage inspection has a real CAS to verify against. -//! -//! See `docs/design/edge-annotations.md` for the on-disk arena -//! layout this suite validates. +//! Edge annotations once their `rdf:reifies` links are indexed: hydration, +//! explain, graph transfers, writes and cascades against the indexed links. #![cfg(feature = "native")] @@ -56,7 +32,7 @@ fn annotated_insert() -> JsonValue { /// Subject-hydration query that exercises /// `HydrationFormatter::inject_annotations` — the only call site -/// that takes the M2b arena path. A flat `select` with +/// that reads the annotation's link. A flat `select` with /// `@annotation` in the `where` clause goes through query /// expansion and the **sync** JSON-LD formatter, which never /// touches `inject_annotations`. The hydration path fires only @@ -98,126 +74,14 @@ fn extract_role_from_hydration(rows: &JsonValue) -> Option { } #[tokio::test] -async fn incremental_arena_seal_then_arena_backed_query() { - // 1. Insert annotated edge. - // 2. Trigger background indexer with attachment-events provider - // attached → arena gets sealed. - // 3. After reindex: snapshot.annotation_index is Some, and the - // subject-hydration query returns the same row it did - // pre-reindex (when the M2a scan path was active). - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = "it/edge-annotations-indexed:incremental-seal"; - - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); - - local - .run_until(async move { - let ledger0 = genesis_ledger(&fluree, ledger_id); - let after_insert = fluree - .insert(ledger0, &annotated_insert()) - .await - .expect("annotated insert"); - - // Force the LedgerManager to load + cache the post-insert - // state via `ledger_cached`. This populates the manager's - // `entries` so the indexer's `AttachmentEventsProvider` - // finds the running ledger when it queries the overlay. - // (`fluree.ledger(...)` bypasses the cache and would - // leave entries empty, causing the provider to return - // None and the indexer to take the defensive-drop path.) - let pre_reindex_handle = fluree - .ledger_cached(ledger_id) - .await - .expect("cached load before reindex"); - let pre_reindex_loaded = pre_reindex_handle.snapshot().await.to_ledger_state(); - assert!( - pre_reindex_loaded - .novelty - .attachments - .iter_event_pairs() - .next() - .is_some(), - "running ledger must have attachments before reindex \ - (provider reads from this overlay)" - ); - // Pre-reindex hydration query — exercises the M2a scan - // path inside `HydrationFormatter::inject_annotations`. - let pre_rows = support::query_jsonld_formatted( - &fluree, - &pre_reindex_loaded, - &annotated_hydration_query(), - ) - .await - .expect("pre-reindex hydration query"); - assert_eq!( - extract_role_from_hydration(&pre_rows).as_deref(), - Some("Engineer"), - "pre-reindex hydration must surface the annotation body via scan path" - ); - - // Trigger reindex. - support::trigger_index_and_wait(&handle, ledger_id, after_insert.receipt.t).await; - support::wait_for_index_application(&fluree, ledger_id, after_insert.receipt.t).await; - - // Post-reindex: arena should be sealed. - let post = fluree - .ledger(ledger_id) - .await - .expect("reload after reindex"); - assert!( - post.snapshot.has_annotations, - "sticky bit must be set after annotated insert + reindex" - ); - assert!( - post.snapshot.annotation_index.is_some(), - "incremental indexer must seal an arena when provider \ - supplies attachment events" - ); - assert!( - post.snapshot.has_arena_reader(), - "snapshot must carry both annotation_index and \ - content_store after a successful reindex" - ); - - // Post-reindex: same hydration query, now via the arena - // path. `inject_annotations` constructs an - // `AnnotationArenaReader` once per response (see - // `HydrationFormatter::new`) and resolves the worksFor - // edge through it instead of issuing a POST scan. - let post_rows = - support::query_jsonld_formatted(&fluree, &post, &annotated_hydration_query()) - .await - .expect("post-reindex hydration query"); - assert_eq!( - extract_role_from_hydration(&post_rows).as_deref(), - Some("Engineer"), - "post-reindex hydration must surface the annotation body via arena path" - ); - assert_eq!( - pre_rows, post_rows, - "arena-backed hydration must produce identical output to scan-based" - ); - }) - .await; -} - -#[tokio::test] -async fn full_rebuild_without_authoritative_falls_back_to_scan() { - // No attachment-events provider → reindex lands the new root - // with `annotation_index = None`. The subject-hydration query - // must still surface the annotation body via the M2a indexed- - // scan-fallback path inside `inject_annotations`. +async fn hydration_reads_indexed_annotations() { + // Subject hydration surfaces an annotation whose link has rolled + // into the index. let fluree = FlureeBuilder::memory() .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build_memory(); let ledger_id = "it/edge-annotations-indexed:scan-fallback"; - // Bare config — NO attachment-events provider. This simulates - // a deployment that hasn't wired the provider yet, or a test - // harness that intentionally exercises the scan-fallback path. let (local, handle) = support::start_background_indexer_local( fluree.backend().clone(), fluree @@ -245,488 +109,20 @@ async fn full_rebuild_without_authoritative_falls_back_to_scan() { .await .expect("reload after reindex"); assert!(post.snapshot.has_annotations, "sticky bit set"); - assert!( - post.snapshot.annotation_index.is_none(), - "without an events provider the indexer must not seal \ - an arena (delta-unknown / Augment-without-base path)" - ); - assert!( - !post.snapshot.has_arena_reader(), - "scan-fallback: hydration goes through the M2a POST scan" - ); - - // Hydration query against the indexed scan-fallback - // snapshot. `inject_annotations` falls through to the - // M2a POST scan because the arena reader can't be - // constructed without `annotation_index`. The data - // lives in the indexed POST (the test reindexed - // above), so this exercises the indexed-scan-fallback - // path — not the novelty-only path. + assert_eq!(post.index_t(), post.t(), "the link is indexed"); let rows = support::query_jsonld_formatted(&fluree, &post, &annotated_hydration_query()) .await - .expect("hydration against scan-fallback snapshot"); + .expect("hydration against the indexed snapshot"); assert_eq!( extract_role_from_hydration(&rows).as_deref(), Some("Engineer"), - "indexed-scan-fallback hydration must surface the annotation body" + "hydration must surface the indexed annotation body" ); }) .await; } -#[tokio::test] -async fn post_defensive_drop_stays_in_scan_fallback() { - // Two-step setup driving a ledger into the - // (has_annotations=true, annotation_index=None) state on the - // base root, then verifying the next reindex with provider - // attached stays scan-only because Augment can't recover - // history without a base arena. - // - // Step A: annotated insert + reindex WITH provider → arena - // sealed. - // Step B: another commit + reindex WITHOUT provider → defensive - // drop; new root has annotation_index = None, - // has_annotations = true. - // Step C: another commit + reindex WITH provider returning - // Augment → arena stays None (the gate I just added). - // - // To run these steps we juggle two workers — one with provider, - // one without. The cleaner alternative would be a switchable - // provider; for the test we just trigger separate workers. - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = "it/edge-annotations-indexed:post-defensive-drop"; - - // Worker A: with provider. - let (local_a, handle_a) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); - - local_a - .run_until(async move { - let ledger0 = genesis_ledger(&fluree, ledger_id); - let after_a = fluree - .insert(ledger0, &annotated_insert()) - .await - .expect("step A insert"); - // Cache the ledger BEFORE trigger so the provider has - // events when the worker dispatches. - let _pre_a = fluree - .ledger_cached(ledger_id) - .await - .expect("pre-step-A cached load"); - support::trigger_index_and_wait(&handle_a, ledger_id, after_a.receipt.t).await; - support::wait_for_index_application(&fluree, ledger_id, after_a.receipt.t).await; - - let after_step_a_handle = fluree - .ledger_cached(ledger_id) - .await - .expect("cached load after step A reindex"); - let after_step_a = after_step_a_handle.snapshot().await.to_ledger_state(); - assert!( - after_step_a.snapshot.annotation_index.is_some(), - "step A: arena sealed via provider" - ); - - // Step B: another commit + reindex without provider. - // We swap to a second worker that has no provider — - // this simulates a deployment that lost provider - // wiring between passes. - let (local_b, handle_b) = support::start_background_indexer_local( - fluree.backend().clone(), - fluree - .nameservice_mode() - .publisher_arc() - .expect("test setup requires ReadWrite nameservice mode"), - IndexerConfig::small(), - ); - local_b - .run_until(async move { - let extra_b = json!({ - "@context": ctx(), - "@id": "ex:bob", - "ex:name": "Bob" - }); - let after_b = fluree - .insert(after_step_a, &extra_b) - .await - .expect("step B insert"); - support::trigger_index_and_wait(&handle_b, ledger_id, after_b.receipt.t).await; - support::wait_for_index_application(&fluree, ledger_id, after_b.receipt.t) - .await; - let after_step_b_handle = fluree - .ledger_cached(ledger_id) - .await - .expect("cached load after step B reindex"); - let after_step_b = after_step_b_handle.snapshot().await.to_ledger_state(); - assert!( - after_step_b.snapshot.has_annotations, - "sticky bit stays set" - ); - assert!( - after_step_b.snapshot.annotation_index.is_none(), - "step B: no provider → defensive drop on the new root" - ); - - // Step C: another commit + reindex with provider. - // Augment without a base arena BUT sticky=true - // → must NOT seal (the gate). - let (local_c, handle_c) = support::start_background_indexer_with_attachments( - &fluree, - IndexerConfig::small(), - ); - local_c - .run_until(async move { - let extra_c = json!({ - "@context": ctx(), - "@id": "ex:carol", - "ex:name": "Carol" - }); - let after_c = fluree - .insert(after_step_b, &extra_c) - .await - .expect("step C insert"); - support::trigger_index_and_wait( - &handle_c, - ledger_id, - after_c.receipt.t, - ) - .await; - support::wait_for_index_application( - &fluree, - ledger_id, - after_c.receipt.t, - ) - .await; - let after_step_c_handle = fluree - .ledger_cached(ledger_id) - .await - .expect("cached load after step C reindex"); - let after_step_c = - after_step_c_handle.snapshot().await.to_ledger_state(); - assert!(after_step_c.snapshot.has_annotations); - assert!( - after_step_c.snapshot.annotation_index.is_none(), - "step C: Augment + no prev arena + sticky=true \ - must stay scan-fallback (gate from prior commit)" - ); - - // Hydration still surfaces the annotation - // body via the M2a indexed-scan path. - let rows = support::query_jsonld_formatted( - &fluree, - &after_step_c, - &annotated_hydration_query(), - ) - .await - .expect("hydration against post-drop snapshot"); - assert_eq!( - extract_role_from_hydration(&rows).as_deref(), - Some("Engineer"), - "indexed-scan-fallback hydration must still \ - surface the annotation body after defensive drop" - ); - }) - .await; - }) - .await; - }) - .await; -} - -#[tokio::test] -async fn had_annotation_arena_sticky_survives_defensive_drop() { - // Pins the High-finding fix: the sticky - // `IndexRoot.had_annotation_arena` bit must be set the first - // time an arena is sealed and must persist across a defensive - // drop (when the indexer drops the arena because it received - // Unknown / None coverage). Without this persistence, the - // post-drop root looks identical to a fresh annotation-bearing - // import and the provider's bootstrap base-index scan-fallback - // could resurrect a live-only `Authoritative` arena, losing - // historical retract/reassert rows. - // - // Setup mirrors `post_defensive_drop_stays_in_scan_fallback`: - // step A seals an arena with provider attached; step B drops - // it by running the indexer without provider. The new - // assertion is on the sticky bit itself. - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = "it/edge-annotations-indexed:sticky-arena-bit"; - - let (local_a, handle_a) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); - - local_a - .run_until(async move { - let ledger0 = genesis_ledger(&fluree, ledger_id); - let after_a = fluree - .insert(ledger0, &annotated_insert()) - .await - .expect("step A insert"); - let _pre_a = fluree - .ledger_cached(ledger_id) - .await - .expect("pre-step-A cached load"); - support::trigger_index_and_wait(&handle_a, ledger_id, after_a.receipt.t).await; - support::wait_for_index_application(&fluree, ledger_id, after_a.receipt.t).await; - - let after_step_a = fluree - .ledger_cached(ledger_id) - .await - .expect("cached load after step A reindex") - .snapshot() - .await - .to_ledger_state(); - assert!( - after_step_a.snapshot.annotation_index.is_some(), - "step A: arena sealed via provider" - ); - assert!( - after_step_a.snapshot.had_annotation_arena, - "step A: sticky bit set on first seal" - ); - - let (local_b, handle_b) = support::start_background_indexer_local( - fluree.backend().clone(), - fluree - .nameservice_mode() - .publisher_arc() - .expect("test setup requires ReadWrite nameservice mode"), - IndexerConfig::small(), - ); - local_b - .run_until(async move { - let extra_b = json!({ - "@context": ctx(), - "@id": "ex:bob", - "ex:name": "Bob" - }); - let after_b = fluree - .insert(after_step_a, &extra_b) - .await - .expect("step B insert"); - support::trigger_index_and_wait(&handle_b, ledger_id, after_b.receipt.t).await; - support::wait_for_index_application(&fluree, ledger_id, after_b.receipt.t) - .await; - let after_step_b = fluree - .ledger_cached(ledger_id) - .await - .expect("cached load after step B reindex") - .snapshot() - .await - .to_ledger_state(); - assert!( - after_step_b.snapshot.annotation_index.is_none(), - "step B: no provider → defensive drop on the new root" - ); - assert!( - after_step_b.snapshot.has_annotations, - "step B: f:reifies* sticky bit persists" - ); - assert!( - after_step_b.snapshot.had_annotation_arena, - "step B: had_annotation_arena MUST persist through \ - defensive drop — without this the provider's \ - bootstrap scan-fallback can't distinguish defensive \ - drop from fresh import and would seal a live-only \ - arena, losing retract/reassert history" - ); - }) - .await; - }) - .await; -} - -#[tokio::test] -async fn indexer_pass_without_provider_marks_arena_history_owned() { - // A non-bulk-import ledger that accumulates annotation flakes - // through the normal commit pipeline and gets indexed *without* - // a provider must still mark the resulting root's sticky bit. - // - // Without this coercion, the root would land at - // (has_annotations=true, annotation_index=None, - // had_annotation_arena=false) — the same shape as a fresh - // bulk-import — and a later empty-overlay provider pass could - // misclassify it as bootstrap-eligible and re-seal a - // live-only `Authoritative` arena, silently dropping any - // retract/reassert events that lived in the running overlay - // before the no-provider reindex consumed it. - // - // The coercion lives in `IncrementalRootBuilder::build()` / - // `encode_and_write_root_v6`: any indexer-produced root with - // `has_annotations=true` gets the sticky bit set, even when no - // arena is sealed. Bulk import never goes through these paths, - // so its `had_annotation_arena=false` state stays the unique - // bootstrap-eligible shape. - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = "it/edge-annotations-indexed:indexer-owns-annotation-history"; - - // Worker without provider — defensive drop / no-op annotation - // arena handling. - let (local, handle) = support::start_background_indexer_local( - fluree.backend().clone(), - fluree - .nameservice_mode() - .publisher_arc() - .expect("test setup requires ReadWrite nameservice mode"), - IndexerConfig::small(), - ); - - local - .run_until(async move { - let ledger0 = genesis_ledger(&fluree, ledger_id); - let after = fluree - .insert(ledger0, &annotated_insert()) - .await - .expect("annotated insert"); - support::trigger_index_and_wait(&handle, ledger_id, after.receipt.t).await; - support::wait_for_index_application(&fluree, ledger_id, after.receipt.t).await; - - let post = fluree - .ledger_cached(ledger_id) - .await - .expect("cached load after no-provider reindex") - .snapshot() - .await - .to_ledger_state(); - assert!( - post.snapshot.has_annotations, - "f:reifies* predicates were observed → sticky bit is set" - ); - assert!( - post.snapshot.annotation_index.is_none(), - "no provider → no arena sealed" - ); - assert!( - post.snapshot.had_annotation_arena, - "indexer-produced root with has_annotations=true MUST mark \ - had_annotation_arena=true, even when no arena is sealed — \ - otherwise a later empty-overlay provider pass would treat \ - this state like fresh import and bootstrap from a \ - live-only base-index scan, losing history" - ); - }) - .await; -} - -#[tokio::test] -async fn storage_inspection_finds_arena_artifacts() { - // After a successful arena seal, the index root's - // forward_branch_cid and reverse_branch_cid must resolve to - // real bytes in CAS, and those branches must reference real - // leaf CIDs. - use fluree_db_binary_index::annotation_arena::{ - AnnotationForwardBranch, AnnotationReverseBranch, - }; - use fluree_db_core::storage::ContentStore; - - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = "it/edge-annotations-indexed:storage-inspection"; - - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); - - local - .run_until(async move { - let ledger0 = genesis_ledger(&fluree, ledger_id); - let after = fluree - .insert(ledger0, &annotated_insert()) - .await - .expect("annotated insert"); - // Pre-reindex cached load — populates LedgerManager so - // the provider can read attachments. - let _pre = fluree - .ledger_cached(ledger_id) - .await - .expect("pre-reindex cached load"); - support::trigger_index_and_wait(&handle, ledger_id, after.receipt.t).await; - - let post = fluree.ledger(ledger_id).await.expect("reload"); - let ann_root = post - .snapshot - .annotation_index - .as_ref() - .expect("arena sealed"); - let cs = post - .snapshot - .content_store - .as_ref() - .expect("content store handle"); - - // Forward branch resolves to bytes. - let fwd_branch_bytes = cs - .get(&ann_root.forward_branch_cid) - .await - .expect("forward branch bytes in CAS"); - let fwd_branch = AnnotationForwardBranch::decode(&fwd_branch_bytes) - .expect("forward branch decodes cleanly"); - assert!( - !fwd_branch.leaves.is_empty(), - "forward branch must reference at least one leaf" - ); - - // Each forward-leaf CID resolves to bytes in CAS. - for entry in &fwd_branch.leaves { - let leaf_bytes = cs - .get(&entry.leaf_cid) - .await - .expect("forward leaf bytes in CAS"); - assert!( - !leaf_bytes.is_empty(), - "forward leaf must be a non-empty blob" - ); - } - - // Reverse branch + leaves: same shape. - let rev_branch_bytes = cs - .get(&ann_root.reverse_branch_cid) - .await - .expect("reverse branch bytes in CAS"); - let rev_branch = AnnotationReverseBranch::decode(&rev_branch_bytes) - .expect("reverse branch decodes cleanly"); - assert!(!rev_branch.leaves.is_empty()); - for entry in &rev_branch.leaves { - let _leaf_bytes = cs.get(&entry.leaf_cid).await.expect("reverse leaf bytes"); - } - - // Stats line up with the inserted shape: one - // (edge, ann) pair → one distinct edge, one distinct - // annotation, one event row in each direction. - assert_eq!(ann_root.stats.forward_rows, 1); - assert_eq!(ann_root.stats.reverse_rows, 1); - assert_eq!(ann_root.stats.distinct_edges, 1); - assert_eq!(ann_root.stats.distinct_annotations, 1); - assert_eq!(ann_root.stats.live_attachment_pairs, 1); - - // Per-slot NDVs (M3.1 follow-up): the single edge has one - // subject (alice), one predicate (worksFor), one object - // (acme), no named graph, and no language tag. - assert_eq!(ann_root.stats.distinct_reified_subjects, 1); - assert_eq!(ann_root.stats.distinct_reified_predicates, 1); - assert_eq!(ann_root.stats.distinct_reified_objects, 1); - assert_eq!(ann_root.stats.reifies_graph_rows, 0); - assert_eq!(ann_root.stats.distinct_reified_graphs, 0); - assert_eq!(ann_root.stats.distinct_graph_anns, 0); - assert_eq!(ann_root.stats.reifies_lang_rows, 0); - assert_eq!(ann_root.stats.distinct_reified_langs, 0); - assert_eq!(ann_root.stats.distinct_lang_anns, 0); - assert_eq!(ann_root.stats.reifies_list_index_rows, 0); - assert_eq!(ann_root.stats.distinct_list_index_anns, 0); - // f:reifiesDatatype is intentionally NOT synthesized - // from the arena — see `AnnotationStats::reifies_datatype_rows`. - assert_eq!(ann_root.stats.reifies_datatype_rows, 0); - assert_eq!(ann_root.stats.distinct_reified_datatypes, 0); - }) - .await; -} - #[tokio::test] async fn non_annotation_ledger_skips_inject_annotations() { // Hydration on a ledger that has never seen an `f:reifies*` @@ -916,279 +312,9 @@ async fn explain_expands_annotations_as_the_executor_does() { .await; } -#[tokio::test] -async fn reindex_seals_arena_when_caller_supplies_only_a_provider() { - // Regression for the review finding: when the caller passes an - // `IndexerConfig` with `attachment_events_provider: Some(_)` but - // `attachment_events: None`, the api's reindex path must still - // resolve the provider into a concrete envelope — the direct - // rebuild path only consumes the concrete field, so skipping - // resolution would silently land the rebuild in scan-fallback. - use fluree_db_api::ReindexOptions; - - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = "it/edge-annotations-indexed:reindex-seals-arena-caller-provider"; - let ledger0 = genesis_ledger(&fluree, ledger_id); - - fluree - .insert(ledger0, &annotated_insert()) - .await - .expect("annotated insert"); - - // Caller hand-builds an `IndexerConfig` carrying only the - // attachment-events provider — no concrete events. The api's - // reindex path must call the provider during resolution. - let caller_provider = fluree - .attachment_events_provider() - .expect("api provider is available when ledger caching is enabled"); - let caller_config = fluree_db_indexer::IndexerConfig::default() - .with_attachment_events_provider(caller_provider); - - fluree - .reindex( - ledger_id, - ReindexOptions::default().with_indexer_config(caller_config), - ) - .await - .expect("reindex must succeed"); - - let post = fluree.ledger(ledger_id).await.expect("reload"); - assert!( - post.snapshot.annotation_index.is_some(), - "reindex must seal an annotation arena via the caller-supplied \ - AttachmentEventsProvider — got annotation_index=None which means \ - the provider-only IndexerConfig branch is back to scan-fallback" - ); -} - -#[tokio::test] -async fn reindex_seals_arena_when_caching_enabled_no_provider_in_opts() { - // Closes the bulk-import → arena seal gap: a `fluree.reindex(...)` - // call with default options should seal an authoritative arena - // when ledger caching is enabled, even though the caller didn't - // attach an `AttachmentEventsProvider` explicitly. The api's - // admin path attaches its own provider (backed by the running - // `LedgerManager`) and pre-loads the ledger so the provider can - // read its overlay. - // - // Without this wiring, a CLI-driven import + reindex flow would - // silently land in scan-fallback and require a *second* reindex - // through the api to seal the arena. - use fluree_db_api::ReindexOptions; - - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = "it/edge-annotations-indexed:reindex-seals-arena"; - let ledger0 = genesis_ledger(&fluree, ledger_id); - - fluree - .insert(ledger0, &annotated_insert()) - .await - .expect("annotated insert"); - - fluree - .reindex(ledger_id, ReindexOptions::default()) - .await - .expect("reindex must succeed"); - - let post = fluree.ledger(ledger_id).await.expect("reload"); - assert!( - post.snapshot.has_annotations, - "sticky bit set after annotated insert + reindex" - ); - assert!( - post.snapshot.annotation_index.is_some(), - "reindex must seal an annotation arena via the api-attached \ - AttachmentEventsProvider — got annotation_index=None which \ - means the bulk-import → arena gap is still open" - ); - let stats = &post - .snapshot - .annotation_index - .as_ref() - .expect("arena root") - .stats; - assert_eq!(stats.distinct_annotations, 1); - assert_eq!(stats.live_attachment_pairs, 1); -} - -#[tokio::test] -async fn reindex_seals_arena_without_ledger_caching() { - // The CLI builds its client `without_ledger_caching()`, so - // `Fluree::reindex` has no `LedgerManager` and therefore no - // attachment-events provider to ask. That is the configuration behind - // `fluree create --from `, whose one-shot seal pass ended - // with `annotation_index = None` while printing "Annotation arena - // sealed" — every quoted-triple query on the imported ledger then took - // the generic join chain. The reindex must derive coverage from the - // ledger state it loads anyway. - use fluree_db_api::ReindexOptions; - - let fluree = FlureeBuilder::memory() - .without_ledger_caching() - .build_memory(); - let ledger_id = "it/edge-annotations-indexed:reindex-seals-arena-no-cache"; - let ledger0 = genesis_ledger(&fluree, ledger_id); - - fluree - .insert(ledger0, &annotated_insert()) - .await - .expect("annotated insert"); - - fluree - .reindex(ledger_id, ReindexOptions::default()) - .await - .expect("reindex must succeed"); - - let post = fluree.ledger(ledger_id).await.expect("reload"); - assert!(post.snapshot.has_annotations, "sticky bit set"); - let stats = &post - .snapshot - .annotation_index - .as_ref() - .expect( - "reindex without a ledger manager must still seal the arena from the \ - ledger state's attachment events", - ) - .stats; - assert_eq!(stats.distinct_annotations, 1); - assert_eq!(stats.live_attachment_pairs, 1); -} - -#[tokio::test] -async fn wildcard_annotation_query_streams_from_the_arena() { - // StarBench P2 shape, `<< ?s ?p ?o >> :q ?x`, over a sealed arena: the - // annotation-first lane emits every live attachment straight from the - // forward arena — ref, plain-literal and language-tagged objects — with - // no base-edge scan. Rows must be exactly the annotated edges, with the - // object's datatype and language tag intact; the unannotated edge must - // not appear. - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = "it/edge-annotations-indexed:wildcard-enumeration"; - - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); - - local - .run_until(async move { - let ledger0 = genesis_ledger(&fluree, ledger_id); - let after_insert = fluree - .insert( - ledger0, - &json!({ - "@context": ctx(), - "@graph": [ - { - "@id": "ex:alice", - "ex:worksFor": { - "@id": "ex:acme", - "@annotation": {"ex:source": "hr"} - }, - "ex:score": { - "@value": 42, - "@annotation": {"ex:source": "exam"} - }, - "ex:motto": { - "@value": "carpe diem", - "@language": "la", - "@annotation": {"ex:source": "wall"} - } - }, - {"@id": "ex:bob", "ex:worksFor": {"@id": "ex:acme"}} - ] - }), - ) - .await - .expect("annotated insert"); - let _ = fluree - .ledger_cached(ledger_id) - .await - .expect("cached load before reindex"); - - support::trigger_index_and_wait(&handle, ledger_id, after_insert.receipt.t).await; - support::wait_for_index_application(&fluree, ledger_id, after_insert.receipt.t).await; - - let post = fluree - .ledger(ledger_id) - .await - .expect("reload after reindex"); - assert!( - post.snapshot.annotation_index.is_some(), - "arena must be sealed for this test to exercise the enumeration lane" - ); - - let sparql = r" - PREFIX ex: - SELECT ?s ?p ?o ?src WHERE { - << ?s ?p ?o >> ex:source ?src . - } - ORDER BY ?src - "; - let result = support::query_sparql(&fluree, &post, sparql) - .await - .expect("wildcard annotation query"); - let rows = result.to_sparql_json(&post.snapshot).expect("sparql json"); - let bindings = rows["results"]["bindings"] - .as_array() - .expect("bindings array") - .clone(); - let cell = |b: &JsonValue, v: &str| b[v]["value"].as_str().unwrap_or("").to_string(); - let got: Vec<(String, String, String, String)> = bindings - .iter() - .map(|b| (cell(b, "s"), cell(b, "p"), cell(b, "o"), cell(b, "src"))) - .collect(); - let ex = |l: &str| format!("http://example.org/{l}"); - assert_eq!( - got, - vec![ - ( - ex("alice"), - ex("score"), - "42".to_string(), - "exam".to_string() - ), - (ex("alice"), ex("worksFor"), ex("acme"), "hr".to_string()), - ( - ex("alice"), - ex("motto"), - "carpe diem".to_string(), - "wall".to_string() - ), - ], - "exactly the annotated edges, in ?src order: {bindings:#?}" - ); - let motto = bindings - .iter() - .find(|b| b["src"]["value"] == "wall") - .expect("language-tagged row"); - assert_eq!( - motto["o"]["xml:lang"], "la", - "language tag must survive the arena round-trip: {motto:#?}" - ); - let score = bindings - .iter() - .find(|b| b["src"]["value"] == "exam") - .expect("integer row"); - assert_eq!( - score["o"]["datatype"], "http://www.w3.org/2001/XMLSchema#integer", - "datatype must survive the arena round-trip: {score:#?}" - ); - }) - .await; -} - -/// #1467: a reification-aware COPY of a named-graph annotation must survive a -/// reindex. The attachment indexer decodes each bundle via -/// `EdgeKey::from_reifies_facts` at seal time -/// (`indexer_attachment_provider.rs`), so an incorrect re-home would drop the -/// annotation from the durable arena — not just from a live query. This drives -/// the full seal path and then decodes the re-homed bundle out of the indexed -/// snapshot (novelty-free), proving the anchor moved to the dest graph. +/// #1467: a COPY of a named-graph annotation must survive a reindex: the +/// copied link reads back out of the indexed snapshot in the destination +/// graph, and the source's stays. #[tokio::test] async fn transfer_named_to_named_survives_reindex() { let fluree = FlureeBuilder::memory() @@ -1258,37 +384,28 @@ async fn transfer_named_to_named_survives_reindex() { .ledger(ledger_id) .await .expect("reload after reindex"); - let ann_root = post - .snapshot - .annotation_index - .as_ref() - .expect("reindex with the attachment provider must seal an arena"); - - // Index-time correctness gate. The attachment indexer decodes each - // bundle via `EdgeKey::from_reifies_facts` at seal time; a bad - // re-home of the g2 copy (anchor OBJECT still pointing at g1 while - // the flake-level `g` is g2) would be a `GraphMismatch` and get - // dropped from the arena. So the annotation now lives in BOTH the - // original g1 and the copied g2 — two distinct reified graphs, two - // f:reifiesGraph anchor rows in the durable index. Without the fix - // only g1 would survive the seal. - assert_eq!( - ann_root.stats.distinct_reified_graphs, 2, - "both the source (g1) and re-homed (g2) graph anchors must seal" - ); - assert_eq!( - ann_root.stats.reifies_graph_rows, 2, - "two f:reifiesGraph anchor rows (one per graph) must survive the seal" - ); + assert_eq!(post.index_t(), post.t()); + for g in [g1, g2] { + let g_id = post + .snapshot + .graph_registry + .graph_id_for_iri(g) + .expect("registered graph"); + let anns = support::decode_annotations_for_subject( + &post, + g_id, + "http://example.org/alice", + ) + .await; + assert_eq!(anns.len(), 1, "graph {g} holds its own link: {anns:?}"); + } }) .await; } -/// default→named at the DURABLE-INDEX layer (#1483 review: the synthesized -/// anchor was only checked in-memory). `COPY DEFAULT TO ` SYNTHESIZES an -/// `f:reifiesGraph` anchor the source never had; the attachment indexer -/// decodes every bundle via `EdgeKey::from_reifies_facts` at seal time, so a -/// malformed synthesized anchor would be silently dropped from the arena. +/// default→named at the DURABLE-INDEX layer (#1483 review: the copy was only +/// checked in-memory). After `COPY DEFAULT TO ` and a reindex, the link +/// reads back out of both graphs. #[tokio::test] async fn transfer_default_to_named_synthesized_anchor_survives_reindex() { let fluree = FlureeBuilder::memory() @@ -1303,7 +420,7 @@ async fn transfer_default_to_named_synthesized_anchor_survives_reindex() { local .run_until(async move { let ledger0 = genesis_ledger(&fluree, ledger_id); - // Default-graph annotated edge — NO f:reifiesGraph anchor exists. + // Default-graph annotated edge. let seeded = fluree .insert( ledger0, @@ -1353,151 +470,30 @@ async fn transfer_default_to_named_synthesized_anchor_survives_reindex() { .ledger(ledger_id) .await .expect("reload after reindex"); - let ann_root = post + assert_eq!(post.index_t(), post.t()); + let g2_id = post .snapshot - .annotation_index - .as_ref() - .expect("reindex with the attachment provider must seal an arena"); - - // The default-graph bundle carries NO anchor; the g2 copy carries - // exactly the ONE synthesized anchor — and it must decode at seal - // time (a bad synthesis would be dropped as GraphMismatch). - assert_eq!( - ann_root.stats.distinct_reified_graphs, 1, - "exactly the synthesized g2 anchor's graph must seal" - ); - assert_eq!( - ann_root.stats.reifies_graph_rows, 1, - "exactly one synthesized f:reifiesGraph anchor row must survive the seal" - ); + .graph_registry + .graph_id_for_iri(g2) + .expect("registered graph"); + for g_id in [0, g2_id] { + let anns = support::decode_annotations_for_subject( + &post, + g_id, + "http://example.org/alice", + ) + .await; + assert_eq!(anns.len(), 1, "graph {g_id} holds its own link: {anns:?}"); + } }) .await; } -// ============================================================================ -// Insert-flow seal matrix — regression pins for the write-only ingest trap: -// a background index build racing an annotated insert used to run with -// "delta unknown" (ledger not resident in the manager), defensively drop -// the arena, and stamp the sticky bit; an explicit reindex then read a -// stale NsRecord and rebuilt under Augment with no base arena — leaving -// `annotation_index = None` permanently. Three fixes pinned here: -// transient provider load, Augment merge with the previous root's arena, -// and the post-quiesce record re-fetch in `Fluree::reindex`. -// ============================================================================ - -async fn assert_seals(fluree: &fluree_db_api::Fluree, ledger_id: &str, label: &str) { - fluree - .reindex(ledger_id, fluree_db_api::ReindexOptions::default()) - .await - .expect("reindex"); - let post = fluree.ledger(ledger_id).await.expect("reload"); - assert!( - post.snapshot.has_annotations, - "{label}: sticky bit after annotated insert" - ); - assert!( - post.snapshot.annotation_index.is_some(), - "{label}: first reindex must seal the arena" - ); -} - -#[tokio::test] -async fn seal_file_backed_small_insert() { - let tmp = tempfile::tempdir().expect("tempdir"); - let fluree: fluree_db_api::Fluree = - FlureeBuilder::file(tmp.path().to_string_lossy().to_string()) - .build() - .expect("fluree"); - let ledger_id = "bisect/file-small:main"; - let ledger0 = fluree.create_ledger(ledger_id).await.expect("create"); - fluree - .insert(ledger0, &annotated_insert()) - .await - .expect("insert"); - assert_seals(&fluree, ledger_id, "file+create_ledger+small").await; -} - -#[tokio::test] -async fn seal_memory_create_ledger_insert() { - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = "bisect/mem-create:main"; - let ledger0 = fluree.create_ledger(ledger_id).await.expect("create"); - fluree - .insert(ledger0, &annotated_insert()) - .await - .expect("insert"); - assert_seals(&fluree, ledger_id, "memory+create_ledger+small").await; -} - +/// An indexed annotation on a LITERAL edge, read with a variable object: the +/// link keys the literal by (value, datatype, tag) exactly as it was written, +/// so the bound value must match it. #[tokio::test] -async fn seal_memory_lpg_graph_insert() { - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = "bisect/mem-lpg:main"; - let ledger0 = genesis_ledger(&fluree, ledger_id); - fluree - .insert( - ledger0, - &json!({ - "@graph": [ - {"@id": "n0", "@type": "INDEXED", "value": "v0", - "order": {"@id": "n1", "@annotation": {"w": 1}}}, - {"@id": "n1", "@type": "INDEXED", "value": "v1", - "related": {"@id": "n0", "@annotation": {"w": 2}}} - ], - "opts": {"lpgEdgeLifecycle": true} - }), - ) - .await - .expect("insert"); - assert_seals(&fluree, ledger_id, "memory+genesis+lpg-graph").await; -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn seal_survives_background_index_race() { - // Probe-scale: hundreds of annotated edges in one transaction, enough - // novelty for the background indexer to race the explicit reindex. - let tmp = tempfile::tempdir().expect("tempdir"); - let fluree: fluree_db_api::Fluree = - FlureeBuilder::file(tmp.path().to_string_lossy().to_string()) - .build() - .expect("fluree"); - let ledger_id = "bisect/file-large:main"; - let ledger0 = fluree.create_ledger(ledger_id).await.expect("create"); - let nodes = 500usize; - let graph: Vec = (0..nodes) - .map(|i| { - json!({ - "@id": format!("http://kb.example/node/{i}"), - "@type": "INDEXED", - "value": format!("http://kb.example/entity/{i}"), - "order": { - "@id": format!("http://kb.example/node/{}", (i + 1) % nodes), - "@annotation": {"w": 1} - } - }) - }) - .collect(); - fluree - .insert( - ledger0, - &json!({"@graph": graph, "opts": {"lpgEdgeLifecycle": true}}), - ) - .await - .expect("insert"); - assert_seals(&fluree, ledger_id, "file+large-lpg-graph").await; -} - -/// Arena-backed probe over a LITERAL reified triple with a VAR object. -/// The recognized chain accepts a var object; the arena keys literal -/// edges by (value, dt, lang) exactly as they were written, so the probe -/// must match them — previously a runtime literal binding dropped the -/// row (returned nothing) instead of probing. -#[tokio::test] -async fn arena_probe_matches_literal_object_annotation() { +async fn indexed_literal_object_annotation_matches() { let fluree = FlureeBuilder::memory() .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build_memory(); @@ -1535,10 +531,7 @@ async fn arena_probe_matches_literal_object_annotation() { .ledger(ledger_id) .await .expect("reload after reindex"); - assert!( - post.snapshot.annotation_index.is_some(), - "arena must be sealed for this test to exercise the probe lane" - ); + assert_eq!(post.index_t(), post.t()); let sparql = r" PREFIX ex: @@ -1557,7 +550,7 @@ async fn arena_probe_matches_literal_object_annotation() { assert_eq!( bindings.len(), 1, - "the literal reified triple must match via the arena probe: {bindings:#?}" + "the literal reified triple must match its indexed link: {bindings:#?}" ); assert_eq!(bindings[0]["score"]["value"].as_str(), Some("42")); assert_eq!(bindings[0]["conf"]["value"].as_str(), Some("high")); @@ -1566,24 +559,12 @@ async fn arena_probe_matches_literal_object_annotation() { } // ============================================================================= -// Named-graph bundles after indexing: the flake-level `g` seam +// Named-graph links after indexing // ============================================================================= -/// A named-graph annotation must stay writable once its bundle has been -/// indexed. -/// -/// The stage-time single-target invariant compares a reifier's CURRENT bundle -/// against the one this transaction produces, keyed by (graph, s, p, o, dt). -/// Index-decoded flakes carry `g: None` — the graph is the index they came -/// from, not a field on the flake — while a named-graph transaction's flakes -/// carry `g: Some(sid)`. Without stamping the scanned side, the two never -/// match: the net bundle mixes `None` and `Some` and decodes as -/// `MixedFlakeGraphs`, so re-asserting an unchanged annotation, or adding a -/// property to an existing claim, fails with an invariant violation on a -/// perfectly legitimate write. -/// -/// Default-graph annotations are unaffected (both sides are `None`), which is -/// why this needs a named graph AND a published index to reproduce. +/// A named-graph annotation must stay writable once its link has been +/// indexed: re-asserting it, enriching its body, and pointing its reifier at +/// a second edge as well (a reifier may reify several triples). #[tokio::test] async fn indexed_named_graph_annotation_stays_writable() { let fluree = FlureeBuilder::memory().build_memory(); @@ -1610,10 +591,10 @@ async fn indexed_named_graph_annotation_stays_writable() { .expect("annotated named-graph insert"); assert!(committed.ledger.t() > 0); - // Move the bundle out of novelty and into the index. + // Move the link out of novelty and into the index. support::rebuild_and_publish_index(&fluree, ledger_id).await; - // Re-assert the identical annotation against the indexed bundle. + // Re-assert the identical annotation against the indexed link. let reloaded = fluree.ledger(ledger_id).await.expect("reload indexed"); fluree .insert( @@ -1633,10 +614,9 @@ async fn indexed_named_graph_annotation_stays_writable() { .await .expect("adding a property to an indexed named-graph claim must be accepted"); - // The invariant itself still holds: re-pointing that reifier at a - // different edge is refused. + // The same reifier on a second edge names both. let reloaded = fluree.ledger(ledger_id).await.expect("reload indexed"); - let err = fluree + let both = fluree .insert( reloaded, &json!({ @@ -1650,20 +630,26 @@ async fn indexed_named_graph_annotation_stays_writable() { }), ) .await - .expect_err("re-pointing an indexed reifier at a second edge must still be refused"); - assert!( - err.to_string().contains("claim1"), - "the refusal must name the reifier: {err}" - ); + .expect("a reifier on a second edge"); + let g_id = both + .ledger + .snapshot + .graph_registry + .graph_id_for_iri("http://example.org/claims-graph") + .expect("registered graph"); + let anns = + support::decode_annotations_for_subject(&both.ledger, g_id, "http://example.org/alice") + .await; + assert_eq!(anns.len(), 2, "claim1 names both edges: {anns:?}"); } -/// Count the reifiers' own `f:reifies*` system facts across every graph. +/// Count live `rdf:reifies` links across every graph. /// -/// The two cascade tests below need to tell an *orphaned* bundle from a -/// *retracted* one, and the annotation query surface cannot: that lane joins -/// the bundle against its base edge, so once the edge is gone it returns -/// nothing either way. Reading the system facts directly is the only view -/// that distinguishes them. +/// The two cascade tests below need to tell an *orphaned* link from a +/// *retracted* one, and the annotation query surface cannot: annotation +/// syntax reads the link only through its base edge's body join, so once +/// the edge is gone it returns nothing either way. Reading the links +/// directly is the view that distinguishes them. async fn live_reifies_flakes(fluree: &fluree_db_api::Fluree, ledger_id: &str) -> usize { let ledger = fluree.ledger(ledger_id).await.expect("reload"); let mut live = 0; @@ -1672,36 +658,21 @@ async fn live_reifies_flakes(fluree: &fluree_db_api::Fluree, ledger_id: &str) -> &ledger.snapshot, g_id, ledger.novelty.as_ref(), - fluree_db_core::comparator::IndexType::Spot, + fluree_db_core::comparator::IndexType::Psot, fluree_db_core::range::RangeTest::Eq, - fluree_db_core::range::RangeMatch::new(), + fluree_db_core::range::RangeMatch::predicate(fluree_db_core::rdf_reifies_sid().clone()), fluree_db_core::range::RangeOptions::new().with_to_t(ledger.t()), ) .await .unwrap_or_default(); - live += flakes - .iter() - .filter(|f| fluree_db_core::is_reserved_reifies_predicate(&f.p)) - .count(); + live += flakes.len(); } live } /// Deleting a base edge must retract the claim that reifies it, even once the -/// bundle has been indexed. -/// -/// `cascade_attachment_retracts` finds the annotation, scans its bundle back, -/// and decodes it with `EdgeKey::from_reifies_facts` to confirm it really -/// reifies the edge being deleted. That decode reconciles the bundle's -/// `f:reifiesGraph` value against the flake-level `g` — and index-decoded -/// flakes carry `g: None` while the value says `Some(graph)`, so an indexed -/// named-graph bundle decoded as `GraphMismatch`. The cascade's -/// `Err(_) => continue` then swallowed it and the claim outlived the edge it -/// describes, which is a wrong answer rather than untidy storage. -/// -/// Only this combination is affected. The default graph is fine (both sides -/// `None`), and a novelty-resident named-graph bundle is fine (both `Some`), -/// which is why every existing cascade test passed. +/// link has been indexed in a named graph — the cascade's link lookup and the +/// retract it writes are both scoped to the edge's graph. #[tokio::test] async fn deleting_an_indexed_named_graph_edge_cascades_to_its_claim() { let fluree = FlureeBuilder::memory().build_memory(); @@ -1726,12 +697,12 @@ async fn deleting_an_indexed_named_graph_edge_cascades_to_its_claim() { support::rebuild_and_publish_index(&fluree, ledger_id).await; - // Guards the counter itself: if this read cannot see the bundle at all, + // Guards the counter itself: if this read cannot see the link at all, // the post-delete assertion below would pass for the wrong reason. let before = live_reifies_flakes(&fluree, ledger_id).await; assert!( before > 0, - "the indexed bundle must be visible to this read before the delete" + "the indexed link must be visible to this read before the delete" ); let deleted = fluree @@ -1750,20 +721,17 @@ async fn deleting_an_indexed_named_graph_edge_cascades_to_its_claim() { assert_eq!( live_reifies_flakes(&fluree, ledger_id).await, 0, - "the claim's f:reifies* bundle must not outlive the edge it reifies" + "the claim's link must not outlive the edge it reifies" ); } -/// The same `g`-asymmetry defect, reached through the *other* cascade pass. +/// The *other* cascade pass against an indexed named-graph link. /// /// Pass 1 fires when the base edge is deleted. Pass 2 fires when the user /// deletes an annotation's last piece of metadata without touching the edge, -/// which leaves the `f:reifies*` bundle behind with nothing to describe. Both -/// passes scan the reifier's own facts and decode them, so both hit the -/// `GraphMismatch` on an indexed named-graph bundle — a fix to one leaves the -/// other silently broken, which is why this test exists alongside its twin. +/// which would leave the link behind with nothing to describe. #[tokio::test] -async fn deleting_an_indexed_named_graph_claim_body_cascades_its_bundle() { +async fn deleting_an_indexed_named_graph_claim_body_cascades_its_link() { let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "it/edge-annotations-indexed:cascade-named-graph-orphan"; @@ -1792,7 +760,7 @@ async fn deleting_an_indexed_named_graph_claim_body_cascades_its_bundle() { assert!( live_reifies_flakes(&fluree, ledger_id).await > 0, - "the indexed bundle must be visible to this read before the delete" + "the indexed link must be visible to this read before the delete" ); // Delete the claim's only metadata fact, leaving the edge itself alone. @@ -1812,61 +780,6 @@ async fn deleting_an_indexed_named_graph_claim_body_cascades_its_bundle() { assert_eq!( live_reifies_flakes(&fluree, ledger_id).await, 0, - "a bundle whose claim has no body left must not survive as an orphan" - ); -} - -/// A reifier already in the base index is still checked against its stored -/// bundle. -/// -/// `enforce_single_target_reifiers` skips its scan for reifiers that resolve -/// in neither the persisted dictionary nor novelty, because those provably -/// have no prior bundle and the scan for them degrades into a walk of the -/// graph's entire novelty. The skip must not extend to a reifier that *does* -/// resolve: this one is in the base index after the rebuild, so the second -/// write has to see its stored bundle to notice that it would then reify two -/// edges at once. -#[tokio::test] -async fn an_indexed_reifier_pointed_at_a_second_edge_is_still_refused() { - let fluree = FlureeBuilder::memory().build_memory(); - let ledger_id = "it/edge-annotations-indexed:second-edge-refused"; - - let committed = fluree - .insert( - genesis_ledger(&fluree, ledger_id), - &json!({ - "@context": ctx(), - "@id": "ex:alice", - "ex:knows": { - "@id": "ex:bob", - "@annotation": {"@id": "ex:claim1", "ex:confidence": 0.9} - } - }), - ) - .await - .expect("annotated insert"); - - support::rebuild_and_publish_index(&fluree, ledger_id).await; - let indexed = fluree.ledger(ledger_id).await.expect("reload indexed"); - assert!(indexed.t() >= committed.ledger.t()); - - // Same reifier, a different edge, with the first attachment left in place. - let err = fluree - .insert( - indexed, - &json!({ - "@context": ctx(), - "@id": "ex:carol", - "ex:knows": { - "@id": "ex:dave", - "@annotation": {"@id": "ex:claim1", "ex:confidence": 0.5} - } - }), - ) - .await - .expect_err("a reifier on two edges must be refused"); - assert!( - err.to_string().contains("claim1"), - "the error must name the reifier: {err}" + "a link whose claim has no body left must not survive as an orphan" ); } diff --git a/fluree-db-api/tests/it_import.rs b/fluree-db-api/tests/it_import.rs index b95e5136ee..e0f9e637b2 100644 --- a/fluree-db-api/tests/it_import.rs +++ b/fluree-db-api/tests/it_import.rs @@ -1245,41 +1245,32 @@ async fn import_jsonld_user_authored_expanded_reifies_iri_is_rejected() { ); } -/// End-to-end: imported annotation-bearing ledger → follow-up -/// `fluree.reindex(...)` (the same call the CLI's -/// `fluree create --import` auto-seal step makes) seals the -/// annotation arena. -/// -/// Closes the bulk-import seal gap. `ApiAttachmentEventsProvider` -/// scans the base index for `f:reifies*` flakes when the running -/// `AttachmentNovelty` overlay is empty but the snapshot's sticky -/// bit says annotations exist — so the freshly-imported state -/// (where the f:reifies* flakes live in the base index, not the -/// overlay) still produces a complete `Authoritative` event set -/// for the indexer's arena builder. +/// An export written before links carries each annotation as `f:reifies*` +/// triples; importing it stores the annotation's `rdf:reifies` link and not +/// the slots. #[tokio::test] -async fn import_then_reindex_seals_annotation_arena() { +async fn import_of_reifies_slots_writes_the_link() { let db_dir = tempfile::tempdir().expect("db tmpdir"); let data_dir = tempfile::tempdir().expect("data tmpdir"); - let ttl = r" + let ttl = r#" @prefix ex: . @prefix f: . ex:alice ex:worksFor ex:acme . -_:ann1 f:reifiesSubject ex:alice ; - f:reifiesPredicate ex:worksFor ; - f:reifiesObject ex:acme . -"; +ex:ann1 f:reifiesSubject ex:alice ; + f:reifiesPredicate ex:worksFor ; + f:reifiesObject ex:acme ; + ex:role "Engineer" . +"#; let ttl_path = write_ttl(data_dir.path(), "annotated.ttl", ttl); let fluree = FlureeBuilder::file(db_dir.path().to_string_lossy().to_string()) - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build() .expect("build file-backed Fluree"); - let ledger_id = "test/import-then-reindex:main"; + let ledger_id = "test/import-reifies-slots:main"; let result = fluree .create(ledger_id) .import(&ttl_path) @@ -1287,26 +1278,47 @@ _:ann1 f:reifiesSubject ex:alice ; .execute() .await .expect("import should succeed"); - assert!(result.has_annotations, "annotation import must signal"); - // Auto-seal step: same call the CLI's run_bulk_import makes when - // `result.has_annotations` is true. - fluree - .reindex(ledger_id, fluree_db_api::ReindexOptions::default()) - .await - .expect("reindex must succeed"); + let ledger = fluree.ledger(ledger_id).await.expect("reload"); + let sparql = "PREFIX ex: \n\ + PREFIX rdf: \n\ + SELECT ?r ?o ?role WHERE { ?r rdf:reifies <<( ex:alice ex:worksFor ?o )>> ; ex:role ?role }"; + let json = support::query_sparql(&fluree, &ledger, sparql) + .await + .expect("link query") + .to_sparql_json(&ledger.snapshot) + .expect("sparql json"); + let bindings = json["results"]["bindings"].as_array().expect("bindings"); + assert_eq!(bindings.len(), 1, "{json:#}"); + assert_eq!(bindings[0]["r"]["value"], "http://example.org/ann1"); + assert_eq!(bindings[0]["o"]["value"], "http://example.org/acme"); + assert_eq!(bindings[0]["role"]["value"], "Engineer"); - let post = fluree.ledger(ledger_id).await.expect("reload"); - assert!( - post.snapshot.has_annotations, - "sticky bit must survive the seal pass" - ); + let slots = fluree_db_core::range_with_overlay( + &ledger.snapshot, + 0, + ledger.novelty.as_ref(), + fluree_db_core::comparator::IndexType::Spot, + fluree_db_core::range::RangeTest::Eq, + fluree_db_core::range::RangeMatch::subject( + ledger + .snapshot + .encode_iri("http://example.org/ann1") + .expect("ann1"), + ), + fluree_db_core::range::RangeOptions::new().with_to_t(ledger.t()), + ) + .await + .expect("scan ann1"); assert!( - post.snapshot.annotation_index.is_some(), - "annotation arena must be sealed after the auto-seal reindex" + !slots + .iter() + .any(|f| fluree_db_core::is_reserved_reifies_predicate(&f.p)), + "the slots are not stored: {slots:#?}" ); } + // N-Triples (.nt) import tests // // N-Triples is a strict subset of Turtle, so `.nt` files dispatch to the same diff --git a/fluree-db-api/tests/it_import_turtle_star.rs b/fluree-db-api/tests/it_import_turtle_star.rs index 3b78d42a51..87bafff899 100644 --- a/fluree-db-api/tests/it_import_turtle_star.rs +++ b/fluree-db-api/tests/it_import_turtle_star.rs @@ -216,8 +216,8 @@ async fn imported_trig_with_a_version_directive_keeps_its_prefixes() { ); } -/// The imported `f:reifies*` flakes under `graph`, across every reifier. -async fn reifies_flakes_in( +/// The imported `rdf:reifies` links under `graph`, across every reifier. +async fn links_in( fluree: &fluree_db_api::Fluree, alias: &str, graph: &str, @@ -232,24 +232,19 @@ async fn reifies_flakes_in( &ledger.snapshot, g_id, ledger.novelty.as_ref(), - fluree_db_core::comparator::IndexType::Spot, + fluree_db_core::comparator::IndexType::Psot, fluree_db_core::range::RangeTest::Eq, - fluree_db_core::range::RangeMatch::new(), + fluree_db_core::range::RangeMatch::predicate(fluree_db_core::rdf_reifies_sid().clone()), fluree_db_core::range::RangeOptions::new().with_to_t(ledger.t()), ) .await .expect("scan named graph") - .into_iter() - .filter(|f| fluree_db_core::is_reserved_reifies_predicate(&f.p)) - .collect() } -/// TriG import writes the bundle into the named graph whether or not it -/// carries `f:reifiesGraph`, so a graph-scoped annotation query cannot tell -/// the two apart. The edge identity can: the bundle must decode to the -/// block's graph, and deleting that edge must cascade to it. +/// TriG import writes a GRAPH block's link into that graph, naming the +/// block's edge, and deleting that edge must cascade to it. #[tokio::test] -async fn imported_trig_star_bundle_carries_its_graph_and_cascades() { +async fn imported_trig_star_link_lands_in_its_graph_and_cascades() { let alias = "it/import-trig-star:graph-anchored"; let trig = format!( "@prefix ex: .\n\ @@ -257,19 +252,15 @@ async fn imported_trig_star_bundle_carries_its_graph_and_cascades() { ); let (fluree, ledger) = import_dir(&[("claims.trig", &trig)], alias).await; - let graph_sid = ledger.snapshot.encode_iri(CLAIMS_GRAPH).expect("graph sid"); - let mut bundle = reifies_flakes_in(&fluree, alias, CLAIMS_GRAPH).await; - // Index-decoded flakes carry `g: None`; stamp the scanned graph the way - // the cascade in `stage()` does before decoding. - for f in &mut bundle { - f.g = Some(graph_sid.clone()); - } - let key = fluree_db_core::edge::EdgeKey::from_reifies_facts(&bundle) - .unwrap_or_else(|e| panic!("imported bundle must decode: {e:?}; {bundle:#?}")); - assert_eq!( - key.g, - Some(graph_sid), - "imported bundle must be anchored to its GRAPH block" + let alice = ledger + .snapshot + .encode_iri("http://example.org/alice") + .expect("alice sid"); + let links = links_in(&fluree, alias, CLAIMS_GRAPH).await; + assert_eq!(links.len(), 1, "{links:#?}"); + assert!( + matches!(&links[0].o, fluree_db_core::FlakeValue::TripleTerm(term) if term.s == alice), + "the link names the block's edge: {links:#?}" ); fluree @@ -283,10 +274,10 @@ async fn imported_trig_star_bundle_carries_its_graph_and_cascades() { .await .expect("delete the imported base edge"); - let remaining = reifies_flakes_in(&fluree, alias, CLAIMS_GRAPH).await; + let remaining = links_in(&fluree, alias, CLAIMS_GRAPH).await; assert!( remaining.is_empty(), - "the claim's bundle must not outlive the edge it reifies: {remaining:#?}" + "the claim's link must not outlive the edge it reifies: {remaining:#?}" ); } diff --git a/fluree-db-api/tests/it_query_sparql_annotations.rs b/fluree-db-api/tests/it_query_sparql_annotations.rs index 07e8395144..18fdf36581 100644 --- a/fluree-db-api/tests/it_query_sparql_annotations.rs +++ b/fluree-db-api/tests/it_query_sparql_annotations.rs @@ -254,13 +254,37 @@ async fn sparql_insert_data_with_named_blank_reifier_round_trips() { assert_eq!(bindings[0]["since"]["value"].as_str(), Some("2024")); } +/// The triples `ex:ann1` reifies, as `subject predicate object` local names. +async fn ann1_triples(fluree: &MemoryFluree, ledger: &MemoryLedger) -> Vec { + let query = r" + PREFIX ex: + PREFIX rdf: + SELECT ?s ?p ?o WHERE { ex:ann1 rdf:reifies <<( ?s ?p ?o )>> } ORDER BY ?s ?p ?o + "; + let json = support::query_sparql(fluree, ledger, query) + .await + .expect("query") + .to_sparql_json(&ledger.snapshot) + .expect("sparql json"); + let local = |v: &serde_json::Value| { + v["value"] + .as_str() + .and_then(|iri| iri.rsplit('/').next()) + .unwrap_or_default() + .to_string() + }; + json["results"]["bindings"] + .as_array() + .expect("bindings") + .iter() + .map(|b| format!("{} {} {}", local(&b["s"]), local(&b["p"]), local(&b["o"]))) + .collect() +} + #[tokio::test] -async fn sparql_same_id_reifying_two_edges_in_one_txn_is_rejected() { - // Single-txn multi-target: one explicit `@id` reifying two edges - // that share a subject. The `f:reifiesSubject` slot dedupes (same - // subject) so a subject-flake *count* sees one — but the predicate - // and object slots diverge, so the net bundle is multi-target. The - // net-bundle decode catches it; a plain count would not. +async fn sparql_same_id_reifying_two_edges_in_one_txn_names_both() { + // A reifier may reify several triples (RDF 1.2): one explicit id on two + // edges in one transaction links it to both. let fluree = FlureeBuilder::memory().build_memory(); let ledger0 = genesis_ledger(&fluree, "it/sparql-ann-update/multi-target-one-txn"); let update = r#" @@ -271,25 +295,23 @@ async fn sparql_same_id_reifying_two_edges_in_one_txn_is_rejected() { } "#; let txn = lower_update(&ledger0, update); - let err = fluree + let ledger = fluree .stage_owned(ledger0) .txn(txn) .execute() .await - .expect_err("one annotation id reifying two edges in one txn must be rejected"); - let msg = format!("{err:?} {err}"); - assert!( - msg.contains("multi-target") || msg.contains("reify exactly one edge"), - "expected multi-target rejection, got: {msg}" + .expect("one annotation id on two edges") + .ledger; + assert_eq!( + ann1_triples(&fluree, &ledger).await, + ["alice knows bob", "alice worksFor acme"] ); } #[tokio::test] -async fn sparql_reattaching_id_to_different_edge_across_txns_is_rejected() { - // Cross-txn re-point with no retract: the prior attachment lives in - // snapshot/novelty, not in this txn's flake set, so a count over the - // current txn alone sees a single subject assert and passes. The - // net-bundle check folds the prior state in and rejects. +async fn sparql_attaching_id_to_a_second_edge_across_txns_names_both() { + // Attaching an annotated id to a second edge in a later transaction, with + // no retract, adds a link; the first one stays. let fluree = FlureeBuilder::memory().build_memory(); let ledger0 = genesis_ledger(&fluree, "it/sparql-ann-update/repoint-across-txn"); @@ -306,23 +328,21 @@ async fn sparql_reattaching_id_to_different_edge_across_txns_is_rejected() { .expect("first attach") .ledger; - let repoint = r#" + let second = r#" PREFIX ex: INSERT DATA { ex:carol ex:worksFor ex:dave ~ ex:ann1 {| ex:role "Manager" |} . } "#; - let t2 = lower_update(&ledger1, repoint); - let err = fluree + let t2 = lower_update(&ledger1, second); + let ledger2 = fluree .stage_owned(ledger1) .txn(t2) .execute() .await - .expect_err( - "re-pointing an annotation id to a different edge across txns must be rejected", - ); - let msg = format!("{err:?} {err}"); - assert!( - msg.contains("multi-target") || msg.contains("reify exactly one edge"), - "expected multi-target rejection, got: {msg}" + .expect("second attach") + .ledger; + assert_eq!( + ann1_triples(&fluree, &ledger2).await, + ["alice worksFor acme", "carol worksFor dave"] ); } diff --git a/fluree-db-api/tests/it_tracing_spans.rs b/fluree-db-api/tests/it_tracing_spans.rs index 1d04a37c63..a4b6c8de12 100644 --- a/fluree-db-api/tests/it_tracing_spans.rs +++ b/fluree-db-api/tests/it_tracing_spans.rs @@ -846,15 +846,13 @@ async fn all_spans_properly_closed() { } // ============================================================================= -// Annotation read-path spans — inject_annotations + annotation_arena_lookup +// Annotation read-path span — inject_annotations // ============================================================================= #[tokio::test(flavor = "current_thread")] async fn annotation_hydration_emits_inject_annotations_span() { // A subject-hydration query against an annotated edge should emit - // an `inject_annotations` span tagged with the chosen path - // (`scan` here — no arena reader on a memory ledger that hasn't - // been reindexed). On non-annotation ledgers the formatter's + // an `inject_annotations` span. On non-annotation ledgers the formatter's // zero-cost gate skips the span entirely; that contract is // covered by `ac5_zero_noise_at_info` above (any span at all // would fail it on a non-annotation workload — the API layer is @@ -902,26 +900,11 @@ async fn annotation_hydration_emits_inject_annotations_span() { tracing::Level::DEBUG, "inject_annotations must be DEBUG (per CLAUDE.md tracing convention)" ); - let path = inject - .fields - .get("path") - .map(String::as_str) - .unwrap_or(""); - assert_eq!( - path, "scan", - "memory ledger without a sealed arena takes the scan path; got {path:?}" - ); assert!( inject.fields.contains_key("annotation_count"), "inject_annotations must record annotation_count: fields = {:?}", inject.fields ); - - // Without an arena reader the inner span should not fire. - assert!( - !store.has_span("annotation_arena_lookup"), - "annotation_arena_lookup must not fire when no arena is sealed" - ); } #[tokio::test(flavor = "current_thread")] diff --git a/fluree-db-api/tests/it_turtle_star_write_paths.rs b/fluree-db-api/tests/it_turtle_star_write_paths.rs index f9213b715f..8bb959e28d 100644 --- a/fluree-db-api/tests/it_turtle_star_write_paths.rs +++ b/fluree-db-api/tests/it_turtle_star_write_paths.rs @@ -355,36 +355,48 @@ async fn annotated_type_edge_is_accepted_by_insert_and_refused_by_upsert() { } #[tokio::test] -async fn one_named_reifier_on_two_edges_is_rejected_on_every_turtle_path() { - // A reifier denotes exactly one edge. Reusing an explicit reifier on - // two different triples in one document would store a bundle that - // `EdgeKey::from_reifies_facts` rejects, so every reader silently drops - // BOTH annotations. Fail loud at write time instead. +async fn one_named_reifier_on_two_edges_names_both_on_every_turtle_path() { + // A reifier may reify several triples (RDF 1.2): reusing an explicit + // reifier on two triples links it to both, on the direct Turtle path and + // on the JSON-LD-converted one alike. let turtle = with_prefixes( "ex:alice ex:knows ex:bob ~ ex:claim1 .\n\ ex:alice ex:knows ex:carol ~ ex:claim1 .\n", ); + let query = "PREFIX ex: \n\ + PREFIX rdf: \n\ + SELECT ?o WHERE { ex:claim1 rdf:reifies <<( ex:alice ex:knows ?o )>> } \ + ORDER BY ?o"; let fluree = FlureeBuilder::memory().build_memory(); - let err = fluree - .insert_turtle( - genesis_ledger(&fluree, "it/turtle-star-reuse:insert"), - &turtle, - ) - .await - .expect_err("insert_turtle must reject a reifier reused on two edges"); - let msg = err.to_string(); - assert!(msg.contains("claim1"), "must name the reifier: {msg}"); - - let err = fluree - .upsert_turtle( - genesis_ledger(&fluree, "it/turtle-star-reuse:upsert"), - &turtle, - ) - .await - .expect_err("upsert_turtle must reject a reifier reused on two edges"); - let msg = err.to_string(); - assert!(msg.contains("claim1"), "must name the reifier: {msg}"); + for (path, ledger_id) in [ + ("insert", "it/turtle-star-reuse:insert"), + ("upsert", "it/turtle-star-reuse:upsert"), + ] { + let ledger = genesis_ledger(&fluree, ledger_id); + let committed = match path { + "insert" => fluree.insert_turtle(ledger, &turtle).await, + _ => fluree.upsert_turtle(ledger, &turtle).await, + } + .unwrap_or_else(|e| panic!("{path}_turtle: {e}")); + let result = support::query_sparql(&fluree, &committed.ledger, query) + .await + .expect("query"); + let json = result + .to_sparql_json(&committed.ledger.snapshot) + .expect("sparql json"); + let objects: Vec<&str> = json["results"]["bindings"] + .as_array() + .expect("bindings") + .iter() + .filter_map(|b| b["o"]["value"].as_str()) + .collect(); + assert_eq!( + objects, + ["http://example.org/bob", "http://example.org/carol"], + "{path}_turtle: claim1 names both edges" + ); + } } #[tokio::test] diff --git a/fluree-db-api/tests/support/mod.rs b/fluree-db-api/tests/support/mod.rs index 323406fb2b..859d234f3e 100644 --- a/fluree-db-api/tests/support/mod.rs +++ b/fluree-db-api/tests/support/mod.rs @@ -145,19 +145,9 @@ pub async fn query_jsonld_tracked( db.query(fluree).jsonld(query_json).execute_tracked().await } -/// Decode every edge-annotation bundle whose reified SUBJECT is `subject_iri` -/// within graph `g_id` of `ledger`, via `EdgeKey::from_reifies_facts` — the -/// exact path both readers (JSON-LD hydration + attachment indexer) take. -/// -/// Panics on a decode error (`GraphMismatch` / `Duplicate` — the silent-drop -/// failure mode of a bad graph re-home), so a returned key is proof the bundle -/// is self-consistent. Locating the reifier by `f:reifiesSubject` (not by -/// `@id`) works for both anonymous and explicit reifiers. Returned keys are -/// ordered by reifier Sid. -/// -/// Uses point (`Eq`) POST + SPOT lookups rather than an unbounded scan so it -/// works on both the novelty path and the V3 binary-index provider (which -/// rejects `RangeTest::Ge` full scans). +/// The edges reified in graph `g_id` of `ledger` whose subject is +/// `subject_iri`, one per live `rdf:reifies` link, ordered by reifier Sid. +/// Each key's `g` is the graph's Sid (`None` for the default graph). pub async fn decode_annotations_for_subject( ledger: &LedgerState, g_id: fluree_db_core::GraphId, @@ -172,75 +162,44 @@ pub async fn decode_annotations_for_subject( .snapshot .encode_iri(subject_iri) .expect("encode subject IRI"); - let reifies_subject_pid = fluree_db_core::namespaces::reifies_subject_sid().clone(); + let g = (g_id != 0).then(|| { + let iri = ledger + .snapshot + .graph_registry + .iri_for_graph_id(g_id) + .expect("registered graph"); + ledger.snapshot.encode_iri(iri).expect("encode graph IRI") + }); - // Reifier subjects: POST lookup of f:reifiesSubject → subject, in g_id. - let pointers = range_with_overlay( + let mut links = range_with_overlay( &ledger.snapshot, g_id, ledger.novelty.as_ref(), - IndexType::Post, + IndexType::Psot, RangeTest::Eq, - RangeMatch::predicate_object(reifies_subject_pid, FlakeValue::Ref(subject_sid)), + RangeMatch::new().with_predicate(fluree_db_core::rdf_reifies_sid().clone()), RangeOptions::new().with_to_t(ledger.t()), ) .await - .expect("scan f:reifiesSubject pointers"); + .expect("scan rdf:reifies links"); + links.sort_by(|a, b| a.s.cmp(&b.s)); - let mut reifiers: Vec = pointers - .iter() + links + .into_iter() .filter(|f| f.op) - .map(|f| f.s.clone()) - .collect(); - reifiers.sort(); - reifiers.dedup(); - - let mut keys = Vec::with_capacity(reifiers.len()); - for ann_sid in reifiers { - let subject_flakes = range_with_overlay( - &ledger.snapshot, - g_id, - ledger.novelty.as_ref(), - IndexType::Spot, - RangeTest::Eq, - RangeMatch::subject(ann_sid.clone()), - RangeOptions::new().with_to_t(ledger.t()), - ) - .await - .expect("scan annotation subject flakes"); - // Index-decoded flakes carry `g: None` — the graph is the index they - // came from, not a field on the flake — while `f:reifiesGraph` names - // the graph. `from_reifies_facts` reconciles the two, so without this - // stamp an indexed named-graph bundle decodes as `GraphMismatch` and - // this helper reports a defect that isn't there. Production scans of a - // reifier's own facts stamp for the same reason (`stamp_graph` in - // `fluree-db-transact`). - let g_sid = subject_flakes.iter().find_map(|f| { - (f.op && f.p.name.as_ref() == fluree_vocab::db::REIFIES_GRAPH) - .then(|| match &f.o { - FlakeValue::Ref(sid) => Some(sid.clone()), - _ => None, - }) - .flatten() - }); - let bundle: Vec<_> = subject_flakes - .iter() - .filter(|f| f.op && fluree_db_core::is_reserved_reifies_predicate(&f.p)) - .cloned() - .map(|mut f| { - f.g = g_sid.clone(); - f - }) - .collect(); - let key = EdgeKey::from_reifies_facts(&bundle).unwrap_or_else(|e| { - panic!( - "from_reifies_facts failed for reifier {ann_sid} in g_id {g_id}: \ - {e:?}; bundle: {bundle:#?}" - ) - }); - keys.push(key); - } - keys + .filter_map(|f| match f.o { + FlakeValue::TripleTerm(term) if term.s == subject_sid => Some(EdgeKey { + g: g.clone(), + s: term.s, + p: term.p, + o: term.o, + dt: term.dt, + lang: term.lang, + list_i: None, + }), + _ => None, + }) + .collect() } /// Create a genesis ledger state for the given ledger ID. diff --git a/fluree-db-cli/src/cli.rs b/fluree-db-cli/src/cli.rs index a5118ca22c..d28b034dfa 100644 --- a/fluree-db-cli/src/cli.rs +++ b/fluree-db-cli/src/cli.rs @@ -1084,12 +1084,9 @@ pub enum Commands { #[arg(long)] graph: Option, - /// Emit edge annotations as raw `f:reifies*` system triples instead of - /// RDF 1.2 annotation syntax — the output of every release before 4.2. - /// - /// For consumers pinned to those bytes. Fluree's own write surfaces - /// reject hand-written `f:reifies*` triples, so this output only - /// re-imports through `fluree create --from`. + /// Write each edge annotation as its stored link, + /// `r rdf:reifies <<( s p o )>>`, instead of RDF 1.2 annotation syntax + /// on the base edge. JSON-LD keeps `@annotation`. #[arg(long)] raw_reifies: bool, diff --git a/fluree-db-cli/src/commands/export.rs b/fluree-db-cli/src/commands/export.rs index b38578a0cc..99bc510d16 100644 --- a/fluree-db-cli/src/commands/export.rs +++ b/fluree-db-cli/src/commands/export.rs @@ -546,7 +546,7 @@ fn report_rdf_stats(alias: &str, format: ExportFormat, stats: &ExportStats, grap if stats.annotations_unresolved > 0 { eprintln!( " {} {} edge annotations could not be resolved and are NOT in the output; \ - re-run with --raw-reifies to emit them as f:reifies* triples", + re-run with --raw-reifies to emit them as rdf:reifies triples", "warning:".yellow(), stats.annotations_unresolved, ); diff --git a/fluree-db-cli/src/commands/index.rs b/fluree-db-cli/src/commands/index.rs index ca4877f586..a11a33e48a 100644 --- a/fluree-db-cli/src/commands/index.rs +++ b/fluree-db-cli/src/commands/index.rs @@ -73,7 +73,7 @@ pub async fn index_ledger(fluree: &Fluree, ledger_id: &str) -> CliResult )>>", + )) + .stdout(predicate::str::contains(" ~ ").not()) + .stdout(predicate::str::contains("ns.flur.ee/db#reifies").not()); } /// Export → re-ingest → query, per format. @@ -2571,9 +2571,8 @@ fn export_annotations_round_trip_in_every_format() { } } -/// The bundle is up to seven predicates, not the three the issue showed: a -/// plain literal adds `reifiesDatatype`, a language-tagged one adds -/// `reifiesLang`. Each object shape has to reach the output. +/// A link's object keeps its datatype and language tag, so each object shape +/// — a ref, a plain literal, a language-tagged one — has to reach the output. #[test] fn export_annotation_object_shapes() { let tmp = TempDir::new().unwrap(); @@ -2647,13 +2646,10 @@ fn export_without_annotations_is_unchanged() { .stdout(predicate::str::contains("~").not()); } -/// An annotation written inside a named graph is not represented in the -/// output today: the forward lookup export uses is blind to named graphs, -/// though the rows are in the ledger and SPARQL reads them. Export must say -/// so — suppressing the `f:reifies*` rows and then emitting no marker is -/// exactly the silent truncation this work exists to remove. +/// An annotation written inside a named graph is exported like any other: +/// its link lives in the edge's graph, where the export's lookup finds it. #[test] -fn export_reports_annotations_it_could_not_resolve() { +fn export_resolves_named_graph_annotations() { let tmp = TempDir::new().unwrap(); fluree_cmd(&tmp).arg("init").assert().success(); let src = tmp.path().join("gann-src"); @@ -2675,24 +2671,8 @@ fn export_reports_annotations_it_could_not_resolve() { .args(["export", "gann", "--format", "trig", "--all-graphs"]) .assert() .success() - .stderr(predicate::str::contains( - "1 edge annotations could not be resolved", - )) - .stderr(predicate::str::contains("--raw-reifies")); - - // And the named remedy works: the bundle comes out verbatim. - fluree_cmd(&tmp) - .args([ - "export", - "gann", - "--format", - "trig", - "--all-graphs", - "--raw-reifies", - ]) - .assert() - .success() - .stdout(predicate::str::contains("reifiesSubject")); + .stdout(predicate::str::contains("ex:p ex:y ~ ex:cG")) + .stderr(predicate::str::contains("could not be resolved").not()); } /// Two exports of the same ledger must produce the same bytes. @@ -2832,21 +2812,13 @@ fn an_untranslated_annotation_exports_with_its_marker() { /// The counter still fires when an annotation genuinely cannot be resolved. /// -/// Paired with the test above on purpose. "No warning" is satisfied by a +/// Paired with the tests above on purpose. "No warning" is satisfied by a /// counter that has stopped working, so a `MustNotFire` assertion alone /// cannot distinguish "nothing was dropped" from "the accounting is dead". /// -/// The fixture is an annotation inside a named graph, which the base-index -/// seal scan cannot key on this branch. -/// -/// **This canary has a known expiry**, recorded here so the next person does -/// not mistake its retirement for a regression: the stacked seal fix makes -/// named-graph annotations resolve, at which point this stops firing and -/// must be replaced rather than deleted. The replacement wanted is a bundle -/// the decoder rejects outright — a `GraphMismatch` or a malformed bundle — -/// which is a corruption state rather than a defect, and which no supported -/// write surface can produce, since every write path rejects hand-written -/// `f:reifies*`. +/// The fixture is an export written before links whose `f:reifies*` bundle +/// names a triple the file never asserts. Import stores its link, and the +/// export has no base edge to hang the `~` marker on. #[test] fn the_unresolved_counter_still_fires_when_it_should() { let tmp = TempDir::new().unwrap(); @@ -2854,10 +2826,11 @@ fn the_unresolved_counter_still_fires_when_it_should() { let src = tmp.path().join("mf-src"); std::fs::create_dir_all(&src).unwrap(); std::fs::write( - src.join("a.trig"), + src.join("a.ttl"), "@prefix ex: .\n\ - GRAPH { \ - ex:x ex:p ex:y ~ ex:cG {| ex:src ex:d |} . }\n", + @prefix f: .\n\ + ex:cG f:reifiesSubject ex:x ; f:reifiesPredicate ex:p ; \ + f:reifiesObject ex:y ; ex:src ex:d .\n", ) .unwrap(); fluree_cmd(&tmp) @@ -2867,7 +2840,7 @@ fn the_unresolved_counter_still_fires_when_it_should() { .success(); fluree_cmd(&tmp) - .args(["export", "mf", "--format", "trig", "--all-graphs"]) + .args(["export", "mf", "--format", "turtle"]) .assert() .success() .stderr(predicate::str::contains( @@ -2882,8 +2855,8 @@ fn the_unresolved_counter_still_fires_when_it_should() { /// returns `None` for the `NUM_BIG_OVERFLOW` arena, because that arena holds /// both overflow `xsd:integer` and `xsd:decimal` and the o_type alone cannot /// say which. The `else { continue }` then skipped the row before it could -/// be matched against the arena, so the marker was lost on *every* lookup -/// path, sealed arena included, and the reifier came out as an orphan. +/// be matched, so the marker was lost on *every* lookup path and the reifier +/// came out as an orphan. /// /// `resolve_datatype_sid_for_value` exists for exactly that ambiguity — /// added for #1329, where the same gap rendered big numerics with an empty @@ -2953,26 +2926,13 @@ fn big_numeric_objects_keep_their_annotations() { } /// A point-in-time export shows an annotation that was live at that time, -/// even though it has since been retracted. -/// -/// It did not, on the arena-less path. `scan_base_index_for_attachment_events_in` -/// computed its own upper bound as `t.max(snapshot.t)` — right for a seal -/// pass, which wants the whole of history, and wrong for a read at a -/// requested `t`. Raising the bound to HEAD means the range never returns a -/// bundle retracted after the requested time, and the filter below it can -/// only *drop* rows, never restore them. The annotation vanished from an -/// export that should contain it, and because no edge was then known to be -/// annotated, nothing incremented the unresolved counter either — silent. -/// -/// The bound is now the caller's: seal callers clamp it themselves, the -/// export passes the requested time. +/// even though it has since been retracted: export reads the links as of the +/// requested `t`, not HEAD. /// -/// Three things the fixture needs, or it passes without exercising the bug: -/// the ledger must be **indexed past** the requested `t` (otherwise -/// `snapshot.t` is 0 and the clamp is a no-op), the annotation must be -/// **retracted after** it (otherwise it is live at HEAD too), and the scan -/// must be the annotation source (`FLUREE_EXPORT_ANNOTATION_SCAN`), since -/// that is the path the bound belongs to. +/// Two things the fixture needs, or it passes without exercising the bound: +/// the ledger must be **indexed past** the requested `t`, so the link's +/// assert and retract both come from the index, and the annotation must be +/// **retracted after** it (otherwise it is live at HEAD too). #[test] fn a_point_in_time_export_keeps_an_annotation_retracted_later() { let tmp = TempDir::new().unwrap(); @@ -3009,7 +2969,6 @@ fn a_point_in_time_export_keeps_an_annotation_retracted_later() { // At HEAD the annotation is gone — that is the retract working. fluree_cmd(&tmp) .args(["export", "tt", "--format", "turtle"]) - .env("FLUREE_EXPORT_ANNOTATION_SCAN", "1") .assert() .success() .stdout(predicate::str::contains("~ ").not()); @@ -3017,7 +2976,6 @@ fn a_point_in_time_export_keeps_an_annotation_retracted_later() { // At t=1 it was live, so it must be in the output. fluree_cmd(&tmp) .args(["export", "tt", "--format", "turtle", "--at", "1"]) - .env("FLUREE_EXPORT_ANNOTATION_SCAN", "1") .assert() .success() .stdout(predicate::str::contains("~ ")) diff --git a/fluree-db-core/src/lib.rs b/fluree-db-core/src/lib.rs index 16bbff31ba..bf4d7a8287 100644 --- a/fluree-db-core/src/lib.rs +++ b/fluree-db-core/src/lib.rs @@ -148,16 +148,16 @@ pub use ledger_id::{ ParsedLedgerId, COMMIT_PREFIX_MIN_LEN, DEFAULT_BRANCH, LEDGER_URN_PREFIX, TIME_TRAVEL_TAGS, }; pub use namespaces::{ - default_namespace_codes, is_owl_class_class, is_owl_datatype_property_class, - is_owl_equivalent_class, is_owl_equivalent_property, is_owl_functional_property, - is_owl_imports, is_owl_inverse_functional_property, is_owl_inverse_of, - is_owl_object_property_class, is_owl_ontology_class, is_owl_same_as, is_owl_symmetric_property, - is_owl_transitive_property, is_rdf_first, is_rdf_nil, is_rdf_property_class, is_rdf_reifies, - is_rdf_rest, is_rdf_type, is_rdfs_domain, is_rdfs_range, is_rdfs_subclass_of, - is_rdfs_subproperty_of, is_reifies_datatype, is_reifies_graph, is_reifies_lang, - is_reifies_list_index, is_reifies_object, is_reifies_predicate, is_reifies_subject, - is_reserved_reifies_predicate, is_scan_hidden_predicate, is_schema_class, is_schema_predicate, - rdf_reifies_sid, reifies_predicate_sids, triple_term_datatype_sid, + default_namespace_codes, is_annotation_predicate, is_owl_class_class, + is_owl_datatype_property_class, is_owl_equivalent_class, is_owl_equivalent_property, + is_owl_functional_property, is_owl_imports, is_owl_inverse_functional_property, + is_owl_inverse_of, is_owl_object_property_class, is_owl_ontology_class, is_owl_same_as, + is_owl_symmetric_property, is_owl_transitive_property, is_rdf_first, is_rdf_nil, + is_rdf_property_class, is_rdf_reifies, is_rdf_rest, is_rdf_type, is_rdfs_domain, is_rdfs_range, + is_rdfs_subclass_of, is_rdfs_subproperty_of, is_reifies_datatype, is_reifies_graph, + is_reifies_lang, is_reifies_list_index, is_reifies_object, is_reifies_predicate, + is_reifies_subject, is_reserved_reifies_predicate, is_scan_hidden_predicate, is_schema_class, + is_schema_predicate, rdf_reifies_sid, reifies_predicate_sids, triple_term_datatype_sid, }; pub use nonempty::NonEmpty; pub use ns_encoding::{ diff --git a/fluree-db-core/src/namespaces.rs b/fluree-db-core/src/namespaces.rs index 65a046d79c..a4f399d358 100644 --- a/fluree-db-core/src/namespaces.rs +++ b/fluree-db-core/src/namespaces.rs @@ -392,6 +392,13 @@ pub fn is_rdf_reifies(sid: &Sid) -> bool { sid.namespace_code == RDF && sid.name.as_ref() == fluree_vocab::rdf_names::REIFIES } +/// True for a predicate whose presence marks a ledger as annotated: the +/// `rdf:reifies` link, or a legacy `f:reifies*` bundle predicate. +#[inline] +pub fn is_annotation_predicate(sid: &Sid) -> bool { + is_rdf_reifies(sid) || is_reserved_reifies_predicate(sid) +} + /// True for the predicates wildcard scans hide from users: the seven /// `f:reifies*` bundle predicates and, while the RDF 1.2 link form is /// index-internal, `rdf:reifies`. Read-side only; the write firewall is diff --git a/fluree-db-cypher/src/lower/pattern.rs b/fluree-db-cypher/src/lower/pattern.rs index 1e8f2b74a1..54a2d05036 100644 --- a/fluree-db-cypher/src/lower/pattern.rs +++ b/fluree-db-cypher/src/lower/pattern.rs @@ -1143,15 +1143,11 @@ fn build_fixed_chain( } /// Whether a per-edge annotation probe can bind anything in this view: -/// `f:reifies*` must be in the dictionary, and the caller must not have proved -/// (index stats + overlay) that no `f:reifies*` fact exists. When it cannot, -/// every relationship value is the synthesized one and the probe is pure cost. +/// `rdf:reifies` must be in the dictionary, and the caller must not have +/// proved (index stats + overlay) that no link exists. When it cannot, every +/// relationship value is the synthesized one and the probe is pure cost. fn annotation_probe_possible(ctx: &LoweringContext<'_, E>) -> bool { - ctx.reified_edges_possible - && ctx - .encoder - .encode_iri(fluree_vocab::reifies_iris::SUBJECT) - .is_some() + ctx.reified_edges_possible && ctx.encoder.encode_iri(fluree_vocab::rdf::REIFIES).is_some() } /// Build the relationship-list value for a fixed chain — one element per hop — diff --git a/fluree-db-indexer/src/build/root_assembly.rs b/fluree-db-indexer/src/build/root_assembly.rs index 164961b49c..00cc6a4009 100644 --- a/fluree-db-indexer/src/build/root_assembly.rs +++ b/fluree-db-indexer/src/build/root_assembly.rs @@ -296,14 +296,14 @@ pub(crate) async fn encode_and_write_root_v6( .cloned() .collect(); - // Sticky bit: `true` once any `f:reifies*` predicate has been - // observed in the ledger's history. Detection is cheap — if any - // of the seven reserved reifies SIDs appears in the indexer's - // accumulated predicate dictionary, annotations exist (or did). + // Sticky bit: `true` once `rdf:reifies` or a legacy `f:reifies*` + // predicate has been observed in the ledger's history. Detection is + // cheap — if one appears in the indexer's accumulated predicate + // dictionary, annotations exist (or did). // Once a predicate enters the dict it stays there across // reindexes, so this naturally inherits sticky-bit semantics. let has_annotations = inputs.predicate_sids.iter().any(|(ns, name)| { - fluree_db_core::is_reserved_reifies_predicate(&fluree_db_core::Sid::new(*ns, name.as_str())) + fluree_db_core::is_annotation_predicate(&fluree_db_core::Sid::new(*ns, name.as_str())) }); let mut root = IndexRoot { diff --git a/fluree-db-indexer/src/run_index/build/incremental_root.rs b/fluree-db-indexer/src/run_index/build/incremental_root.rs index 1d7b12bca2..0664a7e681 100644 --- a/fluree-db-indexer/src/run_index/build/incremental_root.rs +++ b/fluree-db-indexer/src/run_index/build/incremental_root.rs @@ -142,8 +142,8 @@ impl IncrementalRootBuilder { /// Update inline predicate SIDs. /// - /// Also OR-updates `has_annotations` whenever any of the seven - /// reserved `f:reifies*` predicate SIDs appears in the new dict. + /// Also OR-updates `has_annotations` whenever `rdf:reifies` or a + /// legacy `f:reifies*` predicate SID appears in the new dict. /// The full-rebuild path computes the same bit at root-assembly /// time, but the incremental path clones the old root and would /// otherwise carry forward `has_annotations: false` even when the @@ -152,10 +152,7 @@ impl IncrementalRootBuilder { /// reindexes (predicate dicts only accumulate). pub fn set_predicate_sids(&mut self, sids: Vec<(u16, String)>) { let saw_reifies = sids.iter().any(|(ns, name)| { - fluree_db_core::is_reserved_reifies_predicate(&fluree_db_core::Sid::new( - *ns, - name.as_str(), - )) + fluree_db_core::is_annotation_predicate(&fluree_db_core::Sid::new(*ns, name.as_str())) }); self.root.has_annotations |= saw_reifies; self.root.predicate_sids = sids; diff --git a/fluree-db-novelty/src/lib.rs b/fluree-db-novelty/src/lib.rs index 9efb1c9f7d..1b79097411 100644 --- a/fluree-db-novelty/src/lib.rs +++ b/fluree-db-novelty/src/lib.rs @@ -577,6 +577,9 @@ pub struct Novelty { link_base: Option, /// Every commit at or before this `t` has its links in novelty. links_through: i64, + /// An `rdf:reifies` link has been applied; sticky, like + /// [`AttachmentNovelty::has_annotations`]. + saw_links: bool, } #[inline] @@ -602,9 +605,16 @@ impl Novelty { fact_state: NoveltyFactState::new(), link_base: None, links_through: t, + saw_links: false, } } + /// True once novelty has applied an annotation: an `rdf:reifies` link or + /// a legacy `f:reifies*` bundle flake. + pub fn has_annotations(&self) -> bool { + self.saw_links || self.attachments.has_annotations() + } + /// Resolve a flake's graph ID from its `Flake.g` field. /// /// - `None` → default graph (g_id = 0) @@ -963,6 +973,7 @@ impl Novelty { if fluree_db_core::namespaces::is_reserved_reifies_predicate(&flake.p) { accepted_reifies.push(flake.clone()); } + self.saw_links |= fluree_db_core::is_rdf_reifies(&flake.p); // Asserting OR retracting a hierarchy edge changes the RDFS // schema — invalidate the shared hierarchy cache. Likewise any // SHACL-vocabulary flake invalidates the compiled-shapes cache. @@ -1178,6 +1189,7 @@ impl Novelty { if fluree_db_core::namespaces::is_reserved_reifies_predicate(&flake.p) { accepted_reifies.push(flake.clone()); } + self.saw_links |= fluree_db_core::is_rdf_reifies(&flake.p); self.fact_state.record(g_id, flake); } diff --git a/fluree-db-query/src/eval/metadata.rs b/fluree-db-query/src/eval/metadata.rs index d3876c34c1..0507cedd98 100644 --- a/fluree-db-query/src/eval/metadata.rs +++ b/fluree-db-query/src/eval/metadata.rs @@ -700,7 +700,100 @@ fn node_property_binding(props: Vec, pred_sid: &Sid) -> Binding { } } -/// `type(rel)` → relationship type string from `f:reifiesPredicate`. +/// A reifier's `rdf:reifies` link flakes, raw: the provider's PSOT range, or +/// the overlay alone on a ledger with none. +fn read_link_flakes(ctx: &ExecutionContext<'_>, reifier: &Sid) -> Result> { + let reifies = fluree_db_core::rdf_reifies_sid(); + if ctx.active_snapshot.range_provider.is_some() { + return read_subject_predicate_flakes(ctx, reifier, reifies); + } + let mut flakes = Vec::new(); + if let Some(overlay) = ctx.overlay { + overlay.for_each_overlay_flake( + ctx.binary_g_id, + IndexType::Psot, + None, + None, + true, + ctx.to_t, + &mut |flake| { + if flake.s == *reifier && flake.p == *reifies { + flakes.push(flake.clone()); + } + }, + ); + } + Ok(flakes) +} + +/// The triple a reifier names: its live link's term, the least when it +/// names several. +fn live_term(flakes: Vec) -> Option { + let mut latest: HashMap = HashMap::new(); + for flake in flakes { + let FlakeValue::TripleTerm(term) = flake.o else { + continue; + }; + let entry = latest.entry(*term).or_insert((flake.t, flake.op)); + if flake.t > entry.0 { + *entry = (flake.t, flake.op); + } + } + latest + .into_iter() + .filter_map(|(term, (_, op))| op.then_some(term)) + .min() +} + +/// The triple `reifier` names, without view policy (fail-closed under one). +fn reified_triple( + ctx: &ExecutionContext<'_>, + reifier: &Sid, +) -> Result> { + if !ctx.allow_unfiltered() { + tracing::warn!( + "Cypher relationship lookup reached the sync path under an active view policy; \ + returning empty to avoid leaking unfiltered edges" + ); + return Ok(None); + } + Ok(live_term(read_link_flakes(ctx, reifier)?)) +} + +/// The triple `reifier` names, through the view policy. +async fn reified_triple_filtered( + ctx: &ExecutionContext<'_>, + reifier: &Sid, +) -> Result> { + if ctx.allow_unfiltered() { + return reified_triple(ctx, reifier); + } + let flakes = read_link_flakes(ctx, reifier)?; + let overlay = ctx.overlay.unwrap_or(&NoOverlay); + let flakes = crate::binary_scan::BinaryScanOperator::filter_flakes_by_policy( + ctx, + ctx.active_snapshot, + overlay, + ctx.to_t, + ctx.binary_g_id, + flakes, + ) + .await?; + Ok(live_term(flakes)) +} + +/// A relationship endpoint from the triple it reifies. +fn endpoint(term: fluree_db_core::TripleTermValue, start: bool) -> Option { + if start { + return Some(term.s); + } + match term.o { + FlakeValue::Ref(end) => Some(end), + _ => None, + } +} + +/// `type(rel)` → relationship type string: the reified triple's predicate. pub fn eval_rel_type( args: &[Expression], row: &R, @@ -717,25 +810,16 @@ pub fn eval_rel_type( ctx.tracker.consume_fuel(1)?; // A relationship value carries its predicate intrinsically (e.g. from - // `relationships(p)`); a reifier-node binding (bound `-[r:T]->`) needs the - // `f:reifiesPredicate` lookup. + // `relationships(p)`); a reifier-node binding (bound `-[r:T]->`) reads its + // link. let pred_sid = match &binding { Binding::Rel(rel) => rel.predicate.clone(), _ => { let Some(reifier) = binding_subject_sid(&binding, ctx)? else { return Ok(None); }; - let reifies_pred = ctx - .active_snapshot - .encode_iri(fluree_vocab::reifies_iris::PREDICATE) - .unwrap_or_else(|| { - Sid::new(fluree_vocab::namespaces::FLUREE_DB, "reifiesPredicate") - }); - match lookup_ref_objects(ctx, &reifier, &reifies_pred)? - .into_iter() - .next() - { - Some(p) => p, + match reified_triple(ctx, &reifier)? { + Some(term) => term.p, None => return Ok(None), } } @@ -745,47 +829,32 @@ pub fn eval_rel_type( Ok(name.map(|s| ComparableValue::String(Arc::from(s)))) } -/// `startNode(rel)` → the relationship's start node ref (`f:reifiesSubject`). +/// `startNode(rel)` → the relationship's start node ref. pub fn eval_start_node( args: &[Expression], row: &R, ctx: Option<&ExecutionContext<'_>>, ) -> Result> { - eval_rel_endpoint( - args, - row, - ctx, - fluree_vocab::reifies_iris::SUBJECT, - "reifiesSubject", - "startNode", - ) + eval_rel_endpoint(args, row, ctx, true, "startNode") } -/// `endNode(rel)` → the relationship's end node ref (`f:reifiesObject`). +/// `endNode(rel)` → the relationship's end node ref. pub fn eval_end_node( args: &[Expression], row: &R, ctx: Option<&ExecutionContext<'_>>, ) -> Result> { - eval_rel_endpoint( - args, - row, - ctx, - fluree_vocab::reifies_iris::OBJECT, - "reifiesObject", - "endNode", - ) + eval_rel_endpoint(args, row, ctx, false, "endNode") } -/// Shared body for `startNode` / `endNode`: read the named `f:reifies*` ref off -/// the reifier and return it as a node ref. Mirrors [`eval_rel_type`] but yields -/// the node SID (a ref) rather than a type-name string. +/// Shared body for `startNode` / `endNode`: an endpoint of the triple the +/// reifier names, as a node ref. Mirrors [`eval_rel_type`] but yields the node +/// SID (a ref) rather than a type-name string. fn eval_rel_endpoint( args: &[Expression], row: &R, ctx: Option<&ExecutionContext<'_>>, - reifies_iri: &str, - reifies_local: &'static str, + start: bool, fn_name: &str, ) -> Result> { let arg = arity1(args, fn_name)?; @@ -797,14 +866,9 @@ fn eval_rel_endpoint( }; // A relationship value carries its endpoints intrinsically; a reifier-node - // binding needs the `f:reifiesSubject`/`f:reifiesObject` lookup. `is_start` - // selects the field for the Rel case. + // binding reads its link. if let Binding::Rel(rel) = &binding { - let node = if reifies_iri == fluree_vocab::reifies_iris::SUBJECT { - &rel.start - } else { - &rel.end - }; + let node = if start { &rel.start } else { &rel.end }; return Ok(Some(ComparableValue::Sid(node.clone()))); } @@ -814,12 +878,9 @@ fn eval_rel_endpoint( ctx.tracker.consume_fuel(1)?; - let reifies = ctx - .active_snapshot - .encode_iri(reifies_iri) - .unwrap_or_else(|| Sid::new(fluree_vocab::namespaces::FLUREE_DB, reifies_local)); - let refs = lookup_ref_objects(ctx, &reifier, &reifies)?; - Ok(refs.first().map(|s| ComparableValue::Sid(s.clone()))) + Ok(reified_triple(ctx, &reifier)? + .and_then(|term| endpoint(term, start)) + .map(ComparableValue::Sid)) } // =========================================================================== @@ -903,7 +964,7 @@ pub(crate) async fn eval_node_property_async( } /// `type(rel)` — the `Rel` value carries its predicate intrinsically (no read); -/// a reifier-node binding reads `f:reifiesPredicate` through the policy filter. +/// a reifier-node binding reads its link through the policy filter. pub(crate) async fn eval_rel_type_async( args: &[Expression], row: &R, @@ -920,18 +981,8 @@ pub(crate) async fn eval_rel_type_async( let Some(reifier) = binding_subject_sid(&binding, ctx)? else { return Ok(None); }; - let reifies_pred = ctx - .active_snapshot - .encode_iri(fluree_vocab::reifies_iris::PREDICATE) - .unwrap_or_else(|| { - Sid::new(fluree_vocab::namespaces::FLUREE_DB, "reifiesPredicate") - }); - match lookup_ref_objects_filtered(ctx, &reifier, &reifies_pred) - .await? - .into_iter() - .next() - { - Some(p) => p, + match reified_triple_filtered(ctx, &reifier).await? { + Some(term) => term.p, None => return Ok(None), } } @@ -946,15 +997,7 @@ pub(crate) async fn eval_start_node_async( row: &R, ctx: &ExecutionContext<'_>, ) -> Result> { - eval_rel_endpoint_async( - args, - row, - ctx, - fluree_vocab::reifies_iris::SUBJECT, - "reifiesSubject", - "startNode", - ) - .await + eval_rel_endpoint_async(args, row, ctx, true, "startNode").await } /// `endNode(rel)` against the policy filter (or the intrinsic `Rel` field). @@ -963,15 +1006,7 @@ pub(crate) async fn eval_end_node_async( row: &R, ctx: &ExecutionContext<'_>, ) -> Result> { - eval_rel_endpoint_async( - args, - row, - ctx, - fluree_vocab::reifies_iris::OBJECT, - "reifiesObject", - "endNode", - ) - .await + eval_rel_endpoint_async(args, row, ctx, false, "endNode").await } /// Policy-filtered counterpart of [`eval_rel_endpoint`]. @@ -979,8 +1014,7 @@ async fn eval_rel_endpoint_async( args: &[Expression], row: &R, ctx: &ExecutionContext<'_>, - reifies_iri: &str, - reifies_local: &'static str, + start: bool, fn_name: &str, ) -> Result> { let arg = arity1(args, fn_name)?; @@ -988,23 +1022,17 @@ async fn eval_rel_endpoint_async( return Ok(None); }; if let Binding::Rel(rel) = &binding { - let node = if reifies_iri == fluree_vocab::reifies_iris::SUBJECT { - &rel.start - } else { - &rel.end - }; + let node = if start { &rel.start } else { &rel.end }; return Ok(Some(ComparableValue::Sid(node.clone()))); } let Some(reifier) = binding_subject_sid(&binding, ctx)? else { return Ok(None); }; ctx.tracker.consume_fuel(1)?; - let reifies = ctx - .active_snapshot - .encode_iri(reifies_iri) - .unwrap_or_else(|| Sid::new(fluree_vocab::namespaces::FLUREE_DB, reifies_local)); - let refs = lookup_ref_objects_filtered(ctx, &reifier, &reifies).await?; - Ok(refs.first().map(|s| ComparableValue::Sid(s.clone()))) + Ok(reified_triple_filtered(ctx, &reifier) + .await? + .and_then(|term| endpoint(term, start)) + .map(ComparableValue::Sid)) } /// Evaluate a metadata `Call` for one row through the policy-filtered async diff --git a/fluree-db-server/Cargo.toml b/fluree-db-server/Cargo.toml index 35dd1ba586..ef06eb6a0c 100644 --- a/fluree-db-server/Cargo.toml +++ b/fluree-db-server/Cargo.toml @@ -274,9 +274,3 @@ path = "tests/multi_query_tracing_integration.rs" name = "graph_source_query_logging" path = "tests/graph_source_query_logging.rs" -# Standalone for the same reason: it sets `FLUREE_EXPORT_ANNOTATION_SCAN` to -# select the export's fallback annotation source, and an env var is -# process-global while the assertions around it are per-test. -[[test]] -name = "export_scan_source" -path = "tests/export_scan_source.rs" diff --git a/fluree-db-server/tests/export_omission_headers.rs b/fluree-db-server/tests/export_omission_headers.rs index 9f95035749..348e248de2 100644 --- a/fluree-db-server/tests/export_omission_headers.rs +++ b/fluree-db-server/tests/export_omission_headers.rs @@ -15,18 +15,17 @@ //! ## Reachability of the three counters over HTTP //! //! - `x-fluree-export-named-graphs-omitted` — asserted both ways here. -//! - `x-fluree-export-annotations-unresolved` — asserted absent here, on both -//! annotation sources an HTTP client can reach. Its non-zero case needs the -//! base-index scan, which requires a process-global env var, so it lives in -//! the standalone `export_scan_source` target. +//! - `x-fluree-export-annotations-unresolved` — asserted absent here, before +//! and after an index build. Its non-zero case (a link whose base edge is +//! not asserted) is pinned by the CLI suite. //! - `x-fluree-export-annotations-out-of-scope` — emitted, and asserted //! absent, but **not asserted non-zero**. The counter is -//! `named − in_scope`: a reifier a marker pointed at whose own bundle the +//! `named − in_scope`: a reifier a marker pointed at whose own link the //! export never saw. The write path co-locates the two — an annotation's -//! `f:reifies*` rows are written into the same graph as the edge they -//! describe (`EdgeKey::to_reifies_facts`) — so no selection of graphs can -//! include the marker and exclude the bundle. Like the two below, it is a -//! corruption-class guard rather than a reachable state. +//! `rdf:reifies` link is written into the same graph as the edge it names — +//! so no selection of graphs can include the marker and exclude the link. +//! Like the two below, it is a corruption-class guard rather than a +//! reachable state. //! - `x-fluree-export-rows-skipped` — asserted absent. Every site that //! increments it is a dictionary miss (a subject or predicate id with no //! IRI), which no well-formed request produces; reaching it over the wire @@ -246,25 +245,22 @@ async fn a_complete_export_carries_no_omission_headers() { } } -/// Both annotation sources an HTTP client can reach — the novelty overlay -/// before an index exists, and the sealed arena after `POST /reindex` — carry -/// annotations written inside a named graph. Neither drops anything, so -/// neither reports anything. +/// An annotation written inside a named graph resolves both from novelty, +/// before an index exists, and from the index after `POST /reindex`. Neither +/// drops anything, so neither reports anything. /// /// The pairing is the point. A test that only asserts the header is absent /// passes on any fixture that never reached the code emitting it; this one -/// asserts the marker is present in the body for the same fixture, and its -/// counterpart in `export_scan_source` shows the header firing on the one -/// source that does drop it. +/// asserts the marker is present in the body for the same fixture. #[tokio::test] -async fn an_annotated_named_graph_resolves_from_both_reachable_sources() { +async fn an_annotated_named_graph_resolves_before_and_after_indexing() { let (_tmp, state) = test_state().await; let app = build_router(state); create_ledger(&app, "ann:main").await; upsert_trig(&app, "ann:main", ANNOTATED_NAMED_GRAPH).await; - for stage in ["novelty overlay", "sealed arena"] { - if stage == "sealed arena" { + for stage in ["novelty", "index"] { + if stage == "index" { reindex(&app, "ann:main").await; } let (status, headers, body) = export( diff --git a/fluree-db-server/tests/export_scan_source.rs b/fluree-db-server/tests/export_scan_source.rs deleted file mode 100644 index ce9805eb67..0000000000 --- a/fluree-db-server/tests/export_scan_source.rs +++ /dev/null @@ -1,198 +0,0 @@ -//! The one omission counter that needs its own process. -//! -//! `x-fluree-export-annotations-unresolved` goes non-zero when the export -//! resolves annotations through the **base-index scan**, which is blind to -//! annotations written inside a named graph. Selecting that source means -//! setting `FLUREE_EXPORT_ANNOTATION_SCAN`, which is process-global while -//! these assertions are per-test — so this is a standalone `[[test]]` target -//! rather than a `grp_http` member, for the same reason `telemetry_test` is. -//! -//! This is also the only coverage the kill switch has, and it shows what the -//! switch is for: the sources do *not* agree, and they disagree exactly on -//! named-graph annotations. `export_omission_headers` holds the other half — -//! the same fixture resolving cleanly through the arena and the overlay. - -use axum::body::Body; -use fluree_db_server::{routes::build_router, AppState, ServerConfig, TelemetryConfig}; -use http::{Request, StatusCode}; -use http_body_util::BodyExt; -use serde_json::json; -use std::sync::Arc; -use tempfile::TempDir; -use tower::ServiceExt; - -const UNRESOLVED: &str = "x-fluree-export-annotations-unresolved"; - -/// An edge annotation written inside a named graph — the shape the three -/// annotation sources disagree about. -const ANNOTATED_NAMED_GRAPH: &str = "@prefix ex: .\n\ - GRAPH { ex:x ex:p ex:y ~ ex:cG {| ex:src ex:d |} . }\n"; - -async fn test_state() -> (TempDir, Arc) { - let tmp = tempfile::tempdir().expect("tempdir"); - let cfg = ServerConfig { - cors_enabled: false, - indexing_enabled: false, - storage_path: Some(tmp.path().to_path_buf()), - ..Default::default() - }; - let telemetry = TelemetryConfig::with_server_config(&cfg); - let state = Arc::new(AppState::new(cfg, telemetry).await.expect("AppState::new")); - (tmp, state) -} - -async fn create_ledger(app: &axum::Router, ledger: &str) { - let resp = app - .clone() - .oneshot( - Request::builder() - .method("POST") - .uri("/v1/fluree/create") - .header("content-type", "application/json") - .body(Body::from(json!({ "ledger": ledger }).to_string())) - .expect("request"), - ) - .await - .expect("router response"); - assert!( - resp.status().is_success(), - "create {ledger}: {}", - resp.status() - ); -} - -/// Upsert TriG, which is the only HTTP surface that can place a named graph. -async fn upsert_trig(app: &axum::Router, ledger: &str, body: &str) { - let resp = app - .clone() - .oneshot( - Request::builder() - .method("POST") - .uri(format!("/v1/fluree/upsert/{ledger}")) - .header("content-type", "application/trig") - .body(Body::from(body.to_string())) - .expect("request"), - ) - .await - .expect("router response"); - let status = resp.status(); - let bytes = resp.into_body().collect().await.expect("body").to_bytes(); - assert!( - status.is_success(), - "upsert trig into {ledger}: {status} {}", - String::from_utf8_lossy(&bytes) - ); -} - -/// `POST /reindex` is synchronous, so a test can move a ledger from the -/// novelty overlay to a sealed index without racing a background indexer. -async fn reindex(app: &axum::Router, ledger: &str) { - let resp = app - .clone() - .oneshot( - Request::builder() - .method("POST") - .uri("/v1/fluree/reindex") - .header("content-type", "application/json") - .body(Body::from(json!({ "ledger": ledger }).to_string())) - .expect("request"), - ) - .await - .expect("router response"); - let status = resp.status(); - let bytes = resp.into_body().collect().await.expect("body").to_bytes(); - assert!( - status.is_success(), - "reindex {ledger}: {status} {}", - String::from_utf8_lossy(&bytes) - ); -} - -/// `(status, headers, body)` from a real `POST /export`. -async fn export( - app: &axum::Router, - ledger: &str, - body: serde_json::Value, -) -> (StatusCode, http::HeaderMap, String) { - let resp = app - .clone() - .oneshot( - Request::builder() - .method("POST") - .uri(format!("/v1/fluree/export/{ledger}")) - .header("content-type", "application/json") - .body(Body::from(body.to_string())) - .expect("request"), - ) - .await - .expect("router response"); - let status = resp.status(); - let headers = resp.headers().clone(); - let bytes = resp.into_body().collect().await.expect("body").to_bytes(); - ( - status, - headers, - String::from_utf8_lossy(&bytes).into_owned(), - ) -} - -fn header(headers: &http::HeaderMap, name: &str) -> Option { - headers - .get(name) - .map(|v| v.to_str().expect("header is ascii").to_string()) -} - -#[tokio::test] -async fn the_base_index_scan_reports_the_annotation_it_cannot_see() { - std::env::set_var("FLUREE_EXPORT_ANNOTATION_SCAN", "1"); - - let (_tmp, state) = test_state().await; - let app = build_router(state); - create_ledger(&app, "scan:main").await; - upsert_trig(&app, "scan:main", ANNOTATED_NAMED_GRAPH).await; - // Seal an index, so the scan has a base to read and the overlay — which is - // *not* blind — is no longer the source that answers. - reindex(&app, "scan:main").await; - - let (status, headers, body) = export( - &app, - "scan:main", - json!({ "format": "trig", "all_graphs": true }), - ) - .await; - assert_eq!(status, StatusCode::OK); - assert!( - body.contains("http://example.org/g1"), - "the graph itself must still export, or the header below would be \ - reporting on an export that produced nothing: {body}" - ); - assert!( - !body.contains("~ "), - "this is the source that cannot emit the marker: {body}" - ); - assert_eq!( - header(&headers, UNRESOLVED).as_deref(), - Some("1"), - "the annotation the scan could not see must be reported; headers: {headers:?}" - ); - - // `raw_reifies` is the remedy the header's docs name. Over HTTP it is a - // request field; with it the bundle comes out verbatim, so there is - // nothing dropped left to report. - let (status, headers, body) = export( - &app, - "scan:main", - json!({ "format": "trig", "all_graphs": true, "raw_reifies": true }), - ) - .await; - assert_eq!(status, StatusCode::OK); - assert!( - body.contains("reifiesSubject"), - "raw_reifies must emit the bundle verbatim: {body}" - ); - assert_eq!( - header(&headers, UNRESOLVED), - None, - "raw_reifies drops nothing, so reports nothing; headers: {headers:?}" - ); -} diff --git a/fluree-db-transact/src/flake_sink.rs b/fluree-db-transact/src/flake_sink.rs index f839fecb76..c58ea2023e 100644 --- a/fluree-db-transact/src/flake_sink.rs +++ b/fluree-db-transact/src/flake_sink.rs @@ -7,8 +7,6 @@ use crate::error::TransactError; use crate::generate::{infer_datatype, validate_value_dt_pair}; use crate::namespace::{NamespaceRegistry, NsAllocator}; use crate::value_convert::{convert_native_literal, convert_string_literal}; -#[cfg(test)] -use fluree_db_core::edge::EdgeKey; use fluree_db_core::DatatypeConstraint; use fluree_db_core::{Flake, FlakeMeta, FlakeValue, Sid}; use fluree_graph_ir::{Datatype, GraphSink, LiteralValue, SinkResult, TermId}; @@ -155,9 +153,9 @@ impl<'a> FlakeSink<'a> { // Reserved-predicate firewall (mirrors the JSON-LD and SPARQL UPDATE // surfaces): a user-authored `f:reifies*` statement must not reach - // stage. Annotations are minted only through the RDF 1.2 annotation - // syntax (`~` / `{| |}` / `<< >>`), which arrives via - // `emit_reified_triple` and builds a complete, validated bundle. + // stage. These predicates are the retired attachment encoding; the + // RDF 1.2 annotation syntax (`~` / `{| |}` / `<< >>`) arrives via + // `emit_reified_triple` and writes the `rdf:reifies` link. // Bulk import (`ImportSink`) is the administrative bootstrap path and // deliberately stays permissive so an export round-trips. if fluree_db_core::is_reserved_reifies_predicate(&p) { @@ -325,11 +323,8 @@ impl GraphSink for FlakeSink<'_> { true } - /// Turtle-star reifier attachment → the durable `f:reifies*` bundle, - /// built by the shared [`crate::generate::flakes::reified_triple_bundle`] - /// (bit-identical with the JSON-LD `@annotation` lowering and with - /// `ImportSink`'s bulk path). The base triple has already been emitted - /// by the parser via `emit_triple`. + /// The reified triple's link; the parser has already emitted the base + /// triple through `emit_triple`. fn emit_reified_triple( &mut self, subject: TermId, @@ -350,10 +345,10 @@ impl GraphSink for FlakeSink<'_> { return Ok(()); }; - match crate::generate::flakes::reified_triple_bundle(None, s, p, o, &dtc, &ann, self.t) { - Ok(bundle) => self.flakes.extend(bundle), + match crate::generate::flakes::reified_triple_link(None, s, p, o, &dtc, &ann, self.t) { + Ok(link) => self.flakes.push(link), Err(e) => { - tracing::error!("FlakeSink: invariant violation in reifier bundle, aborting — {e}"); + tracing::error!("FlakeSink: invariant violation in reified triple, aborting — {e}"); if self.invariant_error.is_none() { self.invariant_error = Some(e); } @@ -613,15 +608,22 @@ mod tests { ); } - #[test] - fn test_reified_triple_emits_jsonld_compatible_bundle() { - // Ref-object, default-graph: exactly Subject + Predicate + Object — - // NO f:reifiesDatatype (the JSON-LD-compatible shape), and the - // bundle decodes back to the base edge's EdgeKey. - use fluree_db_core::namespaces::{ - is_reifies_datatype, is_reifies_object, is_reifies_predicate, is_reifies_subject, - }; + /// The base flake and the term its link names. + fn base_and_term(flakes: &[Flake]) -> (&Flake, &fluree_db_core::TripleTermValue) { + assert_eq!(flakes.len(), 2, "base + link: {flakes:?}"); + let (base, link) = (&flakes[0], &flakes[1]); + assert!(fluree_db_core::is_rdf_reifies(&link.p)); + assert_eq!(link.s.name.as_ref(), "reifier"); + assert_eq!(link.dt, *fluree_db_core::triple_term_datatype_sid()); + assert!(link.op && link.t == base.t && link.g.is_none()); + match &link.o { + FlakeValue::TripleTerm(term) => (base, term), + other => panic!("link object is not a triple term: {other:?}"), + } + } + #[test] + fn test_reified_triple_emits_link() { let (mut ns, t, txn_id) = make_sink(); let mut sink = FlakeSink::new(&mut ns, t, txn_id); @@ -633,37 +635,15 @@ mod tests { sink.emit_reified_triple(s, p, o, r).unwrap(); let flakes = sink.into_flakes().expect("no invariant violation"); - // 1 base + 3 bundle flakes. - assert_eq!(flakes.len(), 4); - let base = &flakes[0]; - let bundle = &flakes[1..]; - assert!(bundle.iter().any(|f| is_reifies_subject(&f.p))); - assert!(bundle.iter().any(|f| is_reifies_predicate(&f.p))); - assert!(bundle.iter().any(|f| is_reifies_object(&f.p))); - assert!( - !bundle.iter().any(|f| is_reifies_datatype(&f.p)), - "JSON-LD-compatible bundle must omit f:reifiesDatatype: {bundle:?}" - ); - for f in bundle { - assert!(f.op, "assertion bundle"); - assert_eq!(f.t, t); - assert!(f.g.is_none(), "plain Turtle is default-graph"); - } - let decoded = EdgeKey::from_reifies_facts(bundle).expect("bundle decodes"); + let (base, term) = base_and_term(&flakes); assert_eq!( - decoded, - EdgeKey::from_flake(base), - "decoded EdgeKey must equal the base edge's EdgeKey" + (&term.s, &term.p, &term.o, &term.dt, &term.lang), + (&base.s, &base.p, &base.o, &base.dt, &None) ); } #[test] - fn test_reified_triple_lang_literal_bundle_carries_lang() { - // Language-tagged object: bundle adds f:reifiesLang and the - // f:reifiesObject flake carries m.lang (cascade symmetry with the - // JSON-LD writer — see EdgeKey docs / BUGS-2). - use fluree_db_core::namespaces::{is_reifies_lang, is_reifies_object}; - + fn test_reified_lang_literal_link_carries_lang() { let (mut ns, t, txn_id) = make_sink(); let mut sink = FlakeSink::new(&mut ns, t, txn_id); @@ -675,26 +655,9 @@ mod tests { sink.emit_reified_triple(s, p, o, r).unwrap(); let flakes = sink.into_flakes().expect("no invariant violation"); - // 1 base + 4 bundle flakes (S, P, O, Lang). - assert_eq!(flakes.len(), 5); - let base = &flakes[0]; - let bundle = &flakes[1..]; - let obj = bundle - .iter() - .find(|f| is_reifies_object(&f.p)) - .expect("f:reifiesObject"); - assert_eq!( - obj.m.as_ref().and_then(|m| m.lang.as_deref()), - Some("fr"), - "f:reifiesObject must carry m.lang" - ); - let lang = bundle - .iter() - .find(|f| is_reifies_lang(&f.p)) - .expect("f:reifiesLang"); - assert!(matches!(&lang.o, FlakeValue::String(l) if l == "fr")); - let decoded = EdgeKey::from_reifies_facts(bundle).expect("bundle decodes"); - assert_eq!(decoded, EdgeKey::from_flake(base)); + let (base, term) = base_and_term(&flakes); + assert_eq!((&term.o, &term.dt), (&base.o, &base.dt)); + assert_eq!(term.lang.as_deref(), Some("fr")); } #[test] diff --git a/fluree-db-transact/src/generate/flakes.rs b/fluree-db-transact/src/generate/flakes.rs index 76310fb95b..fd846ebf26 100644 --- a/fluree-db-transact/src/generate/flakes.rs +++ b/fluree-db-transact/src/generate/flakes.rs @@ -609,27 +609,11 @@ pub(crate) fn validate_value_dt_pair(val: &FlakeValue, dt: &Sid) -> Result<()> { Ok(()) } -/// Shared core of the two `emit_reified_triple` impls (`FlakeSink` / -/// `ImportSink`) and of TriG bulk import: a resolved reifier attachment → -/// validated → the `EdgeKey::to_reifies_facts_jsonld_compatible` bundle, -/// so the bit-identical-bundles guarantee (Turtle-star ≡ JSON-LD -/// `@annotation` at the flake level; cascade retracts cancel either -/// surface) cannot drift between those callers — each keeps only its own -/// error channel and emission (Vec-extend vs commit-writer/spool). -/// Transactional TriG (`convert_named_graphs_to_templates` in -/// `fluree-db-api`) is a second builder emitting templates; its parity is -/// pinned by `trig_star_in_graph_block_matches_jsonld_named_graph_annotation`. -/// -/// `g` is `None` for the Turtle sinks (default graph) and the block's graph -/// for TriG import; list-occurrence annotations are deferred in v1 → -/// `list_i = None`. -/// -/// The validation is the same late hard guard as `build_flake` / -/// `push_triple`: a bad (value, dt) pair must fail the whole ingest, not -/// silently drop or corrupt the bundle. The base triple hit the same -/// guard already, so this only fires on shapes the base emission also -/// rejected. -pub(crate) fn reified_triple_bundle( +/// The RDF 1.2 link `ann rdf:reifies <<( s p o )>>` for a reified triple, in +/// the triple's graph (`None` for the default graph). Shared by the Turtle +/// sinks and TriG bulk import. The object pair gets the same late guard as +/// `build_flake`, so a bad (value, datatype) pair fails the ingest. +pub(crate) fn reified_triple_link( g: Option, s: Sid, p: Sid, @@ -637,20 +621,23 @@ pub(crate) fn reified_triple_bundle( dtc: &fluree_db_core::DatatypeConstraint, ann: &Sid, t: i64, -) -> Result> { +) -> Result { let dt = dtc.datatype().clone(); - let lang = dtc.lang_tag().map(std::string::ToString::to_string); validate_value_dt_pair(&o, &dt)?; - let key = fluree_db_core::edge::EdgeKey { - g, + let term = fluree_db_core::TripleTermValue { s, p, o, dt, - lang, - list_i: None, + lang: dtc.lang_tag().map(str::to_string), }; - Ok(key.to_reifies_facts_jsonld_compatible(ann, t, true)) + let reifies = fluree_db_core::rdf_reifies_sid().clone(); + let value = FlakeValue::TripleTerm(Box::new(term)); + let dt = fluree_db_core::triple_term_datatype_sid().clone(); + Ok(match g { + Some(g) => Flake::new_in_graph(g, ann.clone(), reifies, value, dt, t, true, None), + None => Flake::new(ann.clone(), reifies, value, dt, t, true, None), + }) } /// Infer datatype from a FlakeValue diff --git a/fluree-db-transact/src/import.rs b/fluree-db-transact/src/import.rs index 23bc4f1e6d..c1ed7b0132 100644 --- a/fluree-db-transact/src/import.rs +++ b/fluree-db-transact/src/import.rs @@ -684,9 +684,7 @@ mod inner { } } - // TriG-star: `f:reifies*` bundles in the same graph as the edge - // they reify — the bundle the default-graph `ImportSink` emits, - // with `f:reifiesGraph` set. + // TriG-star: each reified triple's link, in the triple's graph. for r in &block.reified { let ann = expand_term(&r.reifier, &block.prefixes, &mut worker_cache, skolem_base)?; let s = expand_term(&r.subject, &block.prefixes, &mut worker_cache, skolem_base)?; @@ -702,45 +700,22 @@ mod inner { Some(lang) => fluree_db_core::DatatypeConstraint::LangTag(Arc::from(lang)), None => fluree_db_core::DatatypeConstraint::Explicit(dt), }; - let bundle = crate::generate::flakes::reified_triple_bundle( + let link = crate::generate::flakes::reified_triple_link( Some(graph_sid.clone()), - s.clone(), - p.clone(), + s, + p, o, &dtc, &ann, new_t, )?; - // The RDF 1.2 link rides alongside the bundle, as on the - // default-graph path. - if let (Some(sc), Some(object)) = ( - spool_ctx.as_mut(), - bundle - .iter() - .find(|f| fluree_db_core::is_reifies_object(&f.p)), - ) { - sc.push_named_graph_link(g_id, &s, &p, object, new_t)?; - } - for flake in bundle { - if let Some(sc) = spool_ctx.as_mut() { - sc.push_named_graph_record( - g_id, - crate::import_sink::FlakeRecord { - s: &flake.s, - p: &flake.p, - o: &flake.o, - dt: &flake.dt, - lang: flake.m.as_ref().and_then(|m| m.lang.as_deref()), - list_index: None, - t: new_t, - }, - )?; - } - writer.push_flake(&flake).map_err(|e| { - TransactError::Parse(format!("failed to encode reifier bundle flake: {e}")) - })?; - op_count += 1; + if let (Some(sc), FlakeValue::TripleTerm(term)) = (spool_ctx.as_mut(), &link.o) { + sc.push_named_graph_link(g_id, &ann, term, new_t)?; } + writer.push_flake(&link).map_err(|e| { + TransactError::Parse(format!("failed to encode link flake: {e}")) + })?; + op_count += 1; } } diff --git a/fluree-db-transact/src/import_sink.rs b/fluree-db-transact/src/import_sink.rs index a0f5a3c908..45cb664fe8 100644 --- a/fluree-db-transact/src/import_sink.rs +++ b/fluree-db-transact/src/import_sink.rs @@ -573,32 +573,26 @@ mod inner { Ok(()) } - /// Spool the RDF 1.2 link `ann rdf:reifies <<( s p o )>>` for a reified - /// base edge. + /// Spool the RDF 1.2 link `ann rdf:reifies <<( s p o )>>`. /// - /// The base edge becomes a pseudo-record in the chunk's term table, - /// resolved as the base triple's own record was, except that a + /// The term becomes a pseudo-record in the chunk's term table, + /// resolved as the base triple's own record would be, except that a /// decimal, big-integer or vector object holds the string id of its /// canonical form. The link record's `o_key` is that entry's ordinal; /// the build remaps the entry to global ids and interns it, replacing /// the ordinal with the term handle. - /// - /// `object` is the bundle's `f:reifiesObject` flake: its subject is - /// the reifier and its object, datatype and tag are the base edge's. fn write_link_record( &mut self, - s: &Sid, - p: &Sid, - object: &Flake, + ann: &Sid, + term: &fluree_db_core::TripleTermValue, t: i64, ) -> Result<(), CommitCodecError> { - let ann = &object.s; - let s_id = self.assign_subject_id(s); - let p_id = self.assign_predicate_id(p); - let dt_id = self.assign_datatype_id(&object.dt)?; + let s_id = self.assign_subject_id(&term.s); + let p_id = self.assign_predicate_id(&term.p); + let dt_id = self.assign_datatype_id(&term.dt)?; // An arena handle names a value only within one graph and // predicate; a term keys the object by its canonical form. - let resolved = match fluree_db_core::triple_term::lexical_term_object(&object.o) { + let resolved = match fluree_db_core::triple_term::lexical_term_object(&term.o) { Some((o_type, form)) => { let o_kind = if o_type == fluree_db_core::o_type::OType::VECTOR { ObjKind::VECTOR_ID @@ -608,15 +602,14 @@ mod inner { let id = self.assign_string_id(&form); Some((o_kind.as_u8(), ObjKey::encode_u32_id(id).as_u64())) } - None => self.resolve_object_value(&object.o, p_id), + None => self.resolve_object_value(&term.o, p_id), }; let Some((o_kind, o_key)) = resolved else { return Ok(()); }; - let lang_id = object - .m - .as_ref() - .and_then(|m| m.lang.as_deref()) + let lang_id = term + .lang + .as_deref() .map(|l| self.assign_lang_id(l)) .unwrap_or(0); let ordinal = self.terms.len() as u64; @@ -697,19 +690,18 @@ mod inner { result } - /// Spool the `rdf:reifies` link of a reified edge in an explicit named - /// graph (`g_id`); see [`Self::write_link_record`]. + /// Spool an `rdf:reifies` link in an explicit named graph (`g_id`); + /// see [`Self::write_link_record`]. pub fn push_named_graph_link( &mut self, g_id: GraphId, - s: &Sid, - p: &Sid, - object: &Flake, + ann: &Sid, + term: &fluree_db_core::TripleTermValue, t: i64, ) -> Result<(), CommitCodecError> { let saved = self.g_id; self.g_id = g_id; - let result = self.write_link_record(s, p, object, t); + let result = self.write_link_record(ann, term, t); self.g_id = saved; result } @@ -759,18 +751,17 @@ mod inner { prefix_map: HashMap, /// Optional spool context for Tier 2 parallel pipeline. spool_ctx: Option, - /// Each reifier's `f:reifiesSubject` / `f:reifiesPredicate` / - /// `f:reifiesObject` flakes seen so far, when its bundle arrives as - /// plain triples (the JSON-LD `@annotation` lowering). + /// Each reifier's `f:reifies*` slots seen so far, when its triple + /// arrives as slots (see [`Self::absorb_slot`]). pending_links: HashMap, } - /// A bundle's link-bearing slots, collected until all three are in. + /// A reifier's slots, collected until all three are in. #[derive(Default)] struct PendingLink { s: Option, p: Option, - object: Option, + o: Option<(FlakeValue, DatatypeConstraint)>, } impl<'a> ImportSink<'a> { @@ -917,6 +908,10 @@ mod inner { let Some((o, dtc)) = self.resolve_object(object) else { return; }; + if fluree_db_core::is_reserved_reifies_predicate(&p) { + self.absorb_slot(s, &p, o, dtc); + return; + } // `rdf:reifies` names a triple term; the reified-triple forms // arrive through `emit_reified_triple`, so an ordinary object @@ -971,6 +966,12 @@ mod inner { // Write spool record only after commit encoding succeeded if let Some(ctx) = &mut self.spool_ctx { + if let FlakeValue::TripleTerm(term) = &o { + if let Err(e) = ctx.write_link_record(&s, term, self.t) { + self.encode_error.get_or_insert(e); + } + return; + } let written = ctx.write_record(FlakeRecord { s: &s, p: &p, @@ -983,42 +984,68 @@ mod inner { if let Err(e) = written { self.encode_error.get_or_insert(e); } - self.observe_bundle_slot(&flake); } } - /// Spool the RDF 1.2 link of a bundle that arrives as plain triples - /// once its subject, predicate and object slots are in, as - /// `emit_reified_triple` does for a reified triple it parses. - fn observe_bundle_slot(&mut self, flake: &Flake) { - let entry = if fluree_db_core::is_reifies_subject(&flake.p) { - let FlakeValue::Ref(sid) = &flake.o else { - return; - }; - let entry = self.pending_links.entry(flake.s.clone()).or_default(); - entry.s = Some(sid.clone()); - entry - } else if fluree_db_core::is_reifies_predicate(&flake.p) { - let FlakeValue::Ref(sid) = &flake.o else { - return; - }; - let entry = self.pending_links.entry(flake.s.clone()).or_default(); - entry.p = Some(sid.clone()); - entry - } else if fluree_db_core::is_reifies_object(&flake.p) { - let entry = self.pending_links.entry(flake.s.clone()).or_default(); - entry.object = Some(flake.clone()); - entry - } else { + /// The JSON-LD annotation lowering, and exports written before + /// links, describe a reified triple by its `f:reifies*` slots; a + /// reifier's slots become its `rdf:reifies` link once its subject, + /// predicate and object are in. The object slot carries the + /// triple's datatype and tag, so the other slots add nothing. + fn absorb_slot(&mut self, ann: Sid, slot: &Sid, o: FlakeValue, dtc: DatatypeConstraint) { + use fluree_vocab::db; + let entry = self.pending_links.entry(ann.clone()).or_default(); + match (slot.name.as_ref(), o) { + (db::REIFIES_SUBJECT, FlakeValue::Ref(s)) => entry.s = Some(s), + (db::REIFIES_PREDICATE, FlakeValue::Ref(p)) => entry.p = Some(p), + (db::REIFIES_OBJECT, o) => entry.o = Some((o, dtc)), + _ => return, + } + if entry.s.is_none() || entry.p.is_none() || entry.o.is_none() { return; + } + let Some(PendingLink { + s: Some(s), + p: Some(p), + o: Some((o, dtc)), + }) = self.pending_links.remove(&ann) + else { + unreachable!("all three slots checked above"); }; - let (Some(s), Some(p), Some(object)) = (&entry.s, &entry.p, &entry.object) else { + self.push_link(s, p, o, &dtc, &ann); + } + + /// Write `ann rdf:reifies <<( s p o )>>` to the commit and the spool. + fn push_link( + &mut self, + s: Sid, + p: Sid, + o: FlakeValue, + dtc: &DatatypeConstraint, + ann: &Sid, + ) { + let link = + match crate::generate::flakes::reified_triple_link(None, s, p, o, dtc, ann, self.t) + { + Ok(link) => link, + Err(e) => { + if self.encode_error.is_none() { + let msg = format!("invariant violation in reified triple: {e}"); + tracing::error!("ImportSink: {msg}"); + self.encode_error = Some(CommitCodecError::InvalidOp(msg)); + } + return; + } + }; + if let Err(e) = self.writer.push_flake(&link) { + if self.encode_error.is_none() { + tracing::error!("ImportSink: link flake encode failed: {}", e); + self.encode_error = Some(e); + } return; - }; - let (s, p, object) = (s.clone(), p.clone(), object.clone()); - self.pending_links.remove(&flake.s); - if let Some(ctx) = &mut self.spool_ctx { - if let Err(e) = ctx.write_link_record(&s, &p, &object, self.t) { + } + if let (Some(ctx), FlakeValue::TripleTerm(term)) = (&mut self.spool_ctx, &link.o) { + if let Err(e) = ctx.write_link_record(ann, term, self.t) { self.encode_error.get_or_insert(e); } } @@ -1144,13 +1171,9 @@ mod inner { true } - /// Turtle-star reifier attachment → the durable `f:reifies*` bundle, - /// built by the shared - /// [`crate::generate::flakes::reified_triple_bundle`] (bit-identical - /// with the JSON-LD `@annotation` lowering and with `FlakeSink`'s - /// transactional path) and streamed through the commit writer (and - /// spool, when attached) exactly like ordinary triples. The base - /// triple has already been emitted by the parser via `emit_triple`. + /// The reified triple's link, written to the commit and the spool + /// like any triple; the parser has already emitted the base triple + /// through `emit_triple`. fn emit_reified_triple( &mut self, subject: TermId, @@ -1171,69 +1194,7 @@ mod inner { return Ok(()); }; - let bundle = match crate::generate::flakes::reified_triple_bundle( - None, s, p, o, &dtc, &ann, self.t, - ) { - Ok(bundle) => bundle, - Err(e) => { - if self.encode_error.is_none() { - let msg = format!("invariant violation in reifier bundle: {e}"); - tracing::error!("ImportSink: {msg}"); - self.encode_error = Some(CommitCodecError::InvalidOp(msg)); - } - return Ok(()); - } - }; - for flake in &bundle { - if let Err(e) = self.writer.push_flake(flake) { - if self.encode_error.is_none() { - tracing::error!("ImportSink: reifier bundle flake encode failed: {}", e); - self.encode_error = Some(e); - } - return Ok(()); // Don't spool a flake that failed to encode - } - if let Some(ctx) = &mut self.spool_ctx { - let written = ctx.write_record(FlakeRecord { - s: &flake.s, - p: &flake.p, - o: &flake.o, - dt: &flake.dt, - lang: flake.m.as_ref().and_then(|m| m.lang.as_deref()), - list_index: None, - t: self.t, - }); - if let Err(e) = written { - self.encode_error.get_or_insert(e); - } - } - } - // The RDF 1.2 link form rides alongside the bundle: the bundle's - // object flake carries the base edge's value, datatype and tag. - if let Some(ctx) = &mut self.spool_ctx { - let obj = bundle - .iter() - .find(|f| fluree_db_core::is_reifies_object(&f.p)); - let subj = bundle - .iter() - .find(|f| fluree_db_core::is_reifies_subject(&f.p)) - .and_then(|f| match &f.o { - FlakeValue::Ref(sid) => Some(sid), - _ => None, - }); - let pred = bundle - .iter() - .find(|f| fluree_db_core::is_reifies_predicate(&f.p)) - .and_then(|f| match &f.o { - FlakeValue::Ref(sid) => Some(sid), - _ => None, - }); - if let (Some(obj), Some(subj), Some(pred)) = (obj, subj, pred) { - let written = ctx.write_link_record(subj, pred, obj, self.t); - if let Err(e) = written { - self.encode_error.get_or_insert(e); - } - } - } + self.push_link(s, p, o, &dtc, &ann); Ok(()) } } diff --git a/fluree-db-transact/src/lower_cypher_update.rs b/fluree-db-transact/src/lower_cypher_update.rs index faf7c0f8d5..f9bcf53667 100644 --- a/fluree-db-transact/src/lower_cypher_update.rs +++ b/fluree-db-transact/src/lower_cypher_update.rs @@ -44,7 +44,7 @@ use fluree_db_cypher::ast::{ RemoveItem, SetClause, SetItem, Statement, Update, Variable, WithClause, WriteClause, }; -use crate::ir::{TemplateTerm, TripleTemplate, Txn, TxnOpts, TxnType}; +use crate::ir::{TemplateTerm, TemplateTripleTerm, TripleTemplate, Txn, TxnOpts, TxnType}; use crate::namespace::NamespaceRegistry; /// Errors raised by Cypher → Txn lowering. @@ -1066,8 +1066,8 @@ impl<'a> CypherLowering<'a> { } /// `SET n = { ... }` — replace all scalar node properties visible to Cypher. - /// Labels (`rdf:type`), relationship edges (ref-valued objects), and - /// `f:reifies*` sidecar facts are not node properties and are preserved. + /// Labels (`rdf:type`), relationship edges (ref-valued objects), and a + /// relationship's `rdf:reifies` link are not properties and are preserved. fn replace_property_map(&mut self, target: &str, map: &MapLit) -> Result<(), LowerCypherError> { self.push_optional_old_data_properties(target); for (key, val_expr) in &map.entries { @@ -1587,12 +1587,12 @@ impl<'a> CypherLowering<'a> { let p_expr = UnresolvedExpression::var(&p_name); let p_str = unresolved_call("str", vec![p_expr.clone()]); let old_expr = UnresolvedExpression::var(&old_name); - let mut filters = Vec::with_capacity(2 + reifies_iris::ALL.len()); + let mut filters = Vec::with_capacity(3 + reifies_iris::ALL.len()); filters.push(unresolved_call( "!=", vec![p_str.clone(), UnresolvedExpression::string(rdf::TYPE)], )); - for iri in reifies_iris::ALL { + for iri in std::iter::once(rdf::REIFIES).chain(reifies_iris::ALL) { filters.push(unresolved_call( "!=", vec![p_str.clone(), UnresolvedExpression::string(iri)], @@ -1734,7 +1734,7 @@ impl<'a> CypherLowering<'a> { )); // LPG semantics: every Cypher-created relationship gets identity — a - // fresh `f:reifies*` reifier bundle — so it is visible to named reads + // fresh reifier and its `rdf:reifies` link — so it is visible to named reads // (`-[r:T]->`), deletable by `DELETE r`, guarded by bare `DELETE n`, // and not collapsed with a parallel edge. The annotation blank node is // freshened per WHERE solution (SPARQL §3.1.3), so batched edge inserts @@ -1758,7 +1758,7 @@ impl<'a> CypherLowering<'a> { } None => self.fresh_bnode(), }; - self.emit_reifier_bundle(&ann, &s, &type_sid, &o)?; + self.emit_reifier_link(&ann, &s, &type_sid, &o); if let Some(props) = &rel.props { self.emit_property_triples(&ann, props)?; } @@ -1766,33 +1766,24 @@ impl<'a> CypherLowering<'a> { Ok(()) } - fn emit_reifier_bundle( + /// `ann rdf:reifies <<( s p o )>>`. + fn emit_reifier_link( &mut self, ann: &TemplateTerm, s: &TemplateTerm, p_sid: &Sid, o: &TemplateTerm, - ) -> Result<(), LowerCypherError> { - let subj_pred = self.ns.sid_for_iri(reifies_iris::SUBJECT); - let pred_pred = self.ns.sid_for_iri(reifies_iris::PREDICATE); - let obj_pred = self.ns.sid_for_iri(reifies_iris::OBJECT); - - self.insert_templates.push(TripleTemplate::new( - ann.clone(), - TemplateTerm::Sid(subj_pred), - s.clone(), - )); + ) { self.insert_templates.push(TripleTemplate::new( ann.clone(), - TemplateTerm::Sid(pred_pred), - TemplateTerm::Sid(p_sid.clone()), - )); - self.insert_templates.push(TripleTemplate::new( - ann.clone(), - TemplateTerm::Sid(obj_pred), - o.clone(), + TemplateTerm::Sid(fluree_db_core::rdf_reifies_sid().clone()), + TemplateTerm::TripleTerm(Box::new(TemplateTripleTerm { + s: s.clone(), + p: TemplateTerm::Sid(p_sid.clone()), + o: o.clone(), + dtc: None, + })), )); - Ok(()) } fn emit_property_triples( diff --git a/fluree-db-transact/src/lower_sparql_update.rs b/fluree-db-transact/src/lower_sparql_update.rs index d1eae2d916..20947970ff 100644 --- a/fluree-db-transact/src/lower_sparql_update.rs +++ b/fluree-db-transact/src/lower_sparql_update.rs @@ -51,14 +51,14 @@ use fluree_db_sparql::ast::{ GraphPattern, GraphRefAll, GraphTransfer, Iri, IriValue, Literal, LiteralValue as SparqlLiteralValue, Load, Modify, PredicateTerm, Prologue, PropertyPath, QuadData, QuadPattern, QuadPatternElement, QueryBody, ReifierId, SparqlAst, SubjectTerm, Term, - TriplePattern, UpdateOperation, + TriplePattern, TripleTerm as SparqlTripleTerm, UpdateOperation, }; use fluree_db_sparql::SourceSpan; use thiserror::Error; use crate::ir::{ GraphMgmtOp, GraphSel, GraphTarget, SparqlWhereClause, TemplateGraph, TemplateTerm, - TripleTemplate, Txn, TxnOpts, TxnType, + TemplateTripleTerm, TripleTemplate, Txn, TxnOpts, TxnType, }; use crate::namespace::NamespaceRegistry; use fluree_vocab::{fluree, xsd}; @@ -273,9 +273,9 @@ fn anon_in_mode_msg(op: &'static str) -> &'static str { } /// Expand any annotated triples in a Vec into the equivalent set of -/// unannotated triples: the base triple, the `f:reifies*` bundle -/// (subject/predicate/object only — graph/datatype/lang/listIndex are -/// derived at flake time), and the body's predicate-object pairs. +/// unannotated triples, as RDF 1.2 defines the annotation syntax: the base +/// triple, `reifier rdf:reifies <<( s p o )>>` per annotation, and the body's +/// predicate-object pairs. /// /// Default-graph only in v1; an annotation tail inside a `GRAPH` block /// is rejected by the caller before this is invoked. @@ -284,8 +284,6 @@ fn expand_annotated_triples( mode: AnnotationExpansionMode, bnodes: &mut BlankNodeCounter, ) -> Result<(), LowerError> { - use fluree_vocab::reifies_iris; - let original = std::mem::take(triples); let mut out: Vec = Vec::with_capacity(original.len()); @@ -295,14 +293,7 @@ fn expand_annotated_triples( continue; }; - // Reject RDF-star quoted-triple subjects explicitly. The - // legacy `<< s p o >>` quoted-triple form has no compatible - // representation in the f:reifies* bundle (the base triple - // would need to embed inside f:reifiesSubject's object slot - // which violates the bundle shape), and `subject_to_object` - // below would otherwise hit its `unreachable!()` panic. - // Surface this as an explicit `UnsupportedFeature` so the - // user sees a real error rather than a transactor panic. + // A reified triple as the annotated triple's subject is deferred. if let SubjectTerm::QuotedTriple(qt) = &tp.subject { return Err(LowerError::UnsupportedFeature { feature: "RDF-star quoted-triple subject combined with an RDF 1.2 \ @@ -311,10 +302,6 @@ fn expand_annotated_triples( }); } - // Same for an RDF 1.2 triple-term subject (`<<( s p o )>>`, - // accept-then-defer, D-1): this expansion pre-pass runs BEFORE the - // quad-pattern lowering whose TripleTerm arms defer cleanly, and - // `subject_to_object` below would hit its `unreachable!()` panic. if let SubjectTerm::TripleTerm(tt) = &tp.subject { return Err(LowerError::UnsupportedFeature { feature: "SPARQL 1.2 triple-term subject combined with an RDF 1.2 \ @@ -323,10 +310,9 @@ fn expand_annotated_triples( }); } - // Reify the base edge and emit base + per-unit bundle + body. // The base triple stripped of its annotation goes through - // unchanged; each annotation unit (`~ r? {| … |}?`) contributes - // its own reifier bundle. + // unchanged; each annotation unit (`~ r? {| … |}?`) contributes its + // own reifier. let span = tp.span; // Base triple (without annotation) @@ -340,54 +326,18 @@ fn expand_annotated_triples( for unit in &annotation.units { let reifier = resolve_reifier(unit, mode, bnodes)?; - // f:reifies* bundle: SUBJECT, PREDICATE, OBJECT, and (for a - // language-tagged object) LANG. f:reifiesGraph is omitted - // (default graph only) — WITH-scoped templates are rejected - // upstream by `reject_with_scoped_annotations` so this default - // identity never gets graph-stamped. f:reifiesDatatype rides on - // the f:reifiesObject flake's flake-level dt (the decoder derives - // it), and f:reifiesListIndex is deferred (v1). - let pred_iri = - |s: &'static str| -> PredicateTerm { PredicateTerm::Iri(Iri::full(s, span)) }; out.push(TriplePattern::new( reifier.clone(), - pred_iri(reifies_iris::SUBJECT), - subject_to_object(&tp.subject), - span, - )); - out.push(TriplePattern::new( - reifier.clone(), - pred_iri(reifies_iris::PREDICATE), - predicate_to_object(&tp.predicate), - span, - )); - out.push(TriplePattern::new( - reifier.clone(), - pred_iri(reifies_iris::OBJECT), - tp.object.clone(), + PredicateTerm::Iri(Iri::full(fluree_vocab::rdf::REIFIES, span)), + Term::TripleTerm(Box::new(SparqlTripleTerm { + subject: tp.subject.clone(), + predicate: tp.predicate.clone(), + object: tp.object.clone(), + span, + })), span, )); - // f:reifiesLang — required for a language-tagged object. - // `EdgeKey::from_reifies_facts` reads `lang` from a dedicated - // f:reifiesLang flake, NOT from the f:reifiesObject flake's - // `m.lang`. Without this triple the decoded EdgeKey carries - // `lang = None` while the base edge's EdgeKey carries - // `lang = Some(tag)`, so the forward-map lookup misses: the - // annotation silently vanishes from `@annotation` hydration - // and the bundle is never cascaded on base-edge retract. - // Mirrors the JSON-LD writer (`build_annotation_sibling`). - if let Term::Literal(lit) = &tp.object { - if let SparqlLiteralValue::LangTagged { lang, .. } = &lit.value { - out.push(TriplePattern::new( - reifier.clone(), - pred_iri(reifies_iris::LANG), - Term::Literal(Literal::string(lang.as_ref(), span)), - span, - )); - } - } - // Body entries become (reifier, ann_pred, ann_obj) triples. // Property-path verbs (legal in query annotation blocks) // have no template meaning — reject with a clear error. @@ -418,33 +368,6 @@ fn expand_annotated_triples( Ok(()) } -/// Convert a SPARQL subject term into the corresponding object term so -/// the `f:reifiesSubject` pointer can carry it. Subjects and objects -/// share the IRI / blank-node / variable cases; literals never appear -/// as subjects so the case is unreachable in practice. -fn subject_to_object(s: &SubjectTerm) -> Term { - match s { - SubjectTerm::Var(v) => Term::Var(v.clone()), - SubjectTerm::Iri(i) => Term::Iri(i.clone()), - SubjectTerm::BlankNode(b) => Term::BlankNode(b.clone()), - SubjectTerm::QuotedTriple(_) => { - unreachable!("RDF-star quoted triples are rejected before annotation expansion") - } - SubjectTerm::TripleTerm(_) => { - unreachable!("SPARQL 1.2 triple-term values are rejected before annotation expansion") - } - } -} - -/// Convert a predicate (IRI or var) into the object slot for -/// `f:reifiesPredicate`. -fn predicate_to_object(p: &PredicateTerm) -> Term { - match p { - PredicateTerm::Var(v) => Term::Var(v.clone()), - PredicateTerm::Iri(i) => Term::Iri(i.clone()), - } -} - /// Walk the QuadPatternElement list and expand every annotated triple /// in-place. Annotation tails inside a GRAPH block are rejected with a /// "deferred to a follow-up" message so the v1 default-graph contract @@ -557,9 +480,8 @@ fn reject_user_authored_reifies( for tp in triples { check_predicate(&tp.predicate, prologue)?; - // `rdf:reifies` names a triple term; the reified-triple object forms - // lower to the bundle, so any other object is a data error the link - // lowering would read as a link. + // `rdf:reifies` names a triple term; any other object is a data + // error the link lowering would read as a link. if let PredicateTerm::Iri(iri) = &tp.predicate { if expand_iri(iri, prologue)? == fluree_vocab::rdf::REIFIES && !matches!( @@ -665,16 +587,9 @@ fn reject_user_authored_reifies_in_quad_pattern( Ok(()) } -/// Reject RDF 1.2 annotation tails on `WITH `-scoped template triples. -/// -/// `WITH ` re-homes default-position template triples into `` *after* -/// annotation expansion, but the v1 expansion omits `f:reifiesGraph` — the -/// synthetic bundle encodes a default-graph edge identity. Stamping the WITH -/// graph id over that bundle would mint graph-tagged reifications whose edge -/// identity is still default-graph, so the forward-map lookup misses: the -/// annotation never hydrates and never cascades on base-edge retract. Reject -/// until graph-aware expansion (emitting `f:reifiesGraph`) lands. Annotation -/// tails inside explicit `GRAPH { ... }` blocks are already rejected by +/// Reject RDF 1.2 annotation tails on `WITH `-scoped template triples: +/// SPARQL UPDATE annotations are default-graph only for now. Annotation tails +/// inside explicit `GRAPH { ... }` blocks are rejected by /// [`expand_annotated_triples_in_quad_pattern`]; this covers the top-level /// (WITH-scoped) triples it would otherwise expand as default-graph. fn reject_with_scoped_annotations(pattern: &QuadPattern) -> Result<(), LowerError> { @@ -1187,15 +1102,35 @@ fn lower_delete_where( for tp in triples { // WHERE side: lower to UnresolvedPattern::Triple with bnodes rewritten as vars let s = subject_to_unresolved_delete_where(&tp.subject, prologue, &mut bnode_vars)?; - let p = predicate_to_unresolved(&tp.predicate, prologue)?; - let obj = object_to_unresolved_delete_where(&tp.object, prologue, &mut bnode_vars)?; - - where_patterns.push(UnresolvedPattern::Triple(UnresolvedTriplePattern { - s, - p, - o: obj.term, - dtc: obj.dtc, - })); + match &tp.object { + Term::TripleTerm(tt) if is_rdf_reifies(&tp.predicate, prologue)? => { + let edge_s = + subject_to_unresolved_delete_where(&tt.subject, prologue, &mut bnode_vars)?; + let edge_p = predicate_to_unresolved(&tt.predicate, prologue)?; + let edge_o = + object_to_unresolved_delete_where(&tt.object, prologue, &mut bnode_vars)?; + where_patterns.push(UnresolvedPattern::AnnotationTarget { + annotation: s, + edge: UnresolvedTriplePattern { + s: edge_s, + p: edge_p, + o: edge_o.term, + dtc: edge_o.dtc, + }, + body: Vec::new(), + }); + } + object => { + let p = predicate_to_unresolved(&tp.predicate, prologue)?; + let obj = object_to_unresolved_delete_where(object, prologue, &mut bnode_vars)?; + where_patterns.push(UnresolvedPattern::Triple(UnresolvedTriplePattern { + s, + p, + o: obj.term, + dtc: obj.dtc, + })); + } + } // DELETE side: lower to TripleTemplate with the same bnode->var mapping delete_templates.push(lower_triple_to_delete_template_delete_where( @@ -1255,7 +1190,7 @@ fn lower_delete_where_with_graphs( with_graph_iri: None, using_default_graph_iris: Vec::new(), using_named_graph_iris: Vec::new(), - pattern: quad_pattern_to_graph_pattern(&rewritten), + pattern: quad_pattern_to_graph_pattern(&rewritten, prologue)?, }; let mut write_graphs = BTreeSet::new(); @@ -1323,15 +1258,21 @@ fn rewrite_blank_nodes_to_vars(pattern: &QuadPattern) -> QuadPattern { let mut rewrite_triple = |tp: &TriplePattern| -> TriplePattern { let mut out = tp.clone(); - if let SubjectTerm::BlankNode(bn) = &tp.subject { - if let Some(v) = rewrite_bnode(bn) { - out.subject = SubjectTerm::Var(v); + let mut rewrite = |subject: &mut SubjectTerm, object: &mut Term| { + if let SubjectTerm::BlankNode(bn) = subject { + if let Some(v) = rewrite_bnode(bn) { + *subject = SubjectTerm::Var(v); + } } - } - if let Term::BlankNode(bn) = &tp.object { - if let Some(v) = rewrite_bnode(bn) { - out.object = Term::Var(v); + if let Term::BlankNode(bn) = object { + if let Some(v) = rewrite_bnode(bn) { + *object = Term::Var(v); + } } + }; + rewrite(&mut out.subject, &mut out.object); + if let Term::TripleTerm(tt) = &mut out.object { + rewrite(&mut tt.subject, &mut tt.object); } out }; @@ -1363,7 +1304,10 @@ fn rewrite_blank_nodes_to_vars(pattern: &QuadPattern) -> QuadPattern { /// Runs of default-graph triples become one BGP; each `GRAPH |?g { ... }` /// block becomes a `GraphPattern::Graph` wrapping its own BGP. Source order is /// preserved so bindings join exactly as the user wrote them. -fn quad_pattern_to_graph_pattern(pattern: &QuadPattern) -> GraphPattern { +fn quad_pattern_to_graph_pattern( + pattern: &QuadPattern, + prologue: &Prologue, +) -> Result { let span = pattern.span; let mut parts: Vec = Vec::new(); let mut bgp: Vec = Vec::new(); @@ -1376,30 +1320,25 @@ fn quad_pattern_to_graph_pattern(pattern: &QuadPattern) -> GraphPattern { triples, span: g_span, } => { - if !bgp.is_empty() { - parts.push(GraphPattern::Bgp { - patterns: std::mem::take(&mut bgp), - span, - }); - } + parts.extend(triples_to_graph_patterns( + std::mem::take(&mut bgp), + prologue, + span, + )?); + let inner = triples_to_graph_patterns(triples.clone(), prologue, *g_span)?; parts.push(GraphPattern::Graph { name: name.clone(), - pattern: Box::new(GraphPattern::Bgp { - patterns: triples.clone(), - span: *g_span, - }), + pattern: Box::new(group(inner, *g_span)), span: *g_span, }); } } } - if !bgp.is_empty() { - parts.push(GraphPattern::Bgp { - patterns: bgp, - span, - }); - } + parts.extend(triples_to_graph_patterns(bgp, prologue, span)?); + Ok(group(parts, span)) +} +fn group(mut parts: Vec, span: SourceSpan) -> GraphPattern { if parts.len() == 1 { parts.pop().expect("len checked") } else { @@ -1410,6 +1349,45 @@ fn quad_pattern_to_graph_pattern(pattern: &QuadPattern) -> GraphPattern { } } +/// Triples as WHERE patterns: runs of ordinary triples as BGPs, and each +/// `r rdf:reifies <<( s p o )>>` as the reifier pattern the query parser +/// builds for it. +fn triples_to_graph_patterns( + triples: Vec, + prologue: &Prologue, + span: SourceSpan, +) -> Result, LowerError> { + let mut parts = Vec::new(); + let mut bgp = Vec::new(); + for tp in triples { + let reifies = is_rdf_reifies(&tp.predicate, prologue)?; + match tp.object { + Term::TripleTerm(triple_term) if reifies => { + if !bgp.is_empty() { + parts.push(GraphPattern::Bgp { + patterns: std::mem::take(&mut bgp), + span, + }); + } + parts.push(GraphPattern::AnnotationTarget { + reifier: tp.subject, + predicate: tp.predicate, + triple_term, + span: tp.span, + }); + } + object => bgp.push(TriplePattern { object, ..tp }), + } + } + if !bgp.is_empty() { + parts.push(GraphPattern::Bgp { + patterns: bgp, + span, + }); + } + Ok(parts) +} + /// Lower Modify operation (DELETE/INSERT with WHERE). /// /// The most general update form with optional WITH, DELETE, INSERT, and WHERE clauses. @@ -1605,6 +1583,19 @@ fn lower_triple_to_template( let result = literal_to_template(lit, prologue, ns)?; (result.term, result.dtc) } + Term::TripleTerm(tt) if is_rdf_reifies(&triple.predicate, prologue)? => { + let s = subject_to_template(&tt.subject, prologue, ns, vars, bnodes)?; + let p = predicate_to_template(&tt.predicate, prologue, ns, vars)?; + let (o, dtc) = match &tt.object { + Term::Literal(lit) => { + let result = literal_to_template(lit, prologue, ns)?; + (result.term, result.dtc) + } + other => (object_to_template(other, prologue, ns, vars, bnodes)?, None), + }; + let term = TemplateTripleTerm { s, p, o, dtc }; + (TemplateTerm::TripleTerm(Box::new(term)), None) + } other => (object_to_template(other, prologue, ns, vars, bnodes)?, None), }; @@ -1619,6 +1610,14 @@ fn lower_triple_to_template( }) } +/// True when `p` is `rdf:reifies`, whose object is a triple term. +fn is_rdf_reifies(p: &PredicateTerm, prologue: &Prologue) -> Result { + Ok(match p { + PredicateTerm::Iri(iri) => expand_iri(iri, prologue)? == fluree_vocab::rdf::REIFIES, + PredicateTerm::Var(_) => false, + }) +} + // ============================================================================= // Term conversion for WHERE patterns (UnresolvedTerm) // ============================================================================= @@ -1712,8 +1711,40 @@ fn lower_triple_to_delete_template_delete_where( vars: &mut VarRegistry, bnodes: &mut BlankNodeVarNamer, ) -> Result { - // Subject - let subject = match &triple.subject { + let subject = delete_where_subject_template(&triple.subject, prologue, ns, vars, bnodes)?; + let predicate = delete_where_predicate_template(&triple.predicate, prologue, ns, vars)?; + let (object, dtc) = match &triple.object { + Term::TripleTerm(tt) if is_rdf_reifies(&triple.predicate, prologue)? => { + let s = delete_where_subject_template(&tt.subject, prologue, ns, vars, bnodes)?; + let p = delete_where_predicate_template(&tt.predicate, prologue, ns, vars)?; + let (o, dtc) = delete_where_object_template(&tt.object, prologue, ns, vars, bnodes)?; + let term = TemplateTripleTerm { s, p, o, dtc }; + (TemplateTerm::TripleTerm(Box::new(term)), None) + } + other => delete_where_object_template(other, prologue, ns, vars, bnodes)?, + }; + + Ok(TripleTemplate { + subject, + predicate, + object, + dtc, + list_index: None, + graph: TemplateGraph::Default, + graph_from_template_default: false, + }) +} + +/// A DELETE WHERE subject as a template term, a blank node lowered to the +/// variable its WHERE side binds. +fn delete_where_subject_template( + term: &SubjectTerm, + prologue: &Prologue, + ns: &mut NamespaceRegistry, + vars: &mut VarRegistry, + bnodes: &mut BlankNodeVarNamer, +) -> Result { + Ok(match term { SubjectTerm::Var(v) => TemplateTerm::Var(vars.get_or_insert(&format!("?{}", v.name))), SubjectTerm::Iri(iri) => { let expanded = expand_iri(iri, prologue)?; @@ -1746,19 +1777,33 @@ fn lower_triple_to_delete_template_delete_where( span: tt.span, }); } - }; + }) +} - // Predicate - let predicate = match &triple.predicate { +fn delete_where_predicate_template( + term: &PredicateTerm, + prologue: &Prologue, + ns: &mut NamespaceRegistry, + vars: &mut VarRegistry, +) -> Result { + Ok(match term { PredicateTerm::Var(v) => TemplateTerm::Var(vars.get_or_insert(&format!("?{}", v.name))), PredicateTerm::Iri(iri) => { let expanded = expand_iri(iri, prologue)?; TemplateTerm::Sid(ns.sid_for_iri(&expanded)) } - }; + }) +} - // Object + datatype constraint (for literals) - let (object, dtc) = match &triple.object { +/// A DELETE WHERE object as a template term and its datatype constraint. +fn delete_where_object_template( + term: &Term, + prologue: &Prologue, + ns: &mut NamespaceRegistry, + vars: &mut VarRegistry, + bnodes: &mut BlankNodeVarNamer, +) -> Result<(TemplateTerm, Option), LowerError> { + Ok(match term { Term::Var(v) => ( TemplateTerm::Var(vars.get_or_insert(&format!("?{}", v.name))), None, @@ -1797,16 +1842,6 @@ fn lower_triple_to_delete_template_delete_where( span: tt.span, }); } - }; - - Ok(TripleTemplate { - subject, - predicate, - object, - dtc, - list_index: None, - graph: TemplateGraph::Default, - graph_from_template_default: false, }) } diff --git a/fluree-db-transact/src/parse/edge_annotations.rs b/fluree-db-transact/src/parse/edge_annotations.rs index 053a711657..910e11047a 100644 --- a/fluree-db-transact/src/parse/edge_annotations.rs +++ b/fluree-db-transact/src/parse/edge_annotations.rs @@ -2,11 +2,13 @@ //! parser. //! //! Walks the raw transaction document **before** JSON-LD expansion and -//! rewrites every `@annotation` / `@edge` / `@reifies` block into the -//! seven-fact `f:reifies*` system encoding. The output is a document -//! that contains only ordinary IRIs (no `@`-keyword extensions), so -//! the rest of the parsing pipeline (`expand_with_context_policy`, -//! `parse_expanded_triples_with_ctx`) processes it unchanged. +//! rewrites every `@annotation` / `@edge` / `@reifies` block into +//! `f:reifies*` slot keys naming the annotated edge. JSON-LD has no +//! triple-term syntax, so the slots are how the edge travels through +//! `expand_with_context_policy` and `parse_expanded_triples_with_ctx` +//! unchanged; [`fold_slots_into_links`] then turns each annotation's slots +//! into its `rdf:reifies <<( s p o )>>` link, and bulk import's sink does +//! the same. The slots are never stored. //! //! The accepted insert shape is the **inline form** (`@annotation` / //! `@edge` on the *object* node of a predicate): @@ -1138,39 +1140,17 @@ fn build_annotation_delete( // body — same deferral rule as inserts. scan_annotation_keywords_in_map(&ann_map, ctx)?; - // Build the WHERE pattern as a flat triple-pattern node. The - // body properties (remaining in `ann_map`) act as selector - // predicates; the `f:reifies*` triples pin the annotation to - // the (parent_subject, predicate, object_id) edge. We emit - // the system predicates directly rather than the higher-level - // `@reifies` shape because the standard lowering walker - // rejects `@reifies` outside its query-side context, while - // `f:reifies*` IRIs are accepted as ordinary IRIs (the - // user-authored-reifies firewall has already run against the - // original doc, so our synthesized ones aren't re-scanned). - // The JSON-LD-Q query parser still resolves `f:reifies*` - // triple patterns into the same indexed lookups as - // `@reifies` would. + // The WHERE node: the body properties (remaining in `ann_map`) act + // as selector predicates, and the query-side `@reifies` pins the + // annotation to the (parent_subject, predicate, object) edge through + // its `rdf:reifies` link. let mut where_node = ann_map.clone(); where_node.insert("@id".to_string(), Value::String(var.clone())); - where_node.insert( - reifies_iris::SUBJECT.to_string(), - json!({"@id": parent_subject}), - ); - where_node.insert( - reifies_iris::PREDICATE.to_string(), - json!({"@id": predicate}), - ); - where_node.insert(reifies_iris::OBJECT.to_string(), object_payload.clone()); - if let Some(lang) = &lang_payload { - // Emit `f:reifiesLang` as an additional WHERE constraint - // so the selector form binds only annotations whose - // language tag matches — same lexical string across - // different languages must not collide. - where_node.insert(reifies_iris::LANG.to_string(), json!(lang)); - } + let mut edge = Map::new(); + edge.insert("@id".to_string(), json!(parent_subject)); + edge.insert(predicate.to_string(), object_payload.clone()); + where_node.insert(REIFIES_KEY.to_string(), Value::Object(edge)); if let Some(graph) = graph_iri { - where_node.insert(reifies_iris::GRAPH.to_string(), json!({"@id": graph})); // Named-graph case: wrap the node in the JLDQ s-expression // graph form `["graph", "", { ...patterns... }]` so // the WHERE evaluation scopes its triple matches to the @@ -2081,6 +2061,112 @@ fn intercept_annotations_for_predicate( } } +/// Replace each annotation's lowered `f:reifies*` slot templates with its +/// `rdf:reifies` link. The slots are this module's intermediate form: they +/// carry the edge through JSON-LD expansion, which has no triple-term syntax. +/// Each annotation sets each slot once, in its own node, so a slot set again +/// for the same reifier starts that reifier's next link. +pub(crate) fn fold_slots_into_links(templates: &mut Vec) -> Result<()> { + use crate::ir::{TemplateGraph, TemplateTerm, TemplateTripleTerm, TripleTemplate}; + use fluree_db_core::{DatatypeConstraint, FlakeValue}; + use fluree_vocab::db; + + #[derive(Default)] + struct Slots { + s: Option, + p: Option, + o: Option<(TemplateTerm, Option)>, + lang: Option, + } + impl Slots { + fn has(&self, slot: &str) -> bool { + match slot { + db::REIFIES_SUBJECT => self.s.is_some(), + db::REIFIES_PREDICATE => self.p.is_some(), + db::REIFIES_OBJECT => self.o.is_some(), + db::REIFIES_LANG => self.lang.is_some(), + _ => false, + } + } + } + let key = |t: &TripleTemplate| -> Option<(TemplateGraph, String)> { + let subject = match &t.subject { + TemplateTerm::Var(v) => format!("?{v:?}"), + TemplateTerm::Sid(sid) => format!("<{sid}>"), + TemplateTerm::BlankNode(label) => label.clone(), + _ => return None, + }; + Some((t.graph.clone(), subject)) + }; + + let mut reifiers: Vec<((TemplateGraph, String), TripleTemplate, Slots)> = Vec::new(); + let mut kept = Vec::with_capacity(templates.len()); + for t in templates.drain(..) { + let slot = match &t.predicate { + TemplateTerm::Sid(p) if fluree_db_core::is_reserved_reifies_predicate(p) => { + p.name.to_string() + } + _ => { + kept.push(t); + continue; + } + }; + let k = key(&t).ok_or_else(|| { + TransactError::Parse("an annotation's reifier must be a node".to_string()) + })?; + let open = reifiers + .iter() + .rposition(|(rk, _, slots)| *rk == k && !slots.has(&slot)); + let at = match open { + Some(at) => at, + None => { + reifiers.push((k, t.clone(), Slots::default())); + reifiers.len() - 1 + } + }; + let slots = &mut reifiers[at].2; + match slot.as_str() { + db::REIFIES_SUBJECT => slots.s = Some(t.object), + db::REIFIES_PREDICATE => slots.p = Some(t.object), + db::REIFIES_OBJECT => slots.o = Some((t.object, t.dtc)), + db::REIFIES_LANG => { + if let TemplateTerm::Value(FlakeValue::String(lang)) = t.object { + slots.lang = Some(lang); + } + } + // The link lives in its template's graph and names the object's + // datatype itself. + db::REIFIES_GRAPH | db::REIFIES_DATATYPE => {} + other => { + return Err(TransactError::UnsupportedFeature(format!( + "f:{other} is not supported on an annotation" + ))) + } + } + } + + for (_, first, slots) in reifiers { + let (Some(s), Some(p), Some((o, dtc))) = (slots.s, slots.p, slots.o) else { + return Err(TransactError::Parse( + "an annotation must name its triple's subject, predicate and object".to_string(), + )); + }; + let dtc = match slots.lang { + Some(lang) => Some(DatatypeConstraint::LangTag(std::sync::Arc::from(lang))), + None => dtc, + }; + kept.push(TripleTemplate { + predicate: TemplateTerm::Sid(fluree_db_core::rdf_reifies_sid().clone()), + object: TemplateTerm::TripleTerm(Box::new(TemplateTripleTerm { s, p, o, dtc })), + dtc: None, + list_index: None, + ..first + }); + } + *templates = kept; + Ok(()) +} + #[cfg(test)] mod tests { use super::*; @@ -3001,8 +3087,8 @@ mod tests { let lowered = lower_delete(doc).unwrap(); let wn = &wheres(&lowered)[0]; assert_eq!( - wn.get(reifies_iris::OBJECT).unwrap(), - &json!({"@value": "Alice"}) + wn.get(REIFIES_KEY).unwrap(), + &json!({"@id": "ex:alice", "ex:name": {"@value": "Alice"}}) ); let t = &templates(&lowered)[0]; assert_eq!( @@ -3014,7 +3100,7 @@ mod tests { } #[test] - fn delete_by_selector_on_lang_tagged_literal_emits_lang_in_where_and_template() { + fn delete_by_selector_on_lang_tagged_literal_keeps_lang_in_where_and_template() { let doc = json!({ "delete": { "@id": "ex:alice", @@ -3027,14 +3113,12 @@ mod tests { }); let lowered = lower_delete(doc).unwrap(); let wn = &wheres(&lowered)[0]; + // The WHERE selector's `@reifies` keeps the language tag, so the + // same lexical string in another language does not bind. assert_eq!( - wn.get(reifies_iris::OBJECT).unwrap(), - &json!({"@value": "chat", "@language": "fr"}) + wn.get(REIFIES_KEY).unwrap(), + &json!({"@id": "ex:alice", "ex:label": {"@value": "chat", "@language": "fr"}}) ); - // f:reifiesLang on the WHERE selector pins the join to the - // right language tag — same lexical string in another - // language must not bind. - assert_eq!(wn.get(reifies_iris::LANG).unwrap(), &json!("fr")); let t = &templates(&lowered)[0]; assert_eq!(t.get(reifies_iris::LANG).unwrap(), &json!("fr")); } diff --git a/fluree-db-transact/src/parse/jsonld.rs b/fluree-db-transact/src/parse/jsonld.rs index a7eea86ca3..98c2782f72 100644 --- a/fluree-db-transact/src/parse/jsonld.rs +++ b/fluree-db-transact/src/parse/jsonld.rs @@ -159,11 +159,16 @@ pub fn parse_transaction( std::borrow::Cow::Borrowed(json) }; - match txn_type { + let mut txn = match txn_type { TxnType::Insert => parse_insert(&lowered, opts, ns_registry), TxnType::Upsert => parse_upsert(&lowered, opts, ns_registry), TxnType::Update => parse_update(&lowered, opts, ns_registry), + }?; + if matches!(lowered, std::borrow::Cow::Owned(_)) { + super::edge_annotations::fold_slots_into_links(&mut txn.insert_templates)?; + super::edge_annotations::fold_slots_into_links(&mut txn.delete_templates)?; } + Ok(txn) } /// Parse a graph-sync transaction (see [`Txn::sync_graph`]). @@ -224,33 +229,6 @@ pub fn parse_graph_insert( for t in &mut txn.insert_templates { t.graph = TemplateGraph::Iri(Arc::clone(&target)); } - // Edge annotations were lowered against a payload with no graph identity, - // so their `f:reifies*` bundles carry no `f:reifiesGraph`. Re-homing the - // bundle into the target graph without one produces a bundle whose - // flake-level graph disagrees with the edge graph it encodes - // (`EdgeKey::from_reifies_facts` → `GraphMismatch`, refused at stage). - // Anchor every reifier to the target graph, exactly as the named-`@graph` - // lowering does for an annotated edge written inside a graph block. - let reifies_subject = fluree_db_core::Sid::new( - fluree_vocab::namespaces::FLUREE_DB, - fluree_vocab::db::REIFIES_SUBJECT, - ); - let reifies_graph = fluree_db_core::Sid::new( - fluree_vocab::namespaces::FLUREE_DB, - fluree_vocab::db::REIFIES_GRAPH, - ); - let anchors: Vec = txn - .insert_templates - .iter() - .filter(|t| matches!(&t.predicate, TemplateTerm::Sid(p) if *p == reifies_subject)) - .map(|t| { - let mut anchor = t.clone(); - anchor.predicate = TemplateTerm::Sid(reifies_graph.clone()); - anchor.object = TemplateTerm::Sid(ns_registry.sid_for_iri(graph_iri)); - anchor - }) - .collect(); - txn.insert_templates.extend(anchors); txn.write_graphs.insert(graph_iri.clone()); Ok(txn) } diff --git a/fluree-db-transact/src/stage.rs b/fluree-db-transact/src/stage.rs index f9374c3b12..ca24e70672 100644 --- a/fluree-db-transact/src/stage.rs +++ b/fluree-db-transact/src/stage.rs @@ -51,44 +51,20 @@ use fluree_db_shacl::{ShaclCache, ShaclEngine, ValidationReport}; /// Given `graph_sids` (ledger GraphId → Sid), returns the /// inverse mapping. Used by SHACL/policy to determine which graph a flake /// belongs to based on its `Flake.g` field. -/// Generate cascade `f:reifies*` retraction flakes for any base edges -/// being retracted in `flakes`. +/// Retract the annotations a transaction's retracts leave dangling, so a link +/// never names a triple that is no longer asserted (annotation-syntax reads +/// rely on it and skip the base edge). /// -/// For each retract flake whose predicate is *not* a system-controlled -/// `f:reifies*` predicate, build the corresponding `EdgeKey` and look -/// up its currently-asserted annotations against the merged -/// snapshot+novelty view. For each annotation, emit the inverse -/// `f:reifies*` bundle so the durable encoding doesn't keep pointing -/// at a retracted edge. +/// 1. A retracted triple retracts every link naming it. +/// 2. Retracting all of a reifier's body retracts its links. +/// 3. A reifier left with no link loses its body when it is a blank node, or +/// in LPG mode (`opts.lpgEdgeLifecycle`), where deleting a relationship +/// deletes its properties. An IRI reifier's body otherwise stays, as +/// ordinary RDF about a named resource. /// -/// **Read path:** uses scan-based lookup through -/// `range_with_overlay`, which reads novelty + indexed base storage, -/// so annotations that have rolled into base storage post-reindex -/// are still found and cascaded. -/// -/// **Graph context:** the inverse bundle is emitted via -/// `EdgeKey::to_reifies_facts_jsonld_compatible`, which carries the -/// edge's `g` through to each retract flake. Named-graph assertions -/// are retracted in the same named graph; default-graph assertions -/// stay in the default graph. Without this graph-aware emission, -/// named-graph annotations would be orphaned by mismatched-graph -/// retracts. -/// -/// **Performance:** a ledger that has never observed a `f:reifies*` -/// flake pays zero — the dual-gate fast-path below returns immediately -/// (see the gate on `snapshot.has_annotations` + `novelty.attachments -/// .has_annotations()`). Once a ledger does carry annotations, each -/// retract flake pays one POST point lookup against the binary index; -/// the lookup returns empty quickly when nothing matches, so the -/// per-retract cost is bounded by the binary index's empty-range cost. -/// -/// **Cleanup scope:** retracts the `f:reifies*` bundle only. -/// Anonymous-annotation metadata cleanup (RDF mode default) and -/// explicit-IRI metadata cleanup (LPG mode `opts.lpgEdgeLifecycle: -/// true`) are separate cleanup passes — the bundle-only cascade is -/// observably safe (anonymous SIDs become unreachable through -/// `@reifies` once the bundle is gone, and explicit-IRI annotation -/// metadata stays queryable as ordinary RDF). +/// Links derived from legacy `f:reifies*` bundles read like any other link; +/// retracting one leaves the bundle, whose derived assert the retract cancels +/// in novelty and at the next index build alike. async fn cascade_attachment_retracts( flakes: &[Flake], ledger: &LedgerState, @@ -97,34 +73,15 @@ async fn cascade_attachment_retracts( lpg_edge_lifecycle: bool, ) -> Result> { use fluree_db_core::comparator::IndexType; - use fluree_db_core::edge::EdgeKey; use fluree_db_core::range::{RangeMatch, RangeOptions, RangeTest}; - use fluree_db_core::{is_reserved_reifies_predicate, FlakeValue}; - + use fluree_db_core::{is_annotation_predicate, is_rdf_reifies, FlakeValue, TripleTermValue}; use std::collections::BTreeMap; - use tracing::Instrument; let mut cascade = Vec::new(); - - // Fast-path gate: if the ledger has *never* observed a - // `f:reifies*` flake (sticky bit on `IndexRoot`, plumbed through - // `LedgerSnapshot.has_annotations`) AND the in-memory novelty - // hasn't observed one in this run either, skip the cascade - // entirely. Non-annotation ledgers pay zero per-retract cost. - // - // Both gates are required because the indexed bit only updates at - // index-build time — a ledger that just received its first - // annotation in novelty (no reindex yet) has - // `snapshot.has_annotations == false` but - // `novelty.attachments.has_annotations() == true`. - if !ledger.snapshot.has_annotations && !ledger.novelty.attachments.has_annotations() { + if !ledger.snapshot.has_annotations && !ledger.novelty.has_annotations() { return Ok(cascade); } - // Wrap the cascade work in a span — skipped for non-annotation - // ledgers via the gate above so we don't pay tracing cost on - // common-case retracts. `cascade_count` records how many - // f:reifies* retract flakes the cascade emitted. let span = tracing::debug_span!( "cascade_reifies_bundle", retract_input_count = flakes.iter().filter(|f| !f.op).count(), @@ -132,389 +89,166 @@ async fn cascade_attachment_retracts( cascade_count = tracing::field::Empty, ); async { - let f_reifies_subject = Sid::new( - fluree_vocab::namespaces::FLUREE_DB, - fluree_vocab::db::REIFIES_SUBJECT, - ); let to_t = ledger.t(); + let scan = |g_id: GraphId, index: IndexType, rm: RangeMatch| { + fluree_db_core::range_with_overlay( + &ledger.snapshot, + g_id, + ledger.novelty.as_ref(), + index, + RangeTest::Eq, + rm, + RangeOptions::new().with_to_t(to_t), + ) + }; + let reifies = fluree_db_core::rdf_reifies_sid().clone(); + let retract = |f: &Flake, g: Option<&Sid>| { + let mut r = f.clone(); + r.g = g.cloned(); + r.t = new_t; + r.op = false; + r + }; - // Annotation subjects already cascaded in pass 1 (base-edge - // retract). Pass 2 (orphan-cleanup) skips these so the - // `f:reifies*` bundle isn't double-retracted. - let mut cascaded_anns: HashSet<(GraphId, Sid)> = HashSet::new(); + // This transaction's own link ops, by (graph, reifier). + let mut txn_link_retracts: HashSet<(GraphId, Sid, FlakeValue)> = HashSet::new(); + let mut txn_linked: HashSet<(GraphId, Sid)> = HashSet::new(); + let mut explicit: BTreeMap<(GraphId, Sid), Option> = BTreeMap::new(); + for f in flakes.iter().filter(|f| is_rdf_reifies(&f.p)) { + let g_id = resolve_flake_graph_id(f, reverse_graph)?; + if f.op { + txn_linked.insert((g_id, f.s.clone())); + } else { + txn_link_retracts.insert((g_id, f.s.clone(), f.o.clone())); + explicit.insert((g_id, f.s.clone()), f.g.clone()); + } + } - // Reifiers this transaction is *re-pointing*: it asserts at least one - // `f:reifies*` fact for them, so they describe some edge after this - // transaction and are not orphaned by the base-edge retract below. - // - // The cascade must leave their bundles alone, because the delta that - // reached this point is already complete and already minimal. Sync and - // upsert both hand the accumulator the current state as retractions - // and the payload as assertions, and matched pairs cancel — so a slot - // whose value does not change (typically `f:reifiesPredicate` and - // `f:reifiesGraph`) appears in neither list. Cascading the bundle here - // would retract exactly those unchanged slots while nothing re-asserts - // them, and the surviving bundle would be missing a slot: a re-point - // that is entirely well-formed was refused with - // `Missing("f:reifiesPredicate")`, and the advice in that error - // ("retract the prior attachment in the same transaction") described - // what the caller had already done. - let repointed: HashSet = flakes - .iter() - .filter(|f| f.op && is_reserved_reifies_predicate(&f.p)) - .map(|f| f.s.clone()) - .collect(); + // Links this cascade retracts, by (graph, reifier), with the graph + // they are retracted in. + let mut unlinked: BTreeMap<(GraphId, Sid), (Option, Vec)> = + BTreeMap::new(); + // 1. Retracted triples. for flake in flakes { - if flake.op { - continue; // assertion, not a retract — nothing to cascade + if flake.op || is_annotation_predicate(&flake.p) { + continue; } - if is_reserved_reifies_predicate(&flake.p) { - continue; // already a system retract (cascade or upstream); skip + // A list element is never reified. + if flake.m.as_ref().is_some_and(|m| m.i.is_some()) { + continue; } - let edge_key = EdgeKey::from_flake(flake); let g_id = resolve_flake_graph_id(flake, reverse_graph)?; - - // POST scan: find every annotation whose `f:reifiesSubject` - // points at this edge's subject. The flake's `s` is the - // candidate annotation Sid. - let candidates = fluree_db_core::range_with_overlay( - &ledger.snapshot, + let term = FlakeValue::TripleTerm(Box::new(TripleTermValue { + s: flake.s.clone(), + p: flake.p.clone(), + o: flake.o.clone(), + dt: flake.dt.clone(), + lang: flake.m.as_ref().and_then(|m| m.lang.clone()), + })); + let links = scan( g_id, - ledger.novelty.as_ref(), IndexType::Post, - RangeTest::Eq, RangeMatch::new() - .with_predicate(f_reifies_subject.clone()) - .with_object(FlakeValue::Ref(edge_key.s.clone())), - RangeOptions::new().with_to_t(to_t), + .with_predicate(reifies.clone()) + .with_object(term.clone()), ) .await?; - - let mut seen: HashSet = HashSet::new(); - for cand in &candidates { - let ann_sid = cand.s.clone(); - if !seen.insert(ann_sid.clone()) { - continue; - } - if repointed.contains(&ann_sid) { - // This transaction re-points the reifier; its bundle is - // the transaction's to rewrite, not the cascade's to - // retract. `enforce_single_target_reifiers` still checks - // the result, so a genuinely malformed re-point is caught. - continue; - } - - // SPOT scan for ALL of the candidate's flakes (system - // bundle + body metadata). Splitting after the scan - // lets us reuse the same flake set for bundle-validation - // and (for anonymous annotations) metadata cleanup. - let all_ann_flakes = fluree_db_core::range_with_overlay( - &ledger.snapshot, - g_id, - ledger.novelty.as_ref(), - IndexType::Spot, - RangeTest::Eq, - RangeMatch::new().with_subject(ann_sid.clone()), - RangeOptions::new().with_to_t(to_t), - ) - .await?; - // `from_reifies_facts` reconciles the bundle's `f:reifiesGraph` - // value against the flake-level `g`, so an indexed named-graph - // bundle decoded as `GraphMismatch` and the `Err(_) => continue` - // below swallowed it: deleting a base edge left the claim that - // reifies it live, pointing at a triple that no longer exists. - let mut all_ann_flakes = all_ann_flakes; - stamp_graph(&mut all_ann_flakes, flake.g.as_ref()); - // The index-side `rdf:reifies` link is neither bundle nor body. - let (bundle, metadata): (Vec, Vec) = all_ann_flakes - .into_iter() - .filter(|f| !fluree_db_core::is_rdf_reifies(&f.p)) - .partition(|f| is_reserved_reifies_predicate(&f.p)); - if bundle.is_empty() { - continue; - } - let cand_edge = match EdgeKey::from_reifies_facts(&bundle) { - Ok(k) => k, - Err(_) => continue, - }; - if cand_edge != edge_key { + for link in links { + if txn_link_retracts.contains(&(g_id, link.s.clone(), term.clone())) { continue; } - - // Build the f:reifies* retraction bundle. JSON-LD- - // compatible shape (no f:reifiesDatatype) so the - // inverse retract is byte-symmetric with the original - // assertion. - cascade.extend(edge_key.to_reifies_facts_jsonld_compatible(&ann_sid, new_t, false)); - - // Annotation-body metadata cleanup. Two modes: - // - // - **RDF mode** (default, `lpg_edge_lifecycle = false`): - // only anonymous (blank-node) annotation subjects - // are cleaned. Explicit-IRI annotations are - // user-named and remain queryable as ordinary RDF - // on the named subject. - // - **LPG mode** (`lpg_edge_lifecycle = true`, set via - // `opts.lpgEdgeLifecycle: true` on the transaction): - // *every* annotation's body is cleaned, matching - // Cypher's relationship lifecycle where deleting an - // edge deletes its property metadata. - let cleanup_metadata = lpg_edge_lifecycle - || ann_sid.namespace_code == fluree_vocab::namespaces::BLANK_NODE; - if cleanup_metadata { - for asserted in metadata { - // Mirror the asserted shape with `op = false` - // and the new transaction's `t`. Preserves - // graph (`g`), datatype, and metadata so the - // retract matches the assertion's identity. - let mut retract = asserted.clone(); - retract.t = new_t; - retract.op = false; - cascade.push(retract); - } - } - // Track this annotation as already-cascaded so the - // orphan-cleanup pass below doesn't double-retract. - cascaded_anns.insert((g_id, ann_sid)); + cascade.push(retract(&link, flake.g.as_ref())); + let entry = unlinked + .entry((g_id, link.s)) + .or_insert_with(|| (flake.g.clone(), Vec::new())); + entry.1.push(term.clone()); } } - // --------------------------------------------------------------- - // Pass 2: orphan-cleanup for annotation-metadata-only retracts. - // - // When a transaction retracts all the user-property metadata of - // an annotation subject WITHOUT retracting the base edge, the - // base-edge pass above doesn't fire — but the f:reifies* bundle - // is now an orphan in the durable encoding, since the - // annotation has no body left. An inline `@annotation` query - // would still surface the empty annotation subject, which is - // not what the user intended. - // - // Detection: for every retract flake whose subject is an - // annotation subject (has currently-asserted f:reifies* - // facts), check if the user's retract set covers ALL of that - // subject's currently-asserted user-property metadata. If so, - // also retract the bundle. - // - // Both RDF and LPG modes apply this cleanup — the user-intent - // signal (retract every metadata flake) is the same either way. - // The mode flag only controls whether explicit-IRI metadata - // gets retracted from a base-edge cascade. - // Group both retract and assert flakes by `(g_id, subject)`, - // skipping `f:reifies*` predicates. Pass 2 keys off the retract - // groups (only retracts can trigger orphan cleanup) but - // consults the assert groups when computing the post-transaction - // metadata set — otherwise an in-transaction metadata - // replacement (delete old role, insert new role on the same - // annotation subject) would cascade the bundle and orphan the - // freshly-asserted metadata. - let mut grouped_metadata_retracts: BTreeMap<(GraphId, Sid), Vec<&Flake>> = BTreeMap::new(); - let mut grouped_metadata_asserts: BTreeMap<(GraphId, Sid), Vec<&Flake>> = BTreeMap::new(); - for flake in flakes { - if is_reserved_reifies_predicate(&flake.p) { - continue; - } - let g_id = resolve_flake_graph_id(flake, reverse_graph)?; - let key = (g_id, flake.s.clone()); - if flake.op { - grouped_metadata_asserts.entry(key).or_default().push(flake); + // 2. Reifiers whose whole body this transaction retracts. Same-txn + // asserts count toward what survives, so replacing a body value + // keeps the reifier. + type FlakeIdentity = (Sid, FlakeValue, Sid, Option); + let identity = + |f: &Flake| -> FlakeIdentity { (f.p.clone(), f.o.clone(), f.dt.clone(), f.m.clone()) }; + let mut body_retracts: BTreeMap<(GraphId, Sid), (Option, HashSet)> = + BTreeMap::new(); + let mut body_asserts: HashSet<(GraphId, Sid)> = HashSet::new(); + for f in flakes.iter().filter(|f| !is_annotation_predicate(&f.p)) { + let key = (resolve_flake_graph_id(f, reverse_graph)?, f.s.clone()); + if f.op { + body_asserts.insert(key); } else { - grouped_metadata_retracts + body_retracts .entry(key) - .or_default() - .push(flake); + .or_insert_with(|| (f.g.clone(), HashSet::new())) + .1 + .insert(identity(f)); } } - - for ((g_id, ann_sid), retract_set) in grouped_metadata_retracts { - if cascaded_anns.contains(&(g_id, ann_sid.clone())) { - // Already cascaded via the base-edge pass; skip. + for ((g_id, ann), (g_sid, retracted)) in body_retracts { + if body_asserts.contains(&(g_id, ann.clone())) { continue; } - - // Scan all currently-asserted flakes for this subject. - let all_flakes = fluree_db_core::range_with_overlay( - &ledger.snapshot, + let current = scan( g_id, - ledger.novelty.as_ref(), IndexType::Spot, - RangeTest::Eq, - RangeMatch::new().with_subject(ann_sid.clone()), - RangeOptions::new().with_to_t(to_t), + RangeMatch::new().with_subject(ann.clone()), ) .await?; - // Same seam as the base-edge pass above: stamp before decoding. - // The group's own retracts are this transaction's flakes for this - // subject, so they carry the graph the scan dropped. - let mut all_flakes = all_flakes; - stamp_graph( - &mut all_flakes, - retract_set.first().and_then(|f| f.g.as_ref()), - ); - let (bundle, current_metadata): (Vec, Vec) = all_flakes + let (links, body): (Vec, Vec) = current .into_iter() - .filter(|f| !fluree_db_core::is_rdf_reifies(&f.p)) - .partition(|f| is_reserved_reifies_predicate(&f.p)); - if bundle.is_empty() { - continue; // not an annotation subject - } - let edge_key = match EdgeKey::from_reifies_facts(&bundle) { - Ok(k) => k, - Err(_) => continue, // malformed; skip - }; - - // Compute the post-transaction metadata set: - // (current metadata - retracts in this txn) ∪ asserts in this - // txn. Cascade only when post == ∅. This handles the - // metadata-replacement case (delete old fact + insert new - // fact on the same annotation in one transaction): the new - // assertion keeps the post-set non-empty, so the bundle - // stays asserted and the new metadata stays attached. - // - // Identity = `(s, p, o, dt, m)` — the same dimensions - // Fluree uses for assertion/retraction matching, so an - // assertion with a different object value than the - // retracted flake counts as a distinct fact and survives. - type FlakeIdentity = (Sid, Sid, FlakeValue, Sid, Option); - let identity = |f: &Flake| -> FlakeIdentity { - ( - f.s.clone(), - f.p.clone(), - f.o.clone(), - f.dt.clone(), - f.m.clone(), - ) - }; - let retracted_ids: HashSet = - retract_set.iter().map(|f| identity(f)).collect(); - let asserted_ids: HashSet = grouped_metadata_asserts - .get(&(g_id, ann_sid.clone())) - .map(|v| v.iter().map(|f| identity(f)).collect::>()) - .unwrap_or_default(); - - // Surviving current metadata: present today AND not being - // retracted. Plus any new same-txn assertions (which may - // overlap with surviving current; HashSet handles the - // dedupe for free). - let post_metadata: HashSet = current_metadata - .iter() - .map(identity) - .filter(|id| !retracted_ids.contains(id)) - .chain(asserted_ids) - .collect(); - if !post_metadata.is_empty() { + .filter(|f| !fluree_db_core::is_reserved_reifies_predicate(&f.p)) + .partition(|f| is_rdf_reifies(&f.p)); + if links.is_empty() || body.iter().any(|f| !retracted.contains(&identity(f))) { continue; } - - // Annotation has no surviving metadata after the txn → the - // user is disposing of it. Retract the bundle in the - // JSON-LD-compatible shape so it cancels the original - // assertion. - cascade.extend(edge_key.to_reifies_facts_jsonld_compatible(&ann_sid, new_t, false)); - } - - // --------------------------------------------------------------- - // Pass 3: body cleanup for user-explicit bundle retracts. - // - // The by-id retract pre-pass - // (`fluree_db_transact::parse::edge_annotations::lower_delete_annotation_blocks`) - // synthesizes only `f:reifies*` retracts — the bundle, no body. - // Without this pass, the annotation's body metadata persists - // after the bundle is gone. That matches the design contract - // for explicit-IRI annotations in default RDF mode - // (user-named resources stay queryable as ordinary RDF) but - // breaks two cases: - // - // 1. **Anonymous (BLANK_NODE) annotations:** body becomes - // orphaned bnode-keyed flakes that the wildcard-hide filter - // no longer catches (the `f:reifiesSubject` discriminator - // is gone with the bundle). Always cleanup. - // 2. **Explicit-IRI annotations in LPG mode** - // (`opts.lpgEdgeLifecycle: true`): Cypher relationship-delete - // semantics demand that retracting the relationship deletes - // its property metadata too. Cleanup only when the flag is - // set. - // - // Detection: group the user-input retracts by `(g, ann_sid)`, - // keeping only `f:reifies*` flakes. A "complete bundle retract" - // has at least the three required slots (subject + predicate + - // object) present. Partial retracts are user errors we don't - // try to correct. - let mut grouped_bundle_retracts: BTreeMap<(GraphId, Sid), Vec<&Flake>> = BTreeMap::new(); - for flake in flakes { - if flake.op { - continue; // assertion, not a retract - } - if !is_reserved_reifies_predicate(&flake.p) { - continue; // not a bundle flake + for link in links { + let already = unlinked + .get(&(g_id, ann.clone())) + .is_some_and(|(_, terms)| terms.contains(&link.o)); + if already || txn_link_retracts.contains(&(g_id, ann.clone(), link.o.clone())) { + continue; + } + cascade.push(retract(&link, g_sid.as_ref())); + unlinked + .entry((g_id, ann.clone())) + .or_insert_with(|| (g_sid.clone(), Vec::new())) + .1 + .push(link.o); } - let g_id = resolve_flake_graph_id(flake, reverse_graph)?; - grouped_bundle_retracts - .entry((g_id, flake.s.clone())) - .or_default() - .push(flake); } - for ((g_id, ann_sid), bundle_retracts) in grouped_bundle_retracts { - if cascaded_anns.contains(&(g_id, ann_sid.clone())) { - // Already handled by Pass 1's base-edge cascade. - continue; - } - - // Verify all three required slots are retracted at the same - // t. The pre-pass emits exactly those three (plus - // `f:reifiesGraph` for named-graph annotations); a partial - // user-issued retract isn't a "complete bundle retract" - // signal. - let has_required = [ - fluree_vocab::db::REIFIES_SUBJECT, - fluree_vocab::db::REIFIES_PREDICATE, - fluree_vocab::db::REIFIES_OBJECT, - ] - .iter() - .all(|name| { - bundle_retracts.iter().any(|f| { - f.p.namespace_code == fluree_vocab::namespaces::FLUREE_DB - && f.p.name.as_ref() == *name - }) - }); - if !has_required { - continue; - } - - let is_anonymous = ann_sid.namespace_code == fluree_vocab::namespaces::BLANK_NODE; - let cleanup_body = is_anonymous || lpg_edge_lifecycle; - if !cleanup_body { + // 3. Bodies of reifiers left with no link. + for (key, g_sid) in explicit { + unlinked.entry(key).or_insert((g_sid, Vec::new())); + } + for ((g_id, ann), (g_sid, terms)) in unlinked { + let anonymous = ann.namespace_code == fluree_vocab::namespaces::BLANK_NODE; + if !(anonymous || lpg_edge_lifecycle) || txn_linked.contains(&(g_id, ann.clone())) { continue; } - - // Scan currently-asserted flakes for this annotation - // subject. Filter out the bundle (Pass 1's domain) and - // emit retracts mirroring each body assertion. - let all_flakes = fluree_db_core::range_with_overlay( - &ledger.snapshot, + let current = scan( g_id, - ledger.novelty.as_ref(), IndexType::Spot, - RangeTest::Eq, - RangeMatch::new().with_subject(ann_sid.clone()), - RangeOptions::new().with_to_t(to_t), + RangeMatch::new().with_subject(ann.clone()), ) .await?; - for asserted in all_flakes { - // Bundle flakes are retracted by the caller; the index-side - // link is retired by the next index pass, not by a commit. - if is_reserved_reifies_predicate(&asserted.p) - || fluree_db_core::is_rdf_reifies(&asserted.p) - { - continue; - } - let mut retract = asserted.clone(); - retract.t = new_t; - retract.op = false; - cascade.push(retract); + let keeps_link = current.iter().any(|f| { + is_rdf_reifies(&f.p) + && !terms.contains(&f.o) + && !txn_link_retracts.contains(&(g_id, ann.clone(), f.o.clone())) + }); + if keeps_link { + continue; } + let body: Vec = current + .iter() + .filter(|f| !is_annotation_predicate(&f.p)) + .map(|f| retract(f, g_sid.as_ref())) + .collect(); + cascade.extend(body); } tracing::Span::current().record("cascade_count", cascade.len()); @@ -574,20 +308,6 @@ pub struct StageOptions<'a> { /// /// The normal `stage()` path builds this internally from `txn.write_graphs`. pub graph_sids: Option<&'a HashMap>, - - /// These flakes come from a commit that was already authored and written, - /// not from a transaction being authored now. - /// - /// Authoring invariants become advisory: a violation is logged rather than - /// refused. The single-target reifier rule is one. It reached - /// `stage_flakes` with this work, which put it on the push path — and - /// `insert_turtle` and the permissive bulk-import sink did not enforce it - /// before, so a commit written by an older build can hold a reifier on two - /// edges. Refusing that on push would strand the ledger permanently, with - /// no way forward short of rewriting history, to prevent data that is - /// already written. Authoring still refuses it, which is where refusing - /// can still change the outcome. - pub replaying_commit: bool, } impl<'a> StageOptions<'a> { @@ -619,13 +339,6 @@ impl<'a> StageOptions<'a> { self.graph_sids = Some(graph_sids); self } - - /// Mark these flakes as the replay of an already-authored commit, making - /// authoring invariants advisory. See [`StageOptions::replaying_commit`]. - pub fn replaying_commit(mut self) -> Self { - self.replaying_commit = true; - self - } } /// Stage a transaction against a ledger @@ -1084,9 +797,6 @@ pub async fn stage_with_graph_delta( } } - // Stage-time attachment-bundle invariant (shared with `stage_flakes`). - enforce_single_target_reifiers(&ledger, &flakes, &reverse_graph).await?; - // Charge 1 micro-fuel per staged flake. Matches query-side scan fuel, // which also charges per flake without filtering schema flakes. // Fuel exhaustion returns an error so transactions exceeding fuel @@ -1372,8 +1082,8 @@ async fn scan_graph_flakes( return Err(TransactError::WholeGraphScanTooLarge { limit: l }); } } - // Links are derived from the bundles this rewrites, never written. - flakes.retain(|f| !fluree_db_core::is_rdf_reifies(&f.p)); + // A legacy attachment bundle is inert: its link reads and moves as a link. + flakes.retain(|f| !fluree_db_core::is_reserved_reifies_predicate(&f.p)); for f in &mut flakes { f.g = g_sid.cloned(); } @@ -1621,129 +1331,18 @@ async fn stage_graph_mgmt( let dest_contents: HashSet = dest_flakes.iter().map(flake_content).collect(); - // O5: re-homing rewrites the `f:reifiesGraph` anchor per src/ - // dest graph (see the assert loop below), so an - // annotation-bearing graph now transfers cleanly instead of - // orphaning the anchor. One case is not repairable by - // rewriting, though: ADD (`clear_dest == false`) merges the - // source bundle into the destination without retracting dest's - // existing bundles. If the SAME explicit reifier `@id` reifies - // a DIFFERENT edge in the source and the destination, the merged - // bundle carries two `f:reifiesSubject` flakes on one subject; - // `EdgeKey::from_reifies_facts` rejects that as `Duplicate`, so - // both readers (JSON-LD hydration + attachment indexer) silently - // drop BOTH annotations — including dest's previously-valid one. - // COPY/MOVE are immune (`clear_dest` retracts the colliding dest - // bundle before the source bundle is asserted). Blank-minted - // reifiers never collide, so this only trips when a user reuses - // an explicit reifier `@id` across the two graphs. Fail loud. - if !*clear_dest { - // The edge a reifier subject denotes, keyed by subject Sid: - // the set of its `f:reifies*` flake contents EXCLUDING - // `f:reifiesGraph` (the only graph-dependent flake). Same - // subject + different signature == a merge that would - // produce a `Duplicate`; same signature (same edge) dedups - // cleanly against `dest_contents` and is fine. - let edge_signatures = - |bundle: &[Flake]| -> HashMap> { - let mut sigs: HashMap> = HashMap::new(); - for f in bundle { - if fluree_db_core::is_reserved_reifies_predicate(&f.p) - && !fluree_db_core::is_reifies_graph(&f.p) - { - sigs.entry(f.s.clone()) - .or_default() - .insert(flake_content(f)); - } - } - sigs - }; - let dest_sigs = edge_signatures(&dest_flakes); - if !dest_sigs.is_empty() { - let src_sigs = edge_signatures(&src_flakes); - for (subject, src_sig) in &src_sigs { - if let Some(dest_sig) = dest_sigs.get(subject) { - if src_sig != dest_sig { - let iri = ns_registry - .get_prefix(subject.namespace_code) - .map(|prefix| format!("{}{}", prefix, subject.name)) - .unwrap_or_else(|| subject.to_string()); - return Err(TransactError::UnsupportedFeature(format!( - "ADD would merge edge annotations that share \ - reifier <{iri}> but reify different edges in the \ - source and destination graphs, which silently \ - drops both annotations. Use COPY/MOVE, or give the \ - annotations distinct reifier @ids." - ))); - } - } - } - } - } - // Build the source assertions, re-homed into the destination - // graph. Every flake's `g` becomes `dest_sid`; the - // `f:reifiesGraph` anchor additionally encodes the edge's graph - // in its OBJECT, so it needs a reification-aware rewrite rather - // than the generic `g`-only one: - // - named→named : rewrite the anchor OBJECT to the dest graph - // too, so the decoded `g` == flake-level `g` == dest. - // - named→default: DROP the anchor (the default graph carries - // none; the decoder wants `g == None`). - // - default→named: the source carried no anchor, so SYNTHESIZE - // one per reifier subject (keyed off the exactly-one - // `f:reifiesSubject` flake per bundle). - let src_is_default = matches!(from, GraphSel::Default); - // default→named synthesizes ONE anchor per reifier subject - // (below), so size for them exactly; other directions add - // nothing beyond the re-homed source flakes. - let synthesized = if src_is_default { - src_flakes - .iter() - .filter(|f| fluree_db_core::is_reifies_subject(&f.p)) - .count() - } else { - 0 - }; - let mut rehomed: Vec = - Vec::with_capacity(src_flakes.len() + synthesized); - for f in &src_flakes { - if fluree_db_core::is_reifies_graph(&f.p) { - if let Some(dest) = &dest_sid { - let mut a = f.clone(); - a.op = true; - a.t = new_t; - a.g = Some(dest.clone()); - a.o = FlakeValue::Ref(dest.clone()); - rehomed.push(a); - } - // dest default: drop the anchor (emit nothing). - continue; - } - let mut a = f.clone(); - a.op = true; - a.t = new_t; - a.g = dest_sid.clone(); - rehomed.push(a); - } - if src_is_default { - if let Some(dest) = &dest_sid { - for f in &src_flakes { - if fluree_db_core::is_reifies_subject(&f.p) { - rehomed.push(Flake::new_in_graph( - dest.clone(), - f.s.clone(), - fluree_db_core::namespaces::reifies_graph_sid().clone(), - FlakeValue::Ref(dest.clone()), - fluree_db_core::id_datatype_sid(), - new_t, - true, - None, - )); - } - } - } - } + // graph. + let rehomed: Vec = src_flakes + .iter() + .map(|f| { + let mut a = f.clone(); + a.op = true; + a.t = new_t; + a.g = dest_sid.clone(); + a + }) + .collect(); // The content set the transfer will land in the destination — // keyed on the RE-HOMED flakes so COPY/MOVE stay symmetric: a @@ -1842,278 +1441,6 @@ async fn stage_graph_mgmt( .await } -/// Which of `candidates` could already have rows in `ledger`. -/// -/// A subject absent from both the persisted subject dictionary and its -/// graph's novelty provably has no prior facts, so a scan for it can only -/// come back empty. Skipping that scan matters because it is not cheap when -/// it misses: a `Sid` neither dictionary can resolve falls through -/// `binary_range` into `overlay_only_flakes`, which walks the graph's -/// **entire** novelty. Freshly minted reifiers — every anonymous `{| … |}` -/// and `<< … >>` — are exactly that shape, so a bulk Turtle-star insert -/// otherwise pays one whole-novelty walk per annotation in the file. -/// -/// `None` means absence cannot be decided here (no binary store on an -/// already-indexed ledger); callers must then treat every candidate as -/// possibly present and run their scan, so a real load failure surfaces -/// rather than silently skipping work. -/// -/// The novelty walk runs at most once per graph and answers for every -/// candidate in that graph at once. `generate_upsert_deletions` skips absent -/// subjects the same way and for the same reason, and #1657 fixed the same -/// shape in `binary_scan`. -fn subjects_with_prior_rows( - ledger: &LedgerState, - candidates: &[(GraphId, Sid)], -) -> Option> { - use fluree_db_core::comparator::IndexType; - use fluree_db_query::BinaryRangeProvider; - - let binary_store = ledger - .snapshot - .range_provider - .as_ref() - .and_then(|rp| rp.as_any().downcast_ref::()) - .map(|brp| Arc::clone(brp.store())); - - // Genesis with nothing indexed: novelty is the only place a subject can - // be, so it decides on its own. An indexed ledger whose store failed to - // load must stay undecidable — see `generate_upsert_deletions`. - let base_index_absent = ledger.snapshot.range_provider.is_none() && ledger.snapshot.t == 0; - if binary_store.is_none() && !base_index_absent { - return None; - } - - let mut present: HashSet<(GraphId, Sid)> = HashSet::new(); - let mut want_novelty: HashMap> = HashMap::new(); - - for (g_id, sid) in candidates { - let in_base = match binary_store.as_deref() { - None => false, - Some(store) => { - if matches!( - store.find_subject_id_by_parts(sid.namespace_code, &sid.name), - Ok(Some(_)) - ) { - true - } else { - // A namespace code the pre-transaction snapshot cannot - // decode was minted by this transaction, so it names no - // base row. Same reasoning as the upsert skip. - match ledger.snapshot.decode_sid(sid) { - Some(iri) => !matches!(store.find_subject_id(&iri), Ok(None)), - None => false, - } - } - } - }; - if in_base { - present.insert((*g_id, sid.clone())); - } else { - want_novelty.entry(*g_id).or_default().insert(sid); - } - } - - for (g_id, subjects) in want_novelty { - ledger.novelty.for_each_overlay_flake( - g_id, - IndexType::Spot, - None, - None, - true, - ledger.t(), - &mut |flake| { - if subjects.contains(&flake.s) { - present.insert((g_id, flake.s.clone())); - } - }, - ); - } - - Some(present) -} - -/// Re-attach the graph to flakes read back through a scan. -/// -/// Index-decoded flakes carry `g: None` — the graph is the index they came -/// from, not a field on the flake — while a named-graph transaction's own -/// flakes carry `g: Some(sid)`. Any comparison or decode that reads `g` has to -/// put it back first, or an indexed named-graph bundle silently fails to line -/// up with the transaction that is editing it. `scan_graph_flakes` has always -/// done this; every other scan of a reifier's own facts needs it too. -fn stamp_graph(flakes: &mut [Flake], g_sid: Option<&Sid>) { - for f in flakes { - f.g = g_sid.cloned(); - } -} - -/// Stage-time attachment-bundle invariant: an annotation SID may -/// reify exactly one edge. Counting this txn's asserted -/// `f:reifiesSubject` flakes is insufficient — it misses -/// (a) re-pointing an `@id` already attached to a *different* -/// edge in a prior transaction (no retract in this txn), and -/// (b) same-subject / different-slot multiplicity within one txn -/// (the subject slot dedupes while the predicate/object slots -/// diverge). Validate the *net* asserted bundle per touched -/// annotation SID — current snapshot/novelty state, minus this -/// txn's retracts, plus its asserts — by decoding it the way the -/// arena / hydration paths will. A malformed (multi-target) net -/// bundle is rejected here rather than corrupting downstream -/// `EdgeKey::from_reifies_facts`. -/// -/// Runs on every staging entry point — `stage` (JSON-LD / SPARQL) and -/// `stage_flakes` (the Turtle sink and push/import paths) — so a reifier -/// reused on two edges is refused no matter which surface wrote it. -async fn enforce_single_target_reifiers( - ledger: &LedgerState, - flakes: &[Flake], - reverse_graph: &HashMap, -) -> Result<()> { - use fluree_db_core::comparator::IndexType; - use fluree_db_core::edge::EdgeKey; - use fluree_db_core::is_reserved_reifies_predicate; - use fluree_db_core::range::{RangeMatch, RangeOptions, RangeTest}; - - // One pass over `flakes`, grouping every reserved-predicate flake under - // its annotation subject. The per-reifier work below then reads only its - // own group: re-scanning `flakes` per reifier made the check - // O(reifiers x flakes), and since each annotation contributes ~5 flakes - // the flake count grows with the reifier count — quadratic on exactly the - // shape this check exists for. `stage_flakes` puts it on the bulk Turtle - // insert and commit-apply paths, where one transaction can carry every - // reifier in a file. - // - // `touched` keeps the "gate on asserts" property: a reifier is only work - // when the txn ASSERTS one of its facts, so pure retracts (which can only - // shrink a bundle) and non-annotation transactions stay at zero scan cost. - // - // The key is `(graph, reifier)`, not the reifier alone. A bundle names its - // own graph in `f:reifiesGraph`, and `EdgeKey::from_reifies_facts` checks - // that a bundle is graph-uniform before it checks anything else, so - // folding one reifier's flakes from two graphs into one bundle reports - // `MixedFlakeGraphs` — surfaced here as "multi-target", which it is not. - // The same reifier in two graphs is a state graph management produces on - // purpose: `COPY TO ` duplicates annotated edges, reifier IRIs - // included, and `add_same_edge_same_reifier_succeeds` pins that it must - // work. Keying by subject alone refused within one transaction exactly - // what two transactions were free to do. - let mut by_reifier: HashMap<(GraphId, &Sid), Vec<&Flake>> = HashMap::new(); - let mut touched: Vec<(GraphId, &Sid)> = Vec::new(); - for f in flakes { - if !is_reserved_reifies_predicate(&f.p) { - continue; - } - let key = (resolve_flake_graph_id(f, reverse_graph)?, &f.s); - let group = by_reifier.entry(key).or_default(); - if f.op && !group.iter().any(|g| g.op) { - touched.push(key); - } - group.push(f); - } - - if !touched.is_empty() { - let to_t = ledger.t(); - // Set key for a `f:reifies*` fact: graph + subject + - // predicate + object + datatype. Keying as a set gives - // RDF set-semantics, so an idempotent re-assert of an - // existing attachment collapses instead of looking like - // a duplicate slot, while genuinely divergent slots - // (two different edges) remain distinct and trip - // `EdgeKey::from_reifies_facts`'s `Duplicate` check. - type ReifiesKey = (Option, Sid, Sid, FlakeValue, Sid); - let reifies_key = |f: &Flake| -> ReifiesKey { - ( - f.g.clone(), - f.s.clone(), - f.p.clone(), - f.o.clone(), - f.dt.clone(), - ) - }; - - // Resolve each reifier's graph once, then decide in a single pass - // which of them can have a prior bundle at all. Without this the scan - // below ran per reifier, and a reifier this transaction just minted - // resolves in neither dictionary, so each one walked the graph's whole - // novelty: O(new reifiers x novelty) on exactly the bulk Turtle-star - // insert this check exists to guard. - let mut anchors: Vec<((GraphId, &Sid), Option)> = Vec::with_capacity(touched.len()); - for key in &touched { - let anchor = by_reifier[key] - .iter() - .find(|f| f.op) - .expect("touched implies an assert"); - anchors.push((*key, anchor.g.clone())); - } - let candidates: Vec<(GraphId, Sid)> = anchors - .iter() - .map(|((g_id, sid), _)| (*g_id, (*sid).clone())) - .collect(); - let may_have_prior = subjects_with_prior_rows(ledger, &candidates); - - for ((g_id, ann_sid), g_sid) in anchors { - let mine = &by_reifier[&(g_id, ann_sid)]; - - // Current asserted `f:reifies*` bundle for this SID - // (pre-txn snapshot + novelty), as a deduped set. A reifier with - // no prior rows contributes nothing, so the net bundle is exactly - // what this transaction asserts. - let mut current = match &may_have_prior { - Some(present) if !present.contains(&(g_id, ann_sid.clone())) => Vec::new(), - _ => { - fluree_db_core::range_with_overlay( - &ledger.snapshot, - g_id, - ledger.novelty.as_ref(), - IndexType::Spot, - RangeTest::Eq, - RangeMatch::new().with_subject((*ann_sid).clone()), - RangeOptions::new().with_to_t(to_t), - ) - .await? - } - }; - // Index-decoded flakes carry `g: None` — the graph is the index - // they came from, not a field — while this txn's named-graph - // flakes carry `g: Some(sid)`. `reifies_key` includes `g`, so - // without this stamp an indexed named-graph bundle never matches - // the txn's keys: a re-assert would not collapse, and the net - // bundle would mix `None` and `Some` and decode as - // `MixedFlakeGraphs`. `scan_graph_flakes` stamps for the same - // reason. - stamp_graph(&mut current, g_sid.as_ref()); - let mut net: HashMap = HashMap::new(); - for f in current { - if is_reserved_reifies_predicate(&f.p) { - net.insert(reifies_key(&f), f); - } - } - - // Fold this txn's effects for the SID: retracts drop - // the matching fact; asserts add it. - for f in mine { - let key = reifies_key(f); - if f.op { - net.insert(key, (*f).clone()); - } else { - net.remove(&key); - } - } - - let net_bundle: Vec = net.into_values().collect(); - if let Err(e) = EdgeKey::from_reifies_facts(&net_bundle) { - return Err(TransactError::InvariantViolation(format!( - "annotation subject `{ann_sid}` would reify a malformed or \ - multi-target edge after this transaction ({e:?}); an annotation \ - may reify exactly one edge. Retract the prior attachment in the \ - same transaction if you intended to re-point it.", - ))); - } - } - } - Ok(()) -} - /// Stage pre-built flakes against a ledger (bypass WHERE/template pipeline). /// /// This is the fast path for bulk INSERT from Turtle where flakes are already @@ -2172,24 +1499,6 @@ pub async fn stage_flakes( } }; - // 3. Stage-time attachment-bundle invariant: one reifier, one edge. - // Advisory when replaying a commit that was authored elsewhere — - // see `StageOptions::replaying_commit` for why refusing there would - // strand a ledger rather than prevent anything. - match enforce_single_target_reifiers(&ledger, &flakes, &reverse_graph).await { - Ok(()) => {} - Err(e) if options.replaying_commit => { - tracing::warn!( - error = %e, - "replayed commit violates the single-target reifier invariant; \ - applying it anyway because the commit is already authored. \ - Its attachment bundles will not decode, so the annotations \ - involved will not be queryable" - ); - } - Err(e) => return Err(e), - } - // 4. Charge 1 micro-fuel per staged flake. if let Some(tracker) = options.tracker { tracker.consume_fuel(flakes.len() as u64)?; @@ -2493,8 +1802,8 @@ async fn classify_subject_lifecycle( delta: &SubjectDelta, pre_classes: &[Sid], ) -> Result { - // A reifier's link is derived and never retracted by a transaction, so - // it would make every full delete of a reifier look partial. + // A legacy attachment bundle is never retracted by a transaction, so it + // would make every full delete of a reifier look partial. let scan_pre_state = || async move { let rm = fluree_db_core::RangeMatch::new().with_subject(subject.clone()); let opts = fluree_db_core::RangeOptions::new().with_to_t(ledger.t()); @@ -2509,7 +1818,7 @@ async fn classify_subject_lifecycle( ) .await .map(|mut flakes| { - flakes.retain(|f| !fluree_db_core::is_rdf_reifies(&f.p)); + flakes.retain(|f| !fluree_db_core::is_reserved_reifies_predicate(&f.p)); flakes }) }; diff --git a/fluree-db-transact/tests/it_lower_cypher.rs b/fluree-db-transact/tests/it_lower_cypher.rs index ed2643eedb..ce2d068b87 100644 --- a/fluree-db-transact/tests/it_lower_cypher.rs +++ b/fluree-db-transact/tests/it_lower_cypher.rs @@ -46,13 +46,13 @@ fn create_node_with_properties() { } #[test] -fn create_directed_relationship_emits_base_and_reifier_bundle() { +fn create_directed_relationship_emits_base_and_reifier_link() { let txn = lower(r#"CREATE (a:Person {name: "Alice"})-[:KNOWS]->(b:Person {name: "Bob"})"#); // Every Cypher relationship reifies (LPG identity): - // 2 labels + 2 props + 1 base edge + 3 reifier bundle triples = 8 templates. + // 2 labels + 2 props + 1 base edge + 1 reifier link = 6 templates. assert_eq!( txn.insert_templates.len(), - 8, + 6, "templates: {:?}", txn.insert_templates ); @@ -96,10 +96,10 @@ fn optional_match_before_create_is_rejected() { #[test] fn create_relationship_with_properties_adds_body_triples() { let txn = lower("CREATE (a:Person)-[:KNOWS {since: 2020}]->(b:Person)"); - // 2 labels + 1 base + 3 bundle + 1 ann body = 7 + // 2 labels + 1 base + 1 link + 1 ann body = 5 assert_eq!( txn.insert_templates.len(), - 7, + 5, "templates: {:?}", txn.insert_templates ); @@ -115,8 +115,7 @@ fn create_two_parallel_relationships_mints_distinct_annotation_subjects() { (a:Person {name: "Alice"})-[:KNOWS]->(b:Person {name: "Bob"}), (a)-[:KNOWS]->(b)"#, ); - // Verify two distinct annotation subjects appear in the reifies - // bundle. + // Verify two distinct annotation subjects carry the links. let subjects: std::collections::HashSet = txn .insert_templates .iter() @@ -170,11 +169,11 @@ fn create_bare_node_asserts_existence_marker() { #[test] fn create_bare_pattern_needs_no_marker() { // `CREATE ()-[:TempEdge]->()` — the edge anchors both endpoints: - // 1 base edge + 3 reifier bundle triples, no markers. + // 1 base edge + 1 reifier link, no markers. let txn = lower("CREATE ()-[:TempEdge]->()"); assert_eq!( txn.insert_templates.len(), - 4, + 2, "templates: {:?}", txn.insert_templates ); @@ -230,8 +229,8 @@ fn deferred_write_shapes_are_rejected() { // A leading MATCH is allowed only before a *relationship* MERGE — a // node MERGE must stand alone. "MATCH (a:Person) MERGE (n:Person {name: \"A\"})", - // OPTIONAL MATCH before a relationship MERGE risks a partial reifier - // bundle (optionally-unbound endpoint), so it is rejected. + // OPTIONAL MATCH before a relationship MERGE risks a link to an + // optionally-unbound endpoint, so it is rejected. "MATCH (a:Person {name: \"A\"}) OPTIONAL MATCH (b:Person {name: \"B\"}) \ MERGE (a)-[:KNOWS]->(b)", ] { @@ -268,7 +267,7 @@ fn property_bearing_merge_relationship_guards_on_annotation_sidecar() { debug.contains("NotExists") && debug.contains("EdgeAnnotation"), "guard: {debug}" ); - // Create branch fires the endpoints + edge + reifier bundle with props. + // Create branch fires the endpoints + edge + reifier link with props. assert!(!txn.insert_templates.is_empty()); } @@ -645,10 +644,10 @@ fn merge_relationship_emits_path_guard_and_create_branch() { if let UnresolvedPattern::NotExists(guard) = &txn.where_patterns[0] { assert_eq!(guard.len(), 5, "guard: {guard:?}"); } - // Create branch: 2 labels + 2 names + base edge + 3 reifier triples = 8. + // Create branch: 2 labels + 2 names + base edge + reifier link = 6. assert_eq!( txn.insert_templates.len(), - 8, + 6, "inserts: {:?}", txn.insert_templates ); @@ -673,10 +672,10 @@ fn merge_relationship_on_create_set_routes_to_endpoint() { r#"MERGE (a:Person {name: "Alice"})-[:KNOWS]->(b:Person {name: "Bob"}) ON CREATE SET b.note = "new""#, ); - // 8 path inserts + 1 ON CREATE SET = 9. + // 6 path inserts + 1 ON CREATE SET = 7. assert_eq!( txn.insert_templates.len(), - 9, + 7, "inserts: {:?}", txn.insert_templates ); @@ -686,7 +685,7 @@ fn merge_relationship_on_create_set_routes_to_endpoint() { fn merge_relationship_incoming_direction_orients_edge() { // `<-[:KNOWS]-` puts the tail node on the subject side of the base edge. let txn = lower(r#"MERGE (a:Person {name: "Alice"})<-[:KNOWS]-(b:Person {name: "Bob"})"#); - assert_eq!(txn.insert_templates.len(), 8); + assert_eq!(txn.insert_templates.len(), 6); } #[test] @@ -715,10 +714,10 @@ fn merge_relationship_with_bound_endpoints_uses_match_vars() { .expect("a NOT EXISTS guard"); // Bound endpoints add no label/prop guard — just the rel triple. assert_eq!(guard.len(), 1, "guard: {guard:?}"); - // Create branch: only the base edge + 3 reifier triples (endpoints exist). + // Create branch: only the base edge + reifier link (endpoints exist). assert_eq!( txn.insert_templates.len(), - 4, + 2, "inserts: {:?}", txn.insert_templates ); @@ -754,10 +753,10 @@ fn merge_relationship_mixed_bound_and_new_endpoint() { .expect("a NOT EXISTS guard"); // Guard: the new tail's label + name (probe) + the rel triple = 3. assert_eq!(guard.len(), 3, "guard: {guard:?}"); - // Create: new Pet's label + name + base edge + 3 reifier triples = 6. + // Create: new Pet's label + name + base edge + reifier link = 4. assert_eq!( txn.insert_templates.len(), - 6, + 4, "inserts: {:?}", txn.insert_templates ); @@ -788,10 +787,10 @@ fn match_create_relationship_references_bound_nodes() { "where: {:?}", txn.where_patterns ); - // CREATE: base edge + 3 reifier bundle triples = 4 (no new labels). + // CREATE: base edge + reifier link = 2 (no new labels). assert_eq!( txn.insert_templates.len(), - 4, + 2, "inserts: {:?}", txn.insert_templates ); From 074c699b503a81d7160c0c6a60b830bcd924bcb2 Mon Sep 17 00:00:00 2001 From: bplatz Date: Sat, 3 Oct 2026 07:45:16 -0400 Subject: [PATCH 54/92] refactor: retire the annotation arena Annotations are read from their rdf:reifies links, so the forward/reverse annotation arena, its builder, the attachment-events provider that fed it, and novelty's attachment maps are gone. Bulk import and the server's source upload no longer reindex to seal one. Index roots written by earlier releases may still carry the arena section. Both root decoders skip it; IndexRoot keeps only its two branch CIDs so GC expansion still reaches the arena's blobs and the next build releases them as garbage. The header bits are reserved, not reused. Also fixes the import_sink unit test (feature `import`), which still expected the bundle shape. --- BENCHMARKING.md | 2 - Cargo.lock | 1 - docs/design/edge-annotations.md | 2 + docs/design/index-format.md | 11 +- docs/design/storage-traits.md | 4 +- fluree-db-api/Cargo.toml | 8 - fluree-db-api/benches/annotation_hydration.rs | 405 ------ fluree-db-api/benches/annotation_planner.rs | 325 ----- .../examples/cypher_kb_pageload_probe.rs | 12 +- fluree-db-api/examples/cypher_unwind_probe.rs | 18 +- fluree-db-api/src/admin.rs | 77 +- fluree-db-api/src/explain.rs | 86 +- fluree-db-api/src/import.rs | 40 +- .../src/indexer_attachment_provider.rs | 469 ------- fluree-db-api/src/indexer_warm_cache.rs | 40 + fluree-db-api/src/ledger_manager.rs | 118 +- fluree-db-api/src/lib.rs | 149 +-- fluree-db-api/src/view/query.rs | 3 +- .../tests/it_annotation_filter_pushdown.rs | 3 +- fluree-db-api/tests/it_edge_annotations.rs | 89 +- .../tests/it_edge_annotations_indexed.rs | 71 +- fluree-db-api/tests/it_indexing_stats.rs | 2 - fluree-db-api/tests/it_triple_term_links.rs | 24 +- fluree-db-api/tests/support/mod.rs | 51 +- fluree-db-binary-index/Cargo.toml | 1 - .../src/annotation_arena/builder.rs | 757 ----------- .../src/annotation_arena/bundle.rs | 970 -------------- .../src/annotation_arena/format.rs | 450 ------- .../src/annotation_arena/mod.rs | 53 - .../src/annotation_arena/reader.rs | 1164 ----------------- .../src/format/expanded_cas.rs | 535 ++++---- .../src/format/index_root.rs | 506 ++----- fluree-db-binary-index/src/lib.rs | 6 +- .../tests/it_arena_novelty_merge.rs | 232 ---- fluree-db-cli/src/commands/create.rs | 57 - fluree-db-cli/src/commands/index.rs | 34 +- fluree-db-core/src/annotation_index.rs | 189 --- fluree-db-core/src/db.rs | 210 +-- fluree-db-core/src/edge.rs | 855 +----------- fluree-db-core/src/lib.rs | 4 +- fluree-db-core/src/stats_view.rs | 489 ------- .../src/build/annotation_arena.rs | 421 ------ fluree-db-indexer/src/build/incremental.rs | 156 +-- fluree-db-indexer/src/build/mod.rs | 1 - fluree-db-indexer/src/build/rebuild.rs | 1 - fluree-db-indexer/src/build/root_assembly.rs | 206 +-- fluree-db-indexer/src/config.rs | 130 -- fluree-db-indexer/src/drop.rs | 124 +- fluree-db-indexer/src/gc/collector.rs | 5 +- fluree-db-indexer/src/gc/test_support.rs | 3 +- fluree-db-indexer/src/lib.rs | 5 +- fluree-db-indexer/src/orchestrator.rs | 8 - .../src/run_index/build/incremental_root.rs | 163 +-- fluree-db-ledger/src/lib.rs | 6 +- fluree-db-novelty/src/attachments.rs | 882 ------------- fluree-db-novelty/src/lib.rs | 56 +- fluree-db-query/src/planner.rs | 89 +- fluree-db-query/src/stats_cache.rs | 22 +- fluree-db-server/src/routes/import.rs | 23 +- .../tests/flpack_import_integration.rs | 11 +- fluree-db-transact/src/import_sink.rs | 32 +- 61 files changed, 616 insertions(+), 10250 deletions(-) delete mode 100644 fluree-db-api/benches/annotation_hydration.rs delete mode 100644 fluree-db-api/benches/annotation_planner.rs delete mode 100644 fluree-db-api/src/indexer_attachment_provider.rs create mode 100644 fluree-db-api/src/indexer_warm_cache.rs delete mode 100644 fluree-db-binary-index/src/annotation_arena/builder.rs delete mode 100644 fluree-db-binary-index/src/annotation_arena/bundle.rs delete mode 100644 fluree-db-binary-index/src/annotation_arena/format.rs delete mode 100644 fluree-db-binary-index/src/annotation_arena/mod.rs delete mode 100644 fluree-db-binary-index/src/annotation_arena/reader.rs delete mode 100644 fluree-db-binary-index/tests/it_arena_novelty_merge.rs delete mode 100644 fluree-db-core/src/annotation_index.rs delete mode 100644 fluree-db-indexer/src/build/annotation_arena.rs delete mode 100644 fluree-db-novelty/src/attachments.rs diff --git a/BENCHMARKING.md b/BENCHMARKING.md index 3a2cf7cc1e..eab1d14492 100644 --- a/BENCHMARKING.md +++ b/BENCHMARKING.md @@ -86,8 +86,6 @@ this table, so it is the one that rots. Add a row when you add a bench file. | `fluree-db-api` | `query_hot_property_path.rs` | Hot-cache SPARQL property paths, one scenario per execution mode of the operator (`*` closure, sequence, `?`, …) | | `fluree-db-api` | `query_hot_whole_graph_agg.rs` | Cypher aggregate folds from `fast_whole_graph_agg` (whole-graph + class scalars, histograms) against a linear-cost pipeline baseline | | `fluree-db-api` | `query_overlay_matrix.rs` | The same query shapes at four ledger conditions — base / overlay / cached / novelty — so the columnar+novelty merge lane and a cached handle's steady state after background indexing have coverage | -| `fluree-db-api` | `annotation_hydration.rs` | `inject_annotations` hydration cost: index scan vs sealed annotation arena | -| `fluree-db-api` | `annotation_planner.rs` | Planner direction for `f:reifies*` edge-annotation queries: arena-informed row counts vs HLL-only stats | | `fluree-db-query` | `vector_math.rs` | SIMD vs scalar dot/L2/cosine micro-bench | | `fluree-db-spatial` | `spatial_bench.rs` | S2 covering build + within/intersects/radius latency | diff --git a/Cargo.lock b/Cargo.lock index 801761fd70..0181c13cb5 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3007,7 +3007,6 @@ dependencies = [ "chrono", "ciborium", "fluree-db-core", - "fluree-db-novelty", "fluree-db-spatial", "fluree-vocab", "futures", diff --git a/docs/design/edge-annotations.md b/docs/design/edge-annotations.md index b47bdba6cd..2094667a02 100644 --- a/docs/design/edge-annotations.md +++ b/docs/design/edge-annotations.md @@ -63,6 +63,8 @@ Hydration (`@annotation` in subject expansion) and export read links the same wa Earlier releases stored an annotation as an `f:reifies*` bundle (`f:reifiesSubject`, `f:reifiesPredicate`, `f:reifiesObject`, …) on the reifier. Those bundles stay in commits, and the link is derived from them wherever it is needed: an index build derives it from each commit's bundle ops (`link_synth`), and novelty derives it for commits the index has not covered (`fluree-db-novelty/src/links.rs`). Readers see only links. A write that retracts a derived link leaves the bundle in place; the retract cancels the derived assert in novelty and at the next build alike. +Index roots from those releases may also carry an annotation-arena section. Readers skip it, keeping only the arena's two branch CIDs, and the next index build releases the arena's blobs as garbage. + An annotated index built before links has no term dictionary. Link reads on it fail asking for a rebuild (`fluree reindex`) rather than answering without its annotations. An incremental build over it declines once its window holds a triple term — a term dictionary covering the window alone would lift that refusal — and the index build falls back to a full rebuild, which links every annotation in history. ## See also diff --git a/docs/design/index-format.md b/docs/design/index-format.md index 5f1d0f8305..5f619a43b1 100644 --- a/docs/design/index-format.md +++ b/docs/design/index-format.md @@ -110,16 +110,19 @@ Key properties: materializing simple string values during `ORDER BY` comparisons. This flag must be cleared on the first post-import write because incremental dictionary appends break the invariant. When the flag is absent (older roots) or false, query execution must assume no lexical ordering. -- **Header flag bytes.** Byte 5 carries section/feature flags (`HAS_STATS`, `HAS_ANNOTATIONS`, - `HAS_ANNOTATION_INDEX`, …). Byte 6 is the *extended-flags* byte (low byte of the historically-zero - `pad(2)` field; legacy roots wrote `0`): - - bit 0 `HAD_ANNOTATION_ARENA` — sticky, see `docs/design/edge-annotations.md`. +- **Header flag bytes.** Byte 5 carries section/feature flags (`HAS_STATS`, `HAS_ANNOTATIONS`, …). + Bit 7 is reserved: roots from earlier releases set it for an annotation-arena section (`u32` length + + CBOR, before the term dictionary) that readers skip, keeping only its two branch CIDs so the next + build releases the arena's blobs. Byte 6 is the *extended-flags* byte (low byte of the + historically-zero `pad(2)` field; legacy roots wrote `0`): + - bit 0 — reserved (a retired annotation-arena bit; ignored). - bit 1 `LIST_META_TRACKED` + bit 2 `HAS_LIST_META` — three-state `IndexRoot.has_list_meta: Option`. `Some(false)` (tracked, clear) means the indexer observed every row and none carries an RDF-list position (`o_i == LIST_INDEX_NONE`); filtered-DELETE staging then skips its per-retraction list-meta hydration when novelty agrees. `Some(true)` is sticky. A zero byte decodes to `None` (untracked — legacy roots, bulk import) and consumers must assume lists may exist; a full rebuild, or any incremental pass that sees a list row, moves the root out of that state. + - bit 3 `HAS_TERM_DICT` — the root ends with the triple-term dictionary section. At a high level the root contains: diff --git a/docs/design/storage-traits.md b/docs/design/storage-traits.md index 24a5adc842..4f0d07b759 100644 --- a/docs/design/storage-traits.md +++ b/docs/design/storage-traits.md @@ -214,8 +214,8 @@ visible or when it is safe against power loss; that is each backend's contract: | `S3Storage` | Atomic (object PUT). | Acknowledged after replication. | `FileStorage` also narrows durability by `ContentKind`: kinds for which -`ContentKind::is_derived()` is true (index nodes, dictionaries, sketches, -annotation arenas) are written `PageCache` in either mode, because they are +`ContentKind::is_derived()` is true (index nodes, dictionaries, sketches) are +written `PageCache` in either mode, because they are rebuildable from the commit chain and are written at much higher volume than commits. Source-of-truth kinds — `Commit`, `Txn`, `LedgerConfig`, `GraphSourceMapping` — follow the instance setting. The match is exhaustive, so diff --git a/fluree-db-api/Cargo.toml b/fluree-db-api/Cargo.toml index ad8dda3597..d5fbd9caef 100644 --- a/fluree-db-api/Cargo.toml +++ b/fluree-db-api/Cargo.toml @@ -677,14 +677,6 @@ name = "policy_authorization" harness = false required-features = ["credential"] -[[bench]] -name = "annotation_hydration" -harness = false - -[[bench]] -name = "annotation_planner" -harness = false - [[bench]] name = "annotation_varlength_probe" harness = false diff --git a/fluree-db-api/benches/annotation_hydration.rs b/fluree-db-api/benches/annotation_hydration.rs deleted file mode 100644 index 71de8762c7..0000000000 --- a/fluree-db-api/benches/annotation_hydration.rs +++ /dev/null @@ -1,405 +0,0 @@ -//! Annotation hydration: M2a scan vs M2b arena. -//! -//! Compares the wall-clock cost of `HydrationFormatter::inject_annotations` -//! across two read paths. The bench uses a **subject-hydration** query -//! (`select: {"?person": ["*", {"ex:worksFor": ["*"]}]}`) so the -//! `@annotation` body is materialized via the formatter's -//! `inject_annotations` call site — a flat select with `@annotation` -//! in the where clause goes through the sync JSON-LD formatter and -//! never reaches the arena reader. -//! -//! Both ledgers are reindexed before the bench runs: -//! -//! - **scan**: reindex without an `AttachmentEventsProvider` → -//! indexed root carries `has_annotations=true, -//! annotation_index=None`. `inject_annotations` falls through to a -//! POST range query for `f:reifiesSubject` against the indexed -//! base. This is the M2a indexed-scan-fallback path, not the -//! novelty-only path. -//! - **arena**: reindex with the api `AttachmentEventsProvider` → -//! `annotation_index` is sealed. `inject_annotations` constructs -//! an `AnnotationArenaReader` once per response and resolves each -//! edge via the merged forward arena. -//! -//! ## Workload -//! -//! One edge with N attachments. The query hydrates `ex:alice` → -//! the worksFor ref expansion triggers `inject_annotations`, which -//! must surface every live annotation. -//! -//! Counts: 1, 100, 10_000. -//! -//! ## Running -//! -//! cargo bench -p fluree-db-api --bench annotation_hydration -//! -//! Quick validation (1 iteration each, no stats): -//! -//! cargo bench -p fluree-db-api --bench annotation_hydration -- --test - -#![cfg(feature = "native")] - -use criterion::{criterion_group, criterion_main, BenchmarkId, Criterion, Throughput}; -use fluree_db_api::FlureeBuilder; -use fluree_db_indexer::IndexerConfig; -use serde_json::json; -use tokio::runtime::Runtime; - -mod support { - // Local copy of the test-support helpers needed by the bench. - // We can't reach into the integration tests' `support` module - // from a bench target, so the relevant helpers are duplicated. - use async_trait::async_trait; - use fluree_db_api::{Fluree, LedgerManager, NsNotify}; - use fluree_db_indexer::{AttachmentEventCoverage, AttachmentEventsProvider}; - use std::sync::Arc; - use tokio::task::LocalSet; - - pub fn start_background_indexer_with_attachments( - fluree: &Fluree, - config: fluree_db_indexer::IndexerConfig, - ) -> (LocalSet, fluree_db_indexer::IndexerHandle) { - struct TestProvider { - manager: Arc, - } - - impl std::fmt::Debug for TestProvider { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.debug_struct("TestProvider").finish() - } - } - - #[async_trait] - impl AttachmentEventsProvider for TestProvider { - async fn attachment_events( - &self, - ledger_id: &fluree_db_api::LedgerId, - ) -> Option { - use fluree_db_api::ledger_manager::RunningCoverage; - let result = self - .manager - .try_running_attachment_events(ledger_id) - .await?; - Some(match result.coverage { - RunningCoverage::Authoritative => { - AttachmentEventCoverage::Authoritative(result.events) - } - RunningCoverage::Augment => AttachmentEventCoverage::Augment(result.events), - }) - } - } - - let manager = fluree - .ledger_manager() - .expect("ledger caching must be enabled") - .clone(); - let provider: Arc = Arc::new(TestProvider { manager }); - let config = config.with_attachment_events_provider(provider); - - let (worker, handle) = fluree_db_api::BackgroundIndexerWorker::new( - fluree.backend().clone(), - fluree - .nameservice_mode() - .publisher_arc() - .expect("test setup requires ReadWrite nameservice mode"), - config, - ); - let local = LocalSet::new(); - local.spawn_local(worker.run()); - (local, handle) - } - - pub async fn wait_for_index_application(fluree: &Fluree, ledger_id: &str, target_index_t: i64) { - use std::time::{Duration, Instant}; - - let ns_record = fluree - .nameservice() - .lookup(ledger_id) - .await - .expect("ns lookup") - .expect("ns record"); - let canonical = ns_record.ledger_id.clone(); - let mgr = fluree.ledger_manager().expect("caching enabled"); - let _ = mgr - .notify(NsNotify { - ledger_id: canonical.clone(), - record: Some(ns_record), - }) - .await - .expect("notify"); - - let deadline = Instant::now() + Duration::from_secs(10); - loop { - let handle = fluree - .ledger_cached(&canonical) - .await - .expect("ledger_cached"); - let view = handle.snapshot().await; - if view.snapshot.t >= target_index_t { - return; - } - assert!( - Instant::now() < deadline, - "timed out waiting for index application; current snapshot.t={}", - view.snapshot.t - ); - tokio::task::yield_now().await; - } - } -} - -/// Scale-driven attachment counts. Seeding N annotations costs N sequential -/// commits and per-commit work scales with accumulated novelty, so seeding is -/// O(N²); the 10k case runs for many minutes and must stay out of the per-PR -/// `bench-gate` (which runs at `Tiny` via `--test`). Honors `FLUREE_BENCH_SCALE` -/// the same way the other benches (e.g. `fulltext_query.rs`) do. -fn scenario_sizes() -> &'static [usize] { - use fluree_bench_support::BenchScale; - const SIZES: &[usize] = &[1, 100, 10_000]; - match fluree_bench_support::current_scale() { - BenchScale::Tiny => &SIZES[..2], - _ => SIZES, - } -} - -fn bench_annotation_hydration(c: &mut Criterion) { - let runtime = Runtime::new().expect("tokio runtime"); - let mut group = c.benchmark_group("annotation_hydration"); - group.sample_size(20); - - for n in scenario_sizes() { - group.throughput(Throughput::Elements(*n as u64)); - - // ---- Build two ledgers seeded with N attachments ---- - // One stays in scan-fallback (no provider on its worker). - // The other gets an arena via the full provider chain. - - let (fluree_scan, ledger_id_scan) = - runtime.block_on(seed_ledger_and_optionally_seal(*n, false)); - let (fluree_arena, ledger_id_arena) = - runtime.block_on(seed_ledger_and_optionally_seal(*n, true)); - - // Sanity: the arena ledger must have annotation_index set; - // the scan ledger must not. - let arena_state = runtime - .block_on(fluree_arena.ledger(&ledger_id_arena)) - .unwrap(); - assert!( - arena_state.snapshot.has_arena_reader(), - "arena ledger must have arena reader" - ); - let scan_state = runtime - .block_on(fluree_scan.ledger(&ledger_id_scan)) - .unwrap(); - assert!( - !scan_state.snapshot.has_arena_reader(), - "scan ledger must not have arena reader" - ); - - // Subject-hydration query — `inject_annotations` fires - // when the worksFor ref value is materialized during - // expansion of `ex:alice`. A flat `select` with - // `@annotation` in the `where` clause would route through - // the sync JSON-LD formatter and never hit the arena - // reader. - let query = json!({ - "@context": { "ex": "http://example.org/" }, - "select": {"?person": ["*", {"ex:worksFor": ["*"]}]}, - "where": {"@id": "?person", "ex:worksFor": {"@id": "?org"}} - }); - - group.bench_with_input(BenchmarkId::new("scan", n), n, |b, _| { - b.to_async(&runtime).iter(|| async { - let state = fluree_scan.ledger(&ledger_id_scan).await.unwrap(); - let db = fluree_db_api::GraphDb::from_ledger_state(&state); - let result = fluree_scan.query(&db, &query).await.unwrap(); - let _ = result.to_jsonld_async(db.as_graph_db_ref()).await.unwrap(); - }); - }); - - group.bench_with_input(BenchmarkId::new("arena", n), n, |b, _| { - b.to_async(&runtime).iter(|| async { - let state = fluree_arena.ledger(&ledger_id_arena).await.unwrap(); - let db = fluree_db_api::GraphDb::from_ledger_state(&state); - let result = fluree_arena.query(&db, &query).await.unwrap(); - let _ = result.to_jsonld_async(db.as_graph_db_ref()).await.unwrap(); - }); - }); - } - - group.finish(); - bench_non_annotation_baseline(c); -} - -/// Regression benchmark: hydration on a ledger that has **never** -/// seen an `f:reifies*` flake must not pay the per-ref-value POST -/// scan that pre-gate `inject_annotations` did. Compares two N -/// values (1 ref edge, 100 ref edges) so any per-ref overhead would -/// scale visibly. -fn bench_non_annotation_baseline(c: &mut Criterion) { - let runtime = Runtime::new().expect("tokio runtime"); - let mut group = c.benchmark_group("non_annotation_hydration"); - group.sample_size(20); - - for n in &[1usize, 100] { - group.throughput(Throughput::Elements(*n as u64)); - let (fluree, ledger_id) = runtime.block_on(seed_non_annotation_ledger(*n)); - let state = runtime.block_on(fluree.ledger(&ledger_id)).unwrap(); - assert!( - !state.snapshot.has_annotations, - "non-annotation ledger must not have sticky bit set" - ); - assert!( - !state.novelty.attachments.has_annotations(), - "novelty attachments must be empty" - ); - let query = json!({ - "@context": { "ex": "http://example.org/" }, - "select": {"?person": ["*", {"ex:worksFor": ["*"]}]}, - "where": {"@id": "?person", "ex:worksFor": {"@id": "?org"}} - }); - group.bench_with_input(BenchmarkId::new("baseline", n), n, |b, _| { - b.to_async(&runtime).iter(|| async { - let state = fluree.ledger(&ledger_id).await.unwrap(); - let db = fluree_db_api::GraphDb::from_ledger_state(&state); - let result = fluree.query(&db, &query).await.unwrap(); - let _ = result.to_jsonld_async(db.as_graph_db_ref()).await.unwrap(); - }); - }); - } - group.finish(); -} - -/// Seed a ledger with N `(person, worksFor, org)` edges. No -/// annotations anywhere — exercises the -/// `inject_annotations` zero-cost gate in hydration. -async fn seed_non_annotation_ledger(n: usize) -> (fluree_db_api::Fluree, String) { - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = format!("bench/non-annotation-hydration:{n}"); - let mut state = make_genesis(&fluree, &ledger_id); - for i in 0..n { - let txn = json!({ - "@context": { "ex": "http://example.org/" }, - "@id": format!("ex:person-{i}"), - "ex:worksFor": { "@id": format!("ex:org-{i}") } - }); - state = fluree.insert(state, &txn).await.unwrap().ledger; - } - let _ = state; - (fluree, ledger_id) -} - -/// Build a ledger with N annotations on one edge, then reindex. -/// -/// - `seal=true`: reindex with the api `AttachmentEventsProvider` -/// attached → `annotation_index` is populated, hydration takes -/// the arena path. -/// - `seal=false`: reindex with a bare worker (no provider) → -/// indexed root has `has_annotations=true, annotation_index=None`, -/// hydration takes the M2a indexed-scan-fallback path. The data -/// lives in the indexed POST, not in novelty. -async fn seed_ledger_and_optionally_seal(n: usize, seal: bool) -> (fluree_db_api::Fluree, String) { - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = format!( - "bench/annotation-hydration:{}-{}", - n, - if seal { "arena" } else { "scan" } - ); - - // Bulk insert: one base edge plus N annotations on it. We do - // this in chunks of 1000 to keep transaction sizes reasonable. - let mut state = make_genesis(&fluree, &ledger_id); - let chunk_size = 1000usize; - let mut emitted = 0usize; - let base = json!({ - "@context": { "ex": "http://example.org/" }, - "@id": "ex:alice", - "ex:worksFor": { "@id": "ex:acme" } - }); - state = fluree.insert(state, &base).await.unwrap().ledger; - while emitted < n { - let count = (n - emitted).min(chunk_size); - let mut graph = Vec::with_capacity(count); - for i in 0..count { - graph.push(json!({ - "@id": format!("ex:emp/alice-acme-{}", emitted + i), - "ex:worksFor": "@reifiesEdge", - "ex:role": format!("Role-{}", emitted + i) - })); - // Note: above is not a valid annotation shape — we - // construct annotations via the proper inline form below. - // Drop the placeholder. - } - graph.clear(); - - // Inline-annotation form. Because the M1 lowering only - // attaches one annotation per insert (the @annotation key), - // we issue one transaction per annotation. For large N this - // is slow but accurate — bulk-insert APIs for annotations - // aren't part of v1. - for i in 0..count { - let txn = json!({ - "@context": { "ex": "http://example.org/" }, - "@id": "ex:alice", - "ex:worksFor": { - "@id": "ex:acme", - "@annotation": { - "@id": format!("ex:emp/alice-acme-{}", emitted + i), - "ex:role": format!("Role-{}", emitted + i) - } - } - }); - state = fluree.insert(state, &txn).await.unwrap().ledger; - } - emitted += count; - } - - let receipt_t = state.t(); - let (local, handle) = if seal { - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()) - } else { - let (worker, handle) = fluree_db_api::BackgroundIndexerWorker::new( - fluree.backend().clone(), - fluree - .nameservice_mode() - .publisher_arc() - .expect("test setup requires ReadWrite nameservice mode"), - IndexerConfig::small(), - ); - let local = tokio::task::LocalSet::new(); - local.spawn_local(worker.run()); - (local, handle) - }; - local - .run_until(async { - // Cache the ledger pre-trigger so the provider (when - // attached) sees its overlay events. - let _ = fluree.ledger_cached(&ledger_id).await.unwrap(); - let completion = handle - .trigger( - &fluree_db_api::LedgerId::parse(&ledger_id).unwrap(), - receipt_t, - ) - .await; - let _ = completion.wait().await; - support::wait_for_index_application(&fluree, &ledger_id, receipt_t).await; - }) - .await; - - (fluree, ledger_id) -} - -fn make_genesis(fluree: &fluree_db_api::Fluree, ledger_id: &str) -> fluree_db_api::LedgerState { - let canonical = fluree_db_core::ledger_id::normalize_ledger_id(ledger_id) - .unwrap_or_else(|_| ledger_id.to_string()); - let snapshot = fluree_db_core::LedgerSnapshot::genesis(&canonical); - let _ = fluree; - fluree_db_api::LedgerState::new(snapshot, fluree_db_api::Novelty::new(0)) -} - -criterion_group!(benches, bench_annotation_hydration); -criterion_main!(benches); diff --git a/fluree-db-api/benches/annotation_planner.rs b/fluree-db-api/benches/annotation_planner.rs deleted file mode 100644 index 91d18e7199..0000000000 --- a/fluree-db-api/benches/annotation_planner.rs +++ /dev/null @@ -1,325 +0,0 @@ -//! Planner direction benchmark for edge-annotation queries (M3.3). -//! -//! Compares query throughput across two scan paths: -//! -//! - **arena**: ledger reindexed with the api `AttachmentEventsProvider` -//! so `annotation_index` is sealed. The stats-cache merges arena -//! counters into `StatsView` (M3.1), giving the join planner a real -//! row-count for each `f:reifies*` predicate. -//! - **scan**: ledger reindexed without the provider so -//! `annotation_index = None`. Property stats for `f:reifies*` come -//! only from `IndexStats.properties` (the regular HLL). -//! -//! Two query shapes per ledger: -//! -//! - **edge-rooted-selective**: `?ann` is found via a bound-subject -//! edge probe (ex:alice ex:worksFor ?org { @annotation { ?ann } }). -//! Touches one annotation. Sensitive to whether the planner orders -//! the bound-subject edge probe before the `f:reifies*` lookups. -//! - **annotation-rooted-selective**: `?ann ex:role "Director" ; -//! @reifies { ?person ex:worksFor ?org }`. Filters annotations by -//! metadata, returns the edges they reify. -//! -//! ## Workload -//! -//! - 1 base edge per person, N people, N annotations total (one per -//! edge). -//! - All edges share the same predicate (ex:worksFor) and object -//! (ex:acme). Subjects are unique (ex:person-{i}). -//! - Annotations carry `ex:role` from a small set so the -//! annotation-rooted query has a non-trivial filter to drive. -//! -//! Counts: 100, 1000. -//! -//! ## Running -//! -//! cargo bench -p fluree-db-api --bench annotation_planner -//! -//! Quick validation (1 iteration each, no stats): -//! -//! cargo bench -p fluree-db-api --bench annotation_planner -- --test - -#![cfg(feature = "native")] - -use criterion::{criterion_group, criterion_main, BenchmarkId, Criterion, Throughput}; -use fluree_db_api::FlureeBuilder; -use fluree_db_indexer::IndexerConfig; -use serde_json::{json, Value as JsonValue}; -use tokio::runtime::Runtime; - -mod support { - use async_trait::async_trait; - use fluree_db_api::{Fluree, LedgerManager, NsNotify}; - use fluree_db_indexer::{AttachmentEventCoverage, AttachmentEventsProvider}; - use std::sync::Arc; - use tokio::task::LocalSet; - - pub fn start_background_indexer_with_attachments( - fluree: &Fluree, - config: fluree_db_indexer::IndexerConfig, - ) -> (LocalSet, fluree_db_indexer::IndexerHandle) { - struct TestProvider { - manager: Arc, - } - - impl std::fmt::Debug for TestProvider { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.debug_struct("TestProvider").finish() - } - } - - #[async_trait] - impl AttachmentEventsProvider for TestProvider { - async fn attachment_events( - &self, - ledger_id: &fluree_db_api::LedgerId, - ) -> Option { - use fluree_db_api::ledger_manager::RunningCoverage; - let result = self - .manager - .try_running_attachment_events(ledger_id) - .await?; - Some(match result.coverage { - RunningCoverage::Authoritative => { - AttachmentEventCoverage::Authoritative(result.events) - } - RunningCoverage::Augment => AttachmentEventCoverage::Augment(result.events), - }) - } - } - - let manager = fluree - .ledger_manager() - .expect("ledger caching must be enabled") - .clone(); - let provider: Arc = Arc::new(TestProvider { manager }); - let config = config.with_attachment_events_provider(provider); - - let (worker, handle) = fluree_db_api::BackgroundIndexerWorker::new( - fluree.backend().clone(), - fluree - .nameservice_mode() - .publisher_arc() - .expect("test setup requires ReadWrite nameservice mode"), - config, - ); - let local = LocalSet::new(); - local.spawn_local(worker.run()); - (local, handle) - } - - pub async fn wait_for_index_application(fluree: &Fluree, ledger_id: &str, target_index_t: i64) { - use std::time::{Duration, Instant}; - - let ns_record = fluree - .nameservice() - .lookup(ledger_id) - .await - .expect("ns lookup") - .expect("ns record"); - let canonical = ns_record.ledger_id.clone(); - let mgr = fluree.ledger_manager().expect("caching enabled"); - let _ = mgr - .notify(NsNotify { - ledger_id: canonical.clone(), - record: Some(ns_record), - }) - .await - .expect("notify"); - - let deadline = Instant::now() + Duration::from_secs(30); - loop { - let handle = fluree - .ledger_cached(&canonical) - .await - .expect("ledger_cached"); - let view = handle.snapshot().await; - if view.snapshot.t >= target_index_t { - return; - } - assert!( - Instant::now() < deadline, - "timed out waiting for index application; current snapshot.t={}", - view.snapshot.t - ); - tokio::task::yield_now().await; - } - } -} - -const ROLES: &[&str] = &[ - "Engineer", - "Manager", - "Director", - "VicePresident", - "Architect", -]; - -fn edge_rooted_query() -> JsonValue { - json!({ - "@context": { "ex": "http://example.org/" }, - "select": ["?ann", "?org"], - "where": { - "@id": "ex:person-0", - "ex:worksFor": { - "@id": "?org", - "@annotation": { "@id": "?ann" } - } - } - }) -} - -fn annotation_rooted_query() -> JsonValue { - json!({ - "@context": { "ex": "http://example.org/" }, - "select": ["?person", "?org"], - "where": { - "ex:role": "Director", - "@reifies": { - "@id": "?person", - "ex:worksFor": { "@id": "?org" } - } - } - }) -} - -fn bench_annotation_planner(c: &mut Criterion) { - let runtime = Runtime::new().expect("tokio runtime"); - let mut group = c.benchmark_group("annotation_planner"); - group.sample_size(20); - - for n in &[100usize, 1000] { - group.throughput(Throughput::Elements(*n as u64)); - - let (fluree_arena, ledger_arena) = - runtime.block_on(seed_ledger_and_optionally_seal(*n, true)); - let (fluree_scan, ledger_scan) = - runtime.block_on(seed_ledger_and_optionally_seal(*n, false)); - - // Sanity: the arena ledger must have an annotation_index; - // the scan ledger must not. - let arena_state = runtime - .block_on(fluree_arena.ledger(&ledger_arena)) - .unwrap(); - assert!( - arena_state.snapshot.annotation_index.is_some(), - "arena ledger must have annotation_index after seal" - ); - let scan_state = runtime.block_on(fluree_scan.ledger(&ledger_scan)).unwrap(); - assert!( - scan_state.snapshot.annotation_index.is_none(), - "scan ledger must not have annotation_index" - ); - - let edge_q = edge_rooted_query(); - let ann_q = annotation_rooted_query(); - - group.bench_with_input(BenchmarkId::new("edge-rooted-arena", n), n, |b, _| { - b.to_async(&runtime).iter(|| async { - let state = fluree_arena.ledger(&ledger_arena).await.unwrap(); - let db = fluree_db_api::GraphDb::from_ledger_state(&state); - let _ = fluree_arena.query(&db, &edge_q).await.unwrap(); - }); - }); - - group.bench_with_input(BenchmarkId::new("edge-rooted-scan", n), n, |b, _| { - b.to_async(&runtime).iter(|| async { - let state = fluree_scan.ledger(&ledger_scan).await.unwrap(); - let db = fluree_db_api::GraphDb::from_ledger_state(&state); - let _ = fluree_scan.query(&db, &edge_q).await.unwrap(); - }); - }); - - group.bench_with_input(BenchmarkId::new("annotation-rooted-arena", n), n, |b, _| { - b.to_async(&runtime).iter(|| async { - let state = fluree_arena.ledger(&ledger_arena).await.unwrap(); - let db = fluree_db_api::GraphDb::from_ledger_state(&state); - let _ = fluree_arena.query(&db, &ann_q).await.unwrap(); - }); - }); - - group.bench_with_input(BenchmarkId::new("annotation-rooted-scan", n), n, |b, _| { - b.to_async(&runtime).iter(|| async { - let state = fluree_scan.ledger(&ledger_scan).await.unwrap(); - let db = fluree_db_api::GraphDb::from_ledger_state(&state); - let _ = fluree_scan.query(&db, &ann_q).await.unwrap(); - }); - }); - } - - group.finish(); -} - -/// Seed a ledger with N `(person-i, worksFor, acme)` edges, each -/// carrying one annotation cycling through `ROLES`. Optionally seal -/// the annotation arena at the end. -async fn seed_ledger_and_optionally_seal(n: usize, seal: bool) -> (fluree_db_api::Fluree, String) { - let fluree = FlureeBuilder::memory() - .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) - .build_memory(); - let ledger_id = format!( - "bench/annotation-planner:{}-{}", - n, - if seal { "arena" } else { "scan" } - ); - - let mut state = make_genesis(&fluree, &ledger_id); - for i in 0..n { - let role = ROLES[i % ROLES.len()]; - let txn = json!({ - "@context": { "ex": "http://example.org/" }, - "@id": format!("ex:person-{i}"), - "ex:worksFor": { - "@id": "ex:acme", - "@annotation": { - "@id": format!("ex:emp/{i}"), - "ex:role": role - } - } - }); - state = fluree.insert(state, &txn).await.unwrap().ledger; - } - - let receipt_t = state.t(); - let (local, handle) = if seal { - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()) - } else { - let (worker, handle) = fluree_db_api::BackgroundIndexerWorker::new( - fluree.backend().clone(), - fluree - .nameservice_mode() - .publisher_arc() - .expect("test setup requires ReadWrite nameservice mode"), - IndexerConfig::small(), - ); - let local = tokio::task::LocalSet::new(); - local.spawn_local(worker.run()); - (local, handle) - }; - local - .run_until(async { - let _ = fluree.ledger_cached(&ledger_id).await.unwrap(); - let completion = handle - .trigger( - &fluree_db_api::LedgerId::parse(&ledger_id).unwrap(), - receipt_t, - ) - .await; - let _ = completion.wait().await; - support::wait_for_index_application(&fluree, &ledger_id, receipt_t).await; - }) - .await; - - (fluree, ledger_id) -} - -fn make_genesis(fluree: &fluree_db_api::Fluree, ledger_id: &str) -> fluree_db_api::LedgerState { - let canonical = fluree_db_core::ledger_id::normalize_ledger_id(ledger_id) - .unwrap_or_else(|_| ledger_id.to_string()); - let snapshot = fluree_db_core::LedgerSnapshot::genesis(&canonical); - let _ = fluree; - fluree_db_api::LedgerState::new(snapshot, fluree_db_api::Novelty::new(0)) -} - -criterion_group!(benches, bench_annotation_planner); -criterion_main!(benches); diff --git a/fluree-db-api/examples/cypher_kb_pageload_probe.rs b/fluree-db-api/examples/cypher_kb_pageload_probe.rs index 5f0ae21bd8..44dadb1521 100644 --- a/fluree-db-api/examples/cypher_kb_pageload_probe.rs +++ b/fluree-db-api/examples/cypher_kb_pageload_probe.rs @@ -103,9 +103,9 @@ async fn main() { let classes = env_usize("PROBE_CLASSES", 50); let rel_types = env_usize("PROBE_RELTYPES", 50); let uris = env_usize("PROBE_URIS", 1_000); - // import: bulk-import a .jsonl (import-pipeline root, annotation_index - // absent — the customer's `fluree create --from` ledger shape) - // reindex: insert + reindex (indexer-built root WITH the annotation arena) + // import: bulk-import a .jsonl (the customer's `fluree create --from` + // ledger shape) + // reindex: insert + reindex (indexer-built root) let mode = std::env::var("PROBE_MODE").unwrap_or_else(|_| "import".to_string()); let dir = tempfile::tempdir().expect("tempdir"); @@ -167,11 +167,7 @@ async fn main() { ); let db = fluree.db("probe:kbpage").await.expect("db"); - eprintln!( - "annotation_index={} content_store={}", - db.snapshot.annotation_index.is_some(), - db.snapshot.content_store.is_some() - ); + eprintln!("has_annotations={}", db.snapshot.has_annotations); // ---- getObjectsByUris: UNWIND + untyped rel + p.p property read ---- let uri_list: Vec = (0..uris) diff --git a/fluree-db-api/examples/cypher_unwind_probe.rs b/fluree-db-api/examples/cypher_unwind_probe.rs index 44cdc58cac..8395db7c63 100644 --- a/fluree-db-api/examples/cypher_unwind_probe.rs +++ b/fluree-db-api/examples/cypher_unwind_probe.rs @@ -139,23 +139,7 @@ async fn main() { ); let db = fluree.db("probe:kb").await.expect("db"); - eprintln!( - "annotation_index={} content_store={}", - db.snapshot.annotation_index.is_some(), - db.snapshot.content_store.is_some() - ); - if std::env::var("PROBE_REINDEX2").is_ok() { - fluree - .reindex("probe:kb", ReindexOptions::default()) - .await - .expect("reindex2"); - let handle = fluree.ledger("probe:kb").await.expect("reload"); - eprintln!( - "after second reindex (fresh ledger load): annotation_index={} has_annotations={}", - handle.snapshot.annotation_index.is_some(), - handle.snapshot.has_annotations - ); - } + eprintln!("has_annotations={}", db.snapshot.has_annotations); // Baseline: point lookup by known value (customer: single-digit ms). let point = format!( diff --git a/fluree-db-api/src/admin.rs b/fluree-db-api/src/admin.rs index 952a2c552d..ac7a379428 100644 --- a/fluree-db-api/src/admin.rs +++ b/fluree-db-api/src/admin.rs @@ -2179,8 +2179,7 @@ impl crate::Fluree { // background build racing this reindex may have published between // the lookup above and the cancel. The stale record's // `index_head_id` (None, or an older root) would make the rebuild - // lose sight of the just-published root — including a sealed - // annotation arena the Augment merge in root assembly needs. + // lose sight of the just-published root. let record = self .nameservice() .lookup(&ledger_id) @@ -2192,27 +2191,6 @@ impl crate::Fluree { let gc_max_old_indexes = indexer_config.gc_max_old_indexes; let gc_min_time_mins = indexer_config.gc_min_time_mins; - // Resolve attachment events and read fulltext config off a single - // `LedgerState` load so the reindex doesn't pay for two snapshot - // hydrations on the non-annotation hot path. - // - // **Attachment-events resolution.** The rebuild path - // (`rebuild_index_from_commits`) only consumes the concrete - // `IndexerConfig.attachment_events` field; the `attachment_events_provider` - // trait is consumed only by the orchestrator's per-job dispatch - // loop. So we resolve here whenever no concrete envelope is set, - // preferring a caller-supplied provider over the API's own and - // falling back to the API provider when neither is supplied. - // (A caller-supplied **concrete** envelope is respected as-is.) - // - // **Sticky-bit gate.** We only resolve for ledgers that have - // actually observed a `f:reifies*` flake. On non-annotation - // ledgers, going through the provider has been observed to - // disturb novelty bookkeeping for unrelated facts - // (regression caught by - // `it_select_star_novelty_retract::expansion_applies_novelty_retractions`). - // The gate keeps the M2b arena-seal path intact for annotation - // ledgers while restoring the pre-M2b behavior elsewhere. let ledger_state = match self.ledger(&ledger_id).await { Ok(state) => Some(state), Err(e) => { @@ -2224,59 +2202,6 @@ impl crate::Fluree { } }; - if indexer_config.attachment_events.is_none() { - let ledger_has_annotations = ledger_state - .as_ref() - .map(|st| st.snapshot.has_annotations || st.novelty.has_annotations()) - .unwrap_or(false); - if ledger_has_annotations { - // Caller-supplied provider wins; fall back to the API's - // own provider. The provider trait is consumed via a - // ref so we don't `take()` either field — the trait - // implementation may itself be reusable downstream. - let caller_provider = indexer_config.attachment_events_provider.as_deref(); - let api_provider = self.attachment_events_provider(); - let chosen_caller = caller_provider; - let chosen_api = api_provider.as_ref().map(AsRef::as_ref); - let provider_ref: Option<&dyn fluree_db_indexer::AttachmentEventsProvider> = - chosen_caller.or(chosen_api); - if let Some(provider) = provider_ref { - // Pre-load via the cached path so the provider's - // `try_running_attachment_events` finds the running - // ledger handle even when the only LedgerState we - // hold above came from a fresh-load. - let _ = self.ledger_cached(&ledger_id).await; - indexer_config.attachment_events = provider.attachment_events(&ledger_id).await; - } - // No provider (the CLI's client carries no ledger manager) or - // the provider found nothing: derive coverage from the ledger - // state loaded above the way the provider would, including the - // base-index bootstrap for a fresh bulk import. Without this - // the indexer receives `None`, seals no arena, and every - // quoted-triple query on the imported ledger takes the - // generic join chain. - #[cfg(not(target_arch = "wasm32"))] - if indexer_config.attachment_events.is_none() { - if let Some(state) = ledger_state.as_ref() { - indexer_config.attachment_events = - crate::indexer_attachment_provider::attachment_events_from_state(state) - .await; - } - } - match indexer_config.attachment_events.as_ref() { - Some(fluree_db_indexer::AttachmentEventCoverage::Authoritative(ev)) => { - info!(ledger_id = %ledger_id, events = ev.len(), "reindex: sealing annotation arena from authoritative attachment events"); - } - Some(fluree_db_indexer::AttachmentEventCoverage::Augment(ev)) => { - info!(ledger_id = %ledger_id, events = ev.len(), "reindex: augmenting the previous annotation arena"); - } - Some(fluree_db_indexer::AttachmentEventCoverage::Unknown) | None => { - tracing::warn!(ledger_id = %ledger_id, "reindex: no attachment-event coverage resolved; annotation arena will not be sealed this pass"); - } - } - } - } - // Read the current ledger's `f:fullTextDefaults` so the reindex routes // configured plain-string values into BM25 arena building. Reuses the // `ledger_state` we already loaded above. Best-effort: if the existing diff --git a/fluree-db-api/src/explain.rs b/fluree-db-api/src/explain.rs index e6707dbb5f..f1d63ad220 100644 --- a/fluree-db-api/src/explain.rs +++ b/fluree-db-api/src/explain.rs @@ -71,58 +71,6 @@ fn triple_pattern_to_user_object( }) } -/// If the triple's predicate is one of the seven `f:reifies*` system -/// predicates, return a short slot name. Used by `/explain` to tag -/// triples that came from edge-annotation expansion so the planner's -/// chosen ordering (annotation-first vs edge-first probe) is -/// observable. -fn annotation_role_for(tp: &TriplePattern) -> Option<&'static str> { - use fluree_db_core::namespaces::{ - is_reifies_datatype, is_reifies_graph, is_reifies_lang, is_reifies_list_index, - is_reifies_object, is_reifies_predicate, is_reifies_subject, - }; - use fluree_vocab::reifies_iris; - - let by_sid = match &tp.p { - Ref::Sid(sid) => { - if is_reifies_subject(sid) { - Some("subject") - } else if is_reifies_predicate(sid) { - Some("predicate") - } else if is_reifies_object(sid) { - Some("object") - } else if is_reifies_graph(sid) { - Some("graph") - } else if is_reifies_datatype(sid) { - Some("datatype") - } else if is_reifies_lang(sid) { - Some("lang") - } else if is_reifies_list_index(sid) { - Some("listIndex") - } else { - None - } - } - _ => None, - }; - if by_sid.is_some() { - return by_sid; - } - if let Ref::Iri(iri) = &tp.p { - return match iri.as_ref() { - reifies_iris::SUBJECT => Some("subject"), - reifies_iris::PREDICATE => Some("predicate"), - reifies_iris::OBJECT => Some("object"), - reifies_iris::GRAPH => Some("graph"), - reifies_iris::DATATYPE => Some("datatype"), - reifies_iris::LANG => Some("lang"), - reifies_iris::LIST_INDEX => Some("listIndex"), - _ => None, - }; - } - None -} - fn normalize_ref_snap(snapshot: &fluree_db_core::LedgerSnapshot, r: &Ref) -> Ref { match r { Ref::Iri(iri) => snapshot @@ -420,19 +368,12 @@ fn plan_patterns_to_json( inputs.insert("fallback".to_string(), json!(inp.fallback)); } - let mut entry = json!({ + json!({ "type": typ, "pattern": triple_pattern_to_user_object(tp, vars, compactor), "selectivity": selectivity, "inputs": JsonValue::Object(inputs), - }); - if let Some(role) = annotation_role_for(tp) { - entry - .as_object_mut() - .expect("entry is object") - .insert("annotation-role".to_string(), json!(role)); - } - entry + }) }; // Original order is the query's triple pattern order. @@ -546,22 +487,10 @@ fn explain_from_parsed( &normalize_term, ); - // Build a stats view from whatever is available on the snapshot. - // Mirrors `stats_cache::cached_stats_view_for_db`: when only the - // annotation index is present (e.g. on a freshly-arena-built ledger - // before regular stats land), the merged `f:reifies*` entries are - // still useful for planning. Without this, `/explain` would report - // "no stats" while the planner happily uses arena-derived stats. - let stats_view = if snapshot.stats.is_some() || snapshot.annotation_index.is_some() { - let base = snapshot.stats.clone().unwrap_or_default(); - let mut view = StatsView::from_db_stats_with_namespaces(&base, snapshot); - if let Some(ann) = snapshot.annotation_index.as_ref() { - view.merge_annotation_stats(&ann.stats, snapshot.namespaces()); - } - Some(view) - } else { - None - }; + let stats_view = snapshot + .stats + .as_ref() + .map(|stats| StatsView::from_db_stats_with_namespaces(stats, snapshot)); let stats_available = stats_view .as_ref() .map(fluree_db_core::StatsView::has_property_stats) @@ -637,8 +566,7 @@ fn explain_from_parsed( let (original, optimized) = plan_patterns_to_json(&explain, &triples_in_order, vars, &compactor); - // Minimal statistics summary (stable + useful). `total-flakes` is - // zero when only annotation-derived stats are available. + // Minimal statistics summary (stable + useful). let total_flakes = snapshot.stats.as_ref().map(|s| s.flakes).unwrap_or(0); let statistics = json!({ "total-flakes": total_flakes, diff --git a/fluree-db-api/src/import.rs b/fluree-db-api/src/import.rs index 2bc29c024e..64e25e226d 100644 --- a/fluree-db-api/src/import.rs +++ b/fluree-db-api/src/import.rs @@ -656,16 +656,8 @@ pub struct ImportResult { pub index_t: i64, /// Optional summary of top classes, properties, and connections. pub summary: Option, - /// Whether the imported dataset contains at least one - /// `f:reifies*` flake (an edge-annotation bundle). The bulk-import - /// root itself writes `annotation_index: None`, so callers that - /// want a sealed annotation arena available immediately should - /// follow up with `Fluree::reindex(...)` — the api's - /// `ApiAttachmentEventsProvider` scans the base index for - /// `f:reifies*` flakes when the running overlay is empty, so - /// reindex produces an authoritative arena from the just-imported - /// data. The CLI's `fluree create --import` performs this - /// follow-up automatically. `false` when `build_index == false`. + /// Whether the imported dataset contains at least one edge annotation. + /// `false` when `build_index == false`. pub has_annotations: bool, /// Tracking tally (fuel, time) when a tracker was supplied via /// `ImportBuilder::tracker(...)`. `None` when tracking was disabled. @@ -6172,11 +6164,8 @@ struct IndexUploadResult { root_id: fluree_db_core::ContentId, index_t: i64, summary: Option, - /// Sticky bit: at least one `f:reifies*` predicate landed in the - /// imported dataset. Surfaced to `ImportResult.has_annotations` - /// so the CLI can auto-seal the annotation arena via a follow-up - /// `reindex` pass (the bulk-import root currently writes - /// `annotation_index: None` even when annotations are present). + /// Sticky bit: at least one annotation predicate landed in the + /// imported dataset. Surfaced as `ImportResult.has_annotations`. has_annotations: bool, /// Duplicate input statements collapsed out of the index (chunk-level /// dedup + cross-chunk merge dedup). The commit blobs keep the raw ops. @@ -6985,27 +6974,12 @@ where prev_index: None, garbage: None, sketch_ref: None, - // Bulk import path: detect annotations the same way the - // incremental indexer does — any of the seven reserved - // `f:reifies*` SIDs in the predicate dict means the - // ledger has annotations. Computed above before the + // Detected the same way the indexer does: an annotation + // predicate in the predicate dict. Computed above before the // dict moves into this struct literal. has_annotations: import_has_annotations, - annotation_index: None, + legacy_annotation_arena: None, term_dict: term_dict_refs, - // Sticky-bit canonical contract lives on - // `IndexRoot.had_annotation_arena` in - // `fluree-db-binary-index/src/format/index_root.rs`. - // Bulk import is the *only* path that leaves the bit - // false (it bypasses both incremental and full-rebuild - // root-assembly paths, which both coerce the bit on). - // That makes the - // `has_annotations=true && had_annotation_arena=false` - // shape the unique bootstrap-eligible state the - // provider's base-index scan-fallback gates on — a - // later defensive drop carries the sticky bit forward - // and stays out of the bootstrap path. - had_annotation_arena: false, // Every record written through the spool pipeline reports // whether it carried an RDF-list position, OR'd into one sticky // bit on the shared `SpoolConfig` and read after the parse diff --git a/fluree-db-api/src/indexer_attachment_provider.rs b/fluree-db-api/src/indexer_attachment_provider.rs deleted file mode 100644 index 8812f0dc6c..0000000000 --- a/fluree-db-api/src/indexer_attachment_provider.rs +++ /dev/null @@ -1,469 +0,0 @@ -//! API-side `AttachmentEventsProvider` implementation. -//! -//! Resolves a per-ledger attachment-event delta from the running -//! `LedgerManager` so the background indexer can seal authoritative -//! arenas with the live overlay state. -//! -//! ## Late-bound `LedgerManager` -//! -//! `BackgroundIndexerWorker` is constructed before `LedgerManager` -//! in [`FlureeBuilder::finalize_with_backend`], so the provider -//! can't capture a strong `Arc` at worker -//! construction time. Instead, we share a `OnceLock>` -//! between the provider (in the worker) and the builder -//! (post-LedgerManager construction). The builder fills the cell -//! once `LedgerManager` is built; the provider reads through it -//! lazily on each call. -//! -//! Until the cell is filled, the provider returns `None` — -//! "delta unknown" in the indexer's contract — which causes the -//! defensive arena-drop on the new root. Practically the cell is -//! filled before any background indexing job runs since -//! `LedgerManager` finishes construction synchronously after the -//! worker spawns. -//! -//! ## When the ledger isn't loaded -//! -//! `LedgerManager.try_running_attachment_events` returns `None` -//! when the ledger isn't currently loaded into the running -//! registry. That also routes through the indexer's "delta -//! unknown" path. The defensive drop is correct because we cannot -//! produce an authoritative event set without observing the -//! ledger's running novelty. - -#[cfg(not(target_arch = "wasm32"))] -use fluree_db_core::LedgerId; -use std::sync::{Arc, OnceLock}; - -#[cfg(not(target_arch = "wasm32"))] -use async_trait::async_trait; -#[cfg(not(target_arch = "wasm32"))] -use fluree_db_indexer::{AttachmentEventCoverage, AttachmentEventsProvider}; - -use crate::ledger_manager::LedgerManager; -#[cfg(not(target_arch = "wasm32"))] -use crate::ledger_manager::RunningCoverage; - -/// Shared late-binding cell for the api's running `LedgerManager`. -pub(crate) type LedgerManagerCell = Arc>>; - -/// Provider backed by the running `LedgerManager`. Reads the -/// snapshotted attachment overlay for the requested ledger and -/// returns its event-pair view, suitable for direct use as -/// `IndexerConfig.attachment_events`. -#[cfg(not(target_arch = "wasm32"))] -pub(crate) struct ApiAttachmentEventsProvider { - pub(crate) manager: LedgerManagerCell, -} - -#[cfg(not(target_arch = "wasm32"))] -impl std::fmt::Debug for ApiAttachmentEventsProvider { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.debug_struct("ApiAttachmentEventsProvider").finish() - } -} - -/// Resolves the process-shared read cache from the api's `LedgerManager` once -/// it's constructed — the manager owns the `LeafletCache` the query server -/// reads from, so the background indexer can warm-on-write into that exact -/// cache. Yields `None` until the manager cell is filled (and always for a -/// separate-machine indexer, which has no local manager). -#[cfg(not(target_arch = "wasm32"))] -pub(crate) struct LedgerManagerWarmCache { - pub(crate) manager: LedgerManagerCell, -} - -#[cfg(not(target_arch = "wasm32"))] -impl std::fmt::Debug for LedgerManagerWarmCache { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.debug_struct("LedgerManagerWarmCache") - .field("bound", &self.manager.get().is_some()) - .finish() - } -} - -#[cfg(not(target_arch = "wasm32"))] -impl fluree_db_indexer::WarmCacheSource for LedgerManagerWarmCache { - fn warm_cache(&self) -> Option> { - self.manager.get().and_then(|m| m.leaflet_cache().cloned()) - } -} - -#[cfg(not(target_arch = "wasm32"))] -#[async_trait] -impl AttachmentEventsProvider for ApiAttachmentEventsProvider { - async fn attachment_events(&self, ledger_id: &LedgerId) -> Option { - let manager = self.manager.get()?; - // Coverage from LedgerManager: when snapshot.t==0 (no index - // has ever run on this ledger), the AttachmentNovelty was - // built by walking every commit since genesis — provably - // complete. Once snapshot.t > 0, we can't distinguish a - // continuously-running ledger (full history preserved) from - // a reloaded one (post-index tail only), so we fall back to - // Augment so the indexer merges with the base arena's - // events. - let result = match manager.try_running_attachment_events(ledger_id).await { - Some(result) => result, - None => { - // The ledger isn't resident in the manager — the norm for a - // write-only ingest flow (committed, never read) when the - // background indexer picks it up. Without events this pass - // would run with "delta unknown", defensively drop the - // arena, and stamp `had_annotation_arena` — permanently - // blocking every later seal (the post-index reload has an - // empty attachment overlay and `Augment` coverage, and the - // sticky bit blocks the bootstrap scan). A *transient* - // load (never cached — cache insertion from here disturbs - // the running handle's novelty bookkeeping) replays the - // un-indexed commits, so a first-ever build sees the - // complete event history (`Authoritative` at - // snapshot.t == 0). - manager.transient_attachment_events(ledger_id).await? - } - }; - - // Bulk-import seal path. After `fluree create --import`, the - // `f:reifies*` flakes live in the **base index**, not in the - // running `AttachmentNovelty` overlay — so - // `try_running_attachment_events` reports an empty event set - // even though the ledger has annotations. Without a fallback - // here the indexer's arena builder early-returns and the - // arena never seals. - // - // The fallback walks the base index for `f:reifies*` flakes - // and surfaces the result as `Authoritative` — a complete - // event set sourced from the indexed live state. The scan - // **must only fire when no arena has *ever* been sealed** - // for this ledger — i.e. the first reindex after a fresh - // annotation-bearing import. Two states look identical at - // the snapshot level (`has_annotations=true, - // annotation_index=None`): - // - // - fresh bulk import — annotation-bearing flakes - // landed via the bulk-import pipeline, no indexer - // pass has ever processed them - // (`had_annotation_arena=false`). Safe to bootstrap. - // - any indexer-owned state - // (`had_annotation_arena=true`). Must NOT bootstrap. - // Covers both: - // * defensive-drop after a prior arena seal — the - // dropped arena carried historical retract/reassert - // rows that aren't in the currently-live base. - // * indexer pass on an annotation-bearing ledger - // that didn't seal (e.g. no provider attached) — - // the pass observed events that the live base - // doesn't fully reflect. - // In either case a live-only `Authoritative` reseal - // would silently lose history. Stay in scan-fallback; - // reseal will happen on a future pass that supplies - // *explicit* `Authoritative` coverage. Plain `Augment` - // with no base arena still gets refused by the - // indexer (see - // `fluree-db-indexer/src/build/incremental.rs` phase - // 3d's `Augment` branch) — Augment alone cannot - // recover missing history. - // - // The sticky `had_annotation_arena` bit — set whenever - // the indexer produces a root with `has_annotations=true` - // (regardless of whether an arena was sealed), never - // cleared, plumbed through `IndexRoot.had_annotation_arena` - // and surfaced on `LedgerSnapshot.had_annotation_arena` — - // is the only signal that distinguishes "indexer-owned" - // from "fresh bulk-import" states. Despite the name, the - // load-bearing meaning is closer to "base-index bootstrap - // is not allowed"; the bit is true for indexer-touched - // annotation-bearing roots even when no arena was ever - // sealed. - // - // Full gate (in order of cheapness): - // (i) running overlay is empty, - // (ii) loaded view exists, - // (iii) snapshot.has_annotations, - // (iv) snapshot.annotation_index.is_none(), - // (v) !snapshot.had_annotation_arena. - // The scan itself further gates on `snapshot.has_annotations` - // inside `scan_base_index_for_attachment_events`, so - // non-annotation ledgers pay nothing even if the cheap - // checks pass. - if result.events.is_empty() { - // Escape hatch: a provider-less reindex/index pass sets - // `had_annotation_arena=true` even when no arena was ever - // sealed (root_assembly sets it whenever `has_annotations`), - // which permanently blocks the bulk-import bootstrap below. - // For a known-fresh import with no retract history a - // base-index scan is still authoritative, so allow forcing - // past the sticky-bit gate. EXPERIMENTAL — not a substitute - // for fixing the bit to track actual seals/retracts. - let force_bootstrap = force_annotation_bootstrap(); - let load_view = manager.get_loaded_view(ledger_id).await; - let bootstrap_eligible = load_view - .as_ref() - .map(|v| { - v.snapshot.has_annotations - && v.snapshot.annotation_index.is_none() - && (force_bootstrap || !v.snapshot.had_annotation_arena) - }) - .unwrap_or(false); - if bootstrap_eligible { - if let Some(events) = - scan_base_index_for_attachment_events(manager, ledger_id).await - { - return Some(AttachmentEventCoverage::Authoritative(events)); - } - } - } - - Some(match result.coverage { - RunningCoverage::Authoritative => AttachmentEventCoverage::Authoritative(result.events), - RunningCoverage::Augment => AttachmentEventCoverage::Augment(result.events), - }) - } -} - -/// Walk the running ledger's base index for `f:reifies*` flakes and -/// reconstruct the currently-live attachment-event set. Returns `None` -/// when: -/// -/// - the ledger isn't loaded into the manager, -/// - the snapshot's sticky bit is clear (non-annotation ledger), -/// - the snapshot has no range provider (no base index to scan), or -/// - any structural inconsistency would corrupt the event set. -/// -/// **Caller contract:** must only invoke when no arena exists yet -/// (`snapshot.annotation_index.is_none()`). The scan reconstructs the -/// currently-live event set only — it does not see retract/reassert -/// history — so using it to "refresh" an existing arena would drop -/// historical rows. -/// -/// Strategy: for each reserved `f:reifiesSubject` / `f:reifiesPredicate` -/// / `f:reifiesObject` / `f:reifiesLang` predicate, walk PSOT -/// (`predicate = pX`) across every graph the snapshot exposes, group -/// flakes in memory by annotation SID, and decode each bundle through -/// `EdgeKey::from_reifies_facts`. We use PSOT-and-group rather than -/// per-SID SPOT scans because a SPOT scan with a constant blank-node -/// subject (namespace_code = 0) does not return rows reliably across -/// all backends. Malformed bundles are skipped (consistent with -/// `AttachmentNovelty::observe_flakes`). -#[cfg(not(target_arch = "wasm32"))] -async fn scan_base_index_for_attachment_events( - manager: &LedgerManager, - ledger_id: &LedgerId, -) -> Option< - Vec<( - fluree_db_core::edge::EdgeKey, - fluree_db_core::Sid, - i64, - bool, - )>, -> { - let view = manager.get_loaded_view(ledger_id).await?; - scan_base_index_for_attachment_events_in( - &view.snapshot, - view.novelty.as_ref(), - // A seal pass wants every bundle the base carries, so the bound is - // raised to whichever of the two is further ahead. - view.t.max(view.snapshot.t), - ledger_id, - ) - .await -} - -/// True when `FLUREE_FORCE_ANNOTATION_BOOTSTRAP` asks to run the base-index -/// bootstrap even though the sticky `had_annotation_arena` bit is set. -/// EXPERIMENTAL — not a substitute for fixing the bit to track actual -/// seals/retracts. -#[cfg(not(target_arch = "wasm32"))] -fn force_annotation_bootstrap() -> bool { - std::env::var("FLUREE_FORCE_ANNOTATION_BOOTSTRAP") - .is_ok_and(|v| v == "1" || v.eq_ignore_ascii_case("true")) -} - -/// Attachment-event coverage for a reindex that has no provider to ask — -/// the CLI's client carries no `LedgerManager`, so `Fluree::reindex` used -/// to hand the indexer `None` and every `fluree create --from` of an -/// annotation-bearing file ended with no arena while printing "sealed". -/// -/// Derived from an already-loaded `LedgerState` exactly the way -/// [`ApiAttachmentEventsProvider`] derives it from the running handle: the -/// novelty's attachment events under the same `snapshot.t == 0` coverage -/// rule, and the base-index bootstrap scan when the overlay holds no -/// events and the snapshot is a never-sealed bulk import. -#[cfg(not(target_arch = "wasm32"))] -pub(crate) async fn attachment_events_from_state( - state: &fluree_db_ledger::LedgerState, -) -> Option { - let events: Vec<_> = state.novelty.attachments.iter_event_pairs().collect(); - let snapshot = state.snapshot.as_ref(); - tracing::debug!( - ledger_id = %snapshot.ledger_id, - snapshot_t = snapshot.t, - novelty_events = events.len(), - has_annotations = snapshot.has_annotations, - arena = snapshot.annotation_index.is_some(), - had_annotation_arena = snapshot.had_annotation_arena, - range_provider = snapshot.range_provider.is_some(), - "attachment_events_from_state" - ); - if events.is_empty() { - let bootstrap_eligible = snapshot.has_annotations - && snapshot.annotation_index.is_none() - && (force_annotation_bootstrap() || !snapshot.had_annotation_arena); - if bootstrap_eligible { - if let Some(events) = scan_base_index_for_attachment_events_in( - snapshot, - &*state.novelty, - // Seal pass: whole history, as above. - state.t().max(snapshot.t), - &snapshot.ledger_id, - ) - .await - { - return Some(AttachmentEventCoverage::Authoritative(events)); - } - } - } - Some(if snapshot.t == 0 { - AttachmentEventCoverage::Authoritative(events) - } else { - AttachmentEventCoverage::Augment(events) - }) -} - -/// The base-index bootstrap scan over an explicit snapshot + overlay (see -/// [`scan_base_index_for_attachment_events`] for the contract). -/// -/// Also the read-side fallback for `fluree export` on a ledger whose arena was -/// never sealed. The sticky-bit gate that guards the *seal* caller -/// ([`attachment_events_from_state`]) does not apply to a reader: it exists so -/// a live-only scan cannot re-seal an arena and drop retract history the -/// indexer owns, and export writes no arena — it needs the bundles currently -/// asserted at `t`, which is exactly what this returns. -#[cfg(not(target_arch = "wasm32"))] -pub(crate) async fn scan_base_index_for_attachment_events_in( - snapshot: &fluree_db_core::LedgerSnapshot, - overlay: &dyn fluree_db_core::OverlayProvider, - to_t: i64, - ledger_id: &str, -) -> Option< - Vec<( - fluree_db_core::edge::EdgeKey, - fluree_db_core::Sid, - i64, - bool, - )>, -> { - use fluree_db_core::comparator::IndexType; - use fluree_db_core::edge::EdgeKey; - use fluree_db_core::range::{range_with_overlay, RangeMatch, RangeOptions, RangeTest}; - use fluree_db_core::Sid; - use std::collections::{BTreeMap, HashSet}; - - if !snapshot.has_annotations { - return None; - } - snapshot.range_provider.as_ref()?; - tracing::debug!(ledger_id, to_t, "scan_base_index_for_attachment_events"); - - // Annotation flakes may live in the default graph (g_id=0) or - // any named graph. Always include g_id=0 — `GraphRegistry::iter_entries` - // explicitly skips the default graph slot — and add every named - // graph the registry knows about. - let mut graph_ids: HashSet = HashSet::new(); - graph_ids.insert(0); - for (id, _) in snapshot.graph_registry.iter_entries() { - graph_ids.insert(id); - } - let graph_ids: Vec = graph_ids.into_iter().collect(); - - // `to_t` is the caller's, deliberately un-clamped. It used to be - // `t.max(snapshot.t)`, which is right for a seal pass — it wants the - // whole of history the base carries — and wrong for a point-in-time - // read. Raising the bound to HEAD for an `--at` export means a bundle - // retracted *after* the requested `t` is no longer returned by the - // range at all, and a downstream filter can only drop rows, never - // restore them: the annotation disappears from an export that should - // contain it, without bumping any counter, because no edge is known to - // be annotated. Seal callers pass the clamped value themselves. - - let mut events: Vec<(EdgeKey, Sid, i64, bool)> = Vec::new(); - let mut seen: HashSet<(fluree_db_core::GraphId, Sid, i64)> = HashSet::new(); - - for g_id in graph_ids { - // Collect every `f:reifies*` flake in this graph by walking - // each reserved predicate in turn (PSOT). We group in - // memory by annotation SID, which sidesteps a SPOT-scan - // quirk where a constant blank-node subject (namespace_code - // = 0) doesn't return rows reliably across all backends. - // Key by (annotation SID, t) so each transaction's bundle stays - // separate — an annotation asserted across multiple commits - // (distinct t per chunk on the multi-commit bulk-import path) - // must decode as one bundle per t, not a merged slice that - // trips `from_reifies_facts`'s Duplicate check and gets dropped. - let mut by_ann: BTreeMap<(Sid, i64), Vec> = BTreeMap::new(); - for p_iri in fluree_vocab::reifies_iris::ALL { - let Some(p_sid) = snapshot.encode_iri(p_iri) else { - // Predicate IRI never observed on this ledger — skip. - continue; - }; - let flakes = match range_with_overlay( - snapshot, - g_id, - overlay, - IndexType::Psot, - RangeTest::Eq, - RangeMatch::new().with_predicate(p_sid), - RangeOptions::new().with_to_t(to_t), - ) - .await - { - Ok(flakes) => flakes, - Err(e) => { - // The provider trait returns `Option`, so a - // failed scan can only surface as "no coverage". - // Log the underlying error before falling back — - // a silent `None` would let the indexer take - // the M2a scan path on every subsequent reindex - // without anyone knowing the base-index probe - // is broken. - tracing::warn!( - error = %e, - predicate = %p_iri, - ?g_id, - "scan_base_index_for_attachment_events: PSOT scan failed; \ - falling back to no-coverage (indexer will skip arena seal this pass)" - ); - return None; - } - }; - for f in flakes { - if !f.op { - // Skip retracted f:reifies* events: the seal pass - // only cares about currently-live bundles. - continue; - } - by_ann.entry((f.s.clone(), f.t)).or_default().push(f); - } - } - - for ((ann_sid, group_t), bundle) in by_ann { - if !seen.insert((g_id, ann_sid.clone(), group_t)) { - continue; - } - // Decode → EdgeKey. Malformed bundles are skipped - // (consistent with `AttachmentNovelty::observe_flakes`). - // Every flake in this group shares `group_t`, and a - // successful decode guarantees the group carried a valid - // `f:reifiesSubject` row (the decoder returns `Missing` - // otherwise), so `group_t` is the bundle's assertion time — - // trustworthy on the arena's `t` axis without a separate - // f:reifiesSubject lookup. The arena builder applies - // (t, op) latest-wins across the emitted events. - let Ok(edge_key) = EdgeKey::from_reifies_facts(&bundle) else { - continue; - }; - events.push((edge_key, ann_sid, group_t, /* op = */ true)); - } - } - - Some(events) -} diff --git a/fluree-db-api/src/indexer_warm_cache.rs b/fluree-db-api/src/indexer_warm_cache.rs new file mode 100644 index 0000000000..66abc4e1f5 --- /dev/null +++ b/fluree-db-api/src/indexer_warm_cache.rs @@ -0,0 +1,40 @@ +//! Lets the background indexer warm the query server's read cache on write. +//! +//! `BackgroundIndexerWorker` is constructed before `LedgerManager` in +//! [`FlureeBuilder::finalize_with_backend`](crate::FlureeBuilder), so the +//! worker can't capture the manager directly. The builder shares a +//! `OnceLock` cell with it instead and fills the cell once the manager is +//! built. + +use std::sync::{Arc, OnceLock}; + +use crate::ledger_manager::LedgerManager; + +/// Shared late-binding cell for the api's running `LedgerManager`. +pub(crate) type LedgerManagerCell = Arc>>; + +/// Resolves the process-shared read cache from the api's `LedgerManager` once +/// it's constructed — the manager owns the `LeafletCache` the query server +/// reads from, so the background indexer can warm-on-write into that exact +/// cache. Yields `None` until the manager cell is filled (and always for a +/// separate-machine indexer, which has no local manager). +#[cfg(not(target_arch = "wasm32"))] +pub(crate) struct LedgerManagerWarmCache { + pub(crate) manager: LedgerManagerCell, +} + +#[cfg(not(target_arch = "wasm32"))] +impl std::fmt::Debug for LedgerManagerWarmCache { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("LedgerManagerWarmCache") + .field("bound", &self.manager.get().is_some()) + .finish() + } +} + +#[cfg(not(target_arch = "wasm32"))] +impl fluree_db_indexer::WarmCacheSource for LedgerManagerWarmCache { + fn warm_cache(&self) -> Option> { + self.manager.get().and_then(|m| m.leaflet_cache().cloned()) + } +} diff --git a/fluree-db-api/src/ledger_manager.rs b/fluree-db-api/src/ledger_manager.rs index f30298cf58..943f39b189 100644 --- a/fluree-db-api/src/ledger_manager.rs +++ b/fluree-db-api/src/ledger_manager.rs @@ -795,12 +795,6 @@ impl LedgerHandle { ); let snap = Arc::make_mut(&mut state.snapshot); snap.range_provider = Some(Arc::new(provider)); - // Plumb the CAS handle so arena-backed annotation reads can - // resolve `AnnotationIndexRoot.{forward,reverse}_branch_cid`. - // Without this, `LedgerSnapshot::has_arena_reader()` always - // returns false and the formatter / cascade falls back to - // the M2a scan path even on snapshots with on-disk arenas. - snap.content_store = Some(Arc::clone(&cs)); #[cfg(any(target_arch = "wasm32", feature = "residency"))] prefetch_novelty_translation(&arc_store, &state.dict_novelty).await; @@ -1269,13 +1263,6 @@ pub(crate) async fn load_and_attach_binary_store( // loaded BinaryIndexStore, DictNovelty, and runtime dictionary state. let snap = Arc::make_mut(&mut state.snapshot); snap.range_provider = Some(Arc::new(provider)); - // Plumb the CAS handle so arena-backed annotation reads can resolve - // `AnnotationIndexRoot.{forward,reverse}_branch_cid`. Mirror of the - // identical line in `apply_index_v2` — fresh-load path needs the - // same wiring as the cache-update path or `has_arena_reader()` - // would always be false on snapshots loaded outside the - // LedgerManager handle path. - snap.content_store = Some(Arc::clone(&cs)); #[cfg(any(target_arch = "wasm32", feature = "residency"))] prefetch_novelty_translation(&arc_store, &state.dict_novelty).await; @@ -1316,33 +1303,6 @@ fn install_link_base( /// /// Provides single-flight loading (concurrent requests share one I/O operation) /// and idle eviction. -/// Coverage envelope returned alongside the running ledger's -/// attachment events. -/// -/// Distinguishes "we walked every commit since genesis" (safe to -/// publish as `Authoritative`) from "we only have the post-index -/// tail" (must be merged with a base arena via `Augment`). -#[derive(Debug, Clone, Copy)] -pub enum RunningCoverage { - /// Snapshot.t == 0: no index has ever run, so the running - /// `AttachmentNovelty` was built by walking every commit since - /// genesis. Provider can return `Authoritative`. - Authoritative, - /// Snapshot.t > 0: an index has run. The running - /// `AttachmentNovelty` may be the full history (continuously- - /// running ledger) or only the post-index tail (after a - /// reload). We can't distinguish, so the provider must return - /// `Augment`. - Augment, -} - -/// Result of `LedgerManager::try_running_attachment_events`. -#[derive(Debug, Clone)] -pub struct RunningAttachmentEvents { - pub coverage: RunningCoverage, - pub events: Vec<(fluree_db_core::EdgeKey, fluree_db_core::Sid, i64, bool)>, -} - pub struct LedgerManager { /// Cached ledger handles + loading state /// @@ -1453,85 +1413,9 @@ impl LedgerManager { matches!(watermark, Some(w) if w > cached_t) } - /// Snapshot the running ledger's attachment-event delta in the - /// shape the indexer's arena builder expects, plus the coverage - /// envelope describing what the events span. - /// - /// Returns `None` when: - /// - the ledger isn't currently loaded into this manager (no - /// running overlay to snapshot — the indexer treats this as - /// "delta unknown" and defensively drops any base arena), - /// - the ledger is loading (we don't block the indexer's job - /// dispatch on a load). - /// - /// Returns `Some(vec)` (possibly empty) when the snapshot was - /// observed cleanly — the empty case explicitly asserts "no - /// events since the base arena," which the indexer treats as - /// "delta is empty" and seals an authoritative (unchanged) - /// arena. - pub async fn try_running_attachment_events( - &self, - ledger_id: &LedgerId, - ) -> Option { - let handle = self.ready_handle(ledger_id).await?; - let view = handle.snapshot().await; - // Coverage heuristic: when the snapshot's `t` is zero, no - // index has ever run on this ledger, so the running - // `AttachmentNovelty` was built by walking every commit - // since genesis — it carries the complete event history. - // Once `snapshot.t > 0`, we can't distinguish a continuously- - // running ledger (full history preserved across reindexes) - // from a reloaded one (only post-index tail in the overlay), - // so the safe call is `Augment`. - let coverage = if view.snapshot.t == 0 { - RunningCoverage::Authoritative - } else { - RunningCoverage::Augment - }; - let events: Vec<_> = view.novelty.attachments.iter_event_pairs().collect(); - Some(RunningAttachmentEvents { coverage, events }) - } - - /// Side-effect-free variant of `get_or_load` + - /// [`Self::try_running_attachment_events`] for a ledger that is NOT - /// resident in the cache: load a transient `LedgerState` straight from - /// the backend (never inserted into the cache — cache insertion from a - /// background context disturbs the running handle's novelty - /// bookkeeping; see - /// `it_select_star_novelty_retract::expansion_applies_novelty_retractions`) - /// and snapshot its attachment events. The load replays every - /// post-index commit into the transient novelty, so a never-indexed - /// ledger yields the complete event history (`Authoritative` at - /// `snapshot.t == 0`) — exactly what a first background index build - /// needs to seal an authoritative arena for a write-only ingest flow. - pub async fn transient_attachment_events( - &self, - ledger_id: &LedgerId, - ) -> Option { - let canonical_alias = ledger_id.clone(); - let state = LedgerState::load(&self.nameservice_mode, &canonical_alias, &self.backend) - .await - .ok()?; - // Same coverage heuristic as `try_running_attachment_events`: a - // fresh load at snapshot.t == 0 walked every commit since genesis. - let coverage = if state.snapshot.t == 0 { - RunningCoverage::Authoritative - } else { - RunningCoverage::Augment - }; - let events: Vec<_> = state.novelty.attachments.iter_event_pairs().collect(); - Some(RunningAttachmentEvents { coverage, events }) - } - /// Return a read-only `LedgerView` for a currently-loaded ledger /// without forcing a load. Returns `None` when the ledger isn't /// in the cache. - /// - /// Used by `ApiAttachmentEventsProvider`'s bulk-import seal path: - /// when the running overlay reports no events but the snapshot's - /// sticky bit says annotations exist (the post-import state), - /// the provider needs the snapshot + range_provider to scan the - /// base index for `f:reifies*` flakes itself. pub async fn get_loaded_view(&self, ledger_id: &LedgerId) -> Option { let handle = self.ready_handle(ledger_id).await?; Some(handle.snapshot().await) @@ -3228,7 +3112,7 @@ mod tests { // must park on `state`, not on `entries`. let reader_a = { let mgr = Arc::clone(&mgr); - tokio::spawn(async move { mgr.try_running_attachment_events(&id("busy:main")).await }) + tokio::spawn(async move { mgr.get_loaded_view(&id("busy:main")).await }) }; let reader_b = { let mgr = Arc::clone(&mgr); diff --git a/fluree-db-api/src/lib.rs b/fluree-db-api/src/lib.rs index 3ef5694f44..9d86074b34 100644 --- a/fluree-db-api/src/lib.rs +++ b/fluree-db-api/src/lib.rs @@ -72,9 +72,9 @@ pub mod graphql; #[cfg(not(target_arch = "wasm32"))] pub mod import; pub mod import_source; -mod indexer_attachment_provider; #[cfg(not(target_arch = "wasm32"))] mod indexer_fulltext_provider; +mod indexer_warm_cache; mod inline_ontology; #[cfg(feature = "shacl")] mod inline_shapes; @@ -1585,12 +1585,10 @@ struct RuntimeParts { event_bus: Arc, indexing_mode: tx::IndexingMode, index_config: IndexConfig, - /// Late-binding cell for the api's `LedgerManager`, shared with - /// the background indexer's `AttachmentEventsProvider`. The cell - /// is filled in `finalize_with_backend` after `LedgerManager` is - /// constructed; until then, the provider returns `None` (the - /// indexer's "delta unknown" path). - attachment_provider_cell: indexer_attachment_provider::LedgerManagerCell, + /// Late-binding cell for the api's `LedgerManager`, shared with the + /// background indexer's warm-on-write cache source. Filled in + /// `finalize_with_backend` after `LedgerManager` is constructed. + ledger_manager_cell: indexer_warm_cache::LedgerManagerCell, } /// Spawn a background task that subscribes to `event_bus` and refreshes @@ -2261,8 +2259,8 @@ impl FlureeBuilder { /// config is discarded) and is not honored by [`build_with`], /// [`Fluree::from_backend`], or [`Fluree::with_indexing_mode`], which /// construct without an in-process worker. Provider hooks set on the - /// supplied config (`fulltext_config_provider`, - /// `attachment_events_provider`, `warm_cache_source`) are replaced by + /// supplied config (`fulltext_config_provider`, `warm_cache_source`) + /// are replaced by /// the API layer's own wiring at build time. /// /// [`with_indexing_thresholds`]: FlureeBuilder::with_indexing_thresholds @@ -2499,9 +2497,9 @@ impl FlureeBuilder { let backend = StorageBackend::Managed(self.encrypt_if_configured(Arc::new(storage))); let index_config = self.derive_indexing(); let ns_mode = NameServiceMode::ReadWrite(Arc::new(notifying.clone())); - let attachment_provider_cell = Self::new_attachment_provider_cell(); + let ledger_manager_cell = Self::new_ledger_manager_cell(); let indexing_mode = - self.start_background_indexing(&backend, ¬ifying, &attachment_provider_cell); + self.start_background_indexing(&backend, ¬ifying, &ledger_manager_cell); Ok(Self::finalize_with_backend( self.ledger_cache_config, self.config, @@ -2511,7 +2509,7 @@ impl FlureeBuilder { event_bus, indexing_mode, index_config, - attachment_provider_cell, + ledger_manager_cell, }, self.remote_connections, self.remote_mounts, @@ -2576,7 +2574,7 @@ impl FlureeBuilder { event_bus, indexing_mode: tx::IndexingMode::Disabled, index_config, - attachment_provider_cell: Self::new_attachment_provider_cell(), + ledger_manager_cell: Self::new_ledger_manager_cell(), }, self.remote_connections, self.remote_mounts, @@ -2702,7 +2700,7 @@ impl FlureeBuilder { event_bus, indexing_mode: tx::IndexingMode::Disabled, index_config, - attachment_provider_cell: Self::new_attachment_provider_cell(), + ledger_manager_cell: Self::new_ledger_manager_cell(), }, self.remote_connections, self.remote_mounts, @@ -2758,9 +2756,9 @@ impl FlureeBuilder { fluree_db_nameservice::NotifyingNameService::new(nameservice, event_bus.clone()); let ns_mode = NameServiceMode::ReadWrite(Arc::new(notifying.clone())); let index_config = self.derive_indexing(); - let attachment_provider_cell = Self::new_attachment_provider_cell(); + let ledger_manager_cell = Self::new_ledger_manager_cell(); let indexing_mode = - self.start_background_indexing(&backend, ¬ifying, &attachment_provider_cell); + self.start_background_indexing(&backend, ¬ifying, &ledger_manager_cell); Ok(Self::finalize_with_backend( self.ledger_cache_config, self.config, @@ -2770,7 +2768,7 @@ impl FlureeBuilder { event_bus, indexing_mode, index_config, - attachment_provider_cell, + ledger_manager_cell, }, self.remote_connections, self.remote_mounts, @@ -2817,9 +2815,9 @@ impl FlureeBuilder { let ns_mode = NameServiceMode::ReadWrite(Arc::new(notifying.clone())); let index_config = self.derive_indexing(); let backend = StorageBackend::Managed(self.encrypt_if_configured(Arc::new(storage))); - let attachment_provider_cell = Self::new_attachment_provider_cell(); + let ledger_manager_cell = Self::new_ledger_manager_cell(); let indexing_mode = - self.start_background_indexing(&backend, ¬ifying, &attachment_provider_cell); + self.start_background_indexing(&backend, ¬ifying, &ledger_manager_cell); Ok(Self::finalize_with_backend( self.ledger_cache_config, self.config, @@ -2829,7 +2827,7 @@ impl FlureeBuilder { event_bus, indexing_mode, index_config, - attachment_provider_cell, + ledger_manager_cell, }, self.remote_connections, self.remote_mounts, @@ -2899,9 +2897,9 @@ impl FlureeBuilder { let ns_mode = NameServiceMode::ReadWrite(Arc::new(notifying.clone())); let index_config = self.derive_indexing(); let backend = StorageBackend::Managed(self.encrypt_if_configured(Arc::new(storage))); - let attachment_provider_cell = Self::new_attachment_provider_cell(); + let ledger_manager_cell = Self::new_ledger_manager_cell(); let indexing_mode = - self.start_background_indexing(&backend, ¬ifying, &attachment_provider_cell); + self.start_background_indexing(&backend, ¬ifying, &ledger_manager_cell); Ok(Self::finalize_with_backend( self.ledger_cache_config, self.config, @@ -2911,7 +2909,7 @@ impl FlureeBuilder { event_bus, indexing_mode, index_config, - attachment_provider_cell, + ledger_manager_cell, }, self.remote_connections, self.remote_mounts, @@ -2977,7 +2975,7 @@ impl FlureeBuilder { &self, backend: &StorageBackend, nameservice: &N, - attachment_provider_cell: &indexer_attachment_provider::LedgerManagerCell, + ledger_manager_cell: &indexer_warm_cache::LedgerManagerCell, ) -> tx::IndexingMode where N: NameServiceLookup + BranchLifecycle + fluree_db_nameservice::Publisher + Clone + 'static, @@ -2985,14 +2983,13 @@ impl FlureeBuilder { self.start_background_indexing_dyn( backend, Arc::new(nameservice.clone()), - attachment_provider_cell, + ledger_manager_cell, ) } - /// Construct an empty late-binding cell for the api-side - /// `AttachmentEventsProvider`. Filled in `finalize_with_backend` - /// after `LedgerManager` is built. - fn new_attachment_provider_cell() -> indexer_attachment_provider::LedgerManagerCell { + /// Construct an empty late-binding `LedgerManager` cell. Filled in + /// `finalize_with_backend` after `LedgerManager` is built. + fn new_ledger_manager_cell() -> indexer_warm_cache::LedgerManagerCell { Arc::new(std::sync::OnceLock::new()) } @@ -3003,7 +3000,7 @@ impl FlureeBuilder { &self, _backend: &StorageBackend, _nameservice: &N, - _cell: &indexer_attachment_provider::LedgerManagerCell, + _cell: &indexer_warm_cache::LedgerManagerCell, ) -> tx::IndexingMode where N: NameServiceLookup + BranchLifecycle + fluree_db_nameservice::Publisher + Clone + 'static, @@ -3020,7 +3017,7 @@ impl FlureeBuilder { &self, backend: &StorageBackend, nameservice: Arc, - attachment_provider_cell: &indexer_attachment_provider::LedgerManagerCell, + ledger_manager_cell: &indexer_warm_cache::LedgerManagerCell, ) -> tx::IndexingMode { if let Some(ref idx_config) = self.indexing_config { // Attach an api-side full-text config provider so each index @@ -3043,31 +3040,17 @@ impl FlureeBuilder { .unwrap_or_else(|| LedgerManagerConfig::default().cache_dir), }, ) as Arc; - // Attach an api-side attachment-events provider so the - // background indexer can seal authoritative arenas with - // the running ledger's overlay state. The cell is filled - // by `finalize_with_backend` after `LedgerManager` is - // built; until then the provider returns `None` - // (delta-unknown → defensive arena drop in the indexer). - let ann_provider = Arc::new( - crate::indexer_attachment_provider::ApiAttachmentEventsProvider { - manager: Arc::clone(attachment_provider_cell), - }, - ) - as Arc; // Warm-on-write (co-located only): let the background build seed the // query server's shared read cache with the leaflets it just wrote. - // Resolved late from the same LedgerManager cell used above, so the - // worker warms the exact cache readers use. - let warm_cache_source = - Arc::new(crate::indexer_attachment_provider::LedgerManagerWarmCache { - manager: Arc::clone(attachment_provider_cell), - }) as Arc; + // Resolved late from the LedgerManager cell, so the worker warms the + // exact cache readers use. + let warm_cache_source = Arc::new(crate::indexer_warm_cache::LedgerManagerWarmCache { + manager: Arc::clone(ledger_manager_cell), + }) as Arc; let indexer_config = idx_config .indexer_config .clone() .with_fulltext_config_provider(provider) - .with_attachment_events_provider(ann_provider) .with_warm_cache_source(warm_cache_source); // BackgroundIndexerWorker takes an // `Arc` — the combined lookup @@ -3112,7 +3095,7 @@ impl FlureeBuilder { event_bus, indexing_mode, index_config, - attachment_provider_cell, + ledger_manager_cell, } = parts; let (backend, nameservice) = apply_remote_mounts(backend, nameservice, remote_mounts); let leaflet_cache = make_leaflet_cache(&config); @@ -3134,12 +3117,12 @@ impl FlureeBuilder { )) }); - // Fill the late-binding cell so the background indexer's - // attachment-events provider can resolve running ledgers. + // Fill the late-binding cell so the background indexer can warm + // the running ledgers' read cache. // Failure to set is silent — happens if `finalize_with_backend` // ran twice somehow; the first set wins. if let Some(ref mgr) = ledger_manager { - let _ = attachment_provider_cell.set(Arc::clone(mgr)); + let _ = ledger_manager_cell.set(Arc::clone(mgr)); } if let Some(manager) = &ledger_manager { @@ -3239,7 +3222,7 @@ impl FlureeBuilder { let backend = StorageBackend::Managed(storage); let event_bus = self.resolve_event_bus(); let index_config = self.derive_indexing(); - let attachment_provider_cell = Self::new_attachment_provider_cell(); + let ledger_manager_cell = Self::new_ledger_manager_cell(); let (ns_mode, indexing_mode) = match nameservice { Some(ns) => (ns, tx::IndexingMode::Disabled), @@ -3248,7 +3231,7 @@ impl FlureeBuilder { let notifying = fluree_db_nameservice::NotifyingNameService::new(ns, event_bus.clone()); let indexing_mode = - self.start_background_indexing(&backend, ¬ifying, &attachment_provider_cell); + self.start_background_indexing(&backend, ¬ifying, &ledger_manager_cell); ( NameServiceMode::ReadWrite(Arc::new(notifying)), indexing_mode, @@ -3264,7 +3247,7 @@ impl FlureeBuilder { event_bus, indexing_mode, index_config, - attachment_provider_cell, + ledger_manager_cell, }, self.remote_connections, self.remote_mounts, @@ -3316,7 +3299,7 @@ impl FlureeBuilder { let backend = StorageBackend::Managed(storage); let event_bus = self.resolve_event_bus(); let index_config = self.derive_indexing(); - let attachment_provider_cell = Self::new_attachment_provider_cell(); + let ledger_manager_cell = Self::new_ledger_manager_cell(); let (ns_mode, indexing_mode) = match nameservice { Some(ns) => (ns, tx::IndexingMode::Disabled), @@ -3324,11 +3307,8 @@ impl FlureeBuilder { let ns = FileNameService::with_storage(ns_storage); let notifying = fluree_db_nameservice::NotifyingNameService::new(ns, event_bus.clone()); - let indexing_mode = self.start_background_indexing( - &backend, - ¬ifying, - &attachment_provider_cell, - ); + let indexing_mode = + self.start_background_indexing(&backend, ¬ifying, &ledger_manager_cell); ( NameServiceMode::ReadWrite(Arc::new(notifying)), indexing_mode, @@ -3344,7 +3324,7 @@ impl FlureeBuilder { event_bus, indexing_mode, index_config, - attachment_provider_cell, + ledger_manager_cell, }, self.remote_connections, self.remote_mounts, @@ -3389,7 +3369,7 @@ impl FlureeBuilder { let backend = StorageBackend::Managed(storage); let event_bus = self.resolve_event_bus(); let index_config = self.derive_indexing(); - let attachment_provider_cell = Self::new_attachment_provider_cell(); + let ledger_manager_cell = Self::new_ledger_manager_cell(); let (ns_mode, indexing_mode) = match nameservice { Some(ns) => (ns, tx::IndexingMode::Disabled), @@ -3398,7 +3378,7 @@ impl FlureeBuilder { let ns_rw: Arc = aws_handle.nameservice_arc().clone(); let indexing_mode = - self.start_background_indexing_dyn(&backend, ns_rw, &attachment_provider_cell); + self.start_background_indexing_dyn(&backend, ns_rw, &ledger_manager_cell); (NameServiceMode::ReadWrite(ns_arc), indexing_mode) } }; @@ -3411,7 +3391,7 @@ impl FlureeBuilder { event_bus, indexing_mode, index_config, - attachment_provider_cell, + ledger_manager_cell, }, self.remote_connections, self.remote_mounts, @@ -3826,41 +3806,6 @@ impl Fluree { ) } - /// Build a [`fluree_db_indexer::AttachmentEventsProvider`] backed by - /// this connection's running `LedgerManager`. Attach it to the - /// indexer's `IndexerConfig` (via - /// `with_attachment_events_provider`) so every index build — - /// including CLI-driven incremental runs and direct `reindex` - /// calls — picks up the live attachment overlay and seals an - /// authoritative annotation arena. - /// - /// Returns `None` when ledger caching is disabled — without a - /// `LedgerManager`, the provider has nowhere to read the running - /// attachment overlay from. The indexer routes that case through - /// the defensive arena drop, which is correct. - /// - /// The background indexer constructed at `FlureeBuilder::build()` - /// time already attaches one of these automatically. External - /// callers invoking `fluree_db_indexer::build_index_for_ledger` - /// directly (e.g. the CLI's `index` command) need to attach - /// theirs by calling this method. - #[cfg(not(target_arch = "wasm32"))] - pub fn attachment_events_provider( - &self, - ) -> Option> { - use std::sync::OnceLock; - let manager = Arc::clone(self.ledger_manager.as_ref()?); - // The provider's late-binding cell is overkill here (manager - // already exists), but reusing the same provider type keeps - // one source of truth for `RunningCoverage` → - // `AttachmentEventCoverage` translation. - let cell = Arc::new(OnceLock::new()); - let _ = cell.set(manager); - Some(Arc::new( - crate::indexer_attachment_provider::ApiAttachmentEventsProvider { manager: cell }, - )) - } - /// Per-instance cache for cross-ledger governance artifacts. /// /// Exposed so the resolver in `cross_ledger::resolver` can plug diff --git a/fluree-db-api/src/view/query.rs b/fluree-db-api/src/view/query.rs index 76e9f31218..e6129c9cfd 100644 --- a/fluree-db-api/src/view/query.rs +++ b/fluree-db-api/src/view/query.rs @@ -752,14 +752,13 @@ impl Fluree { } /// Explain may show query structure, but a scoped caller must not learn - /// unrestricted cardinalities (including annotation-derived counts). + /// unrestricted cardinalities. /// Keep the normal snapshot on the root path; only governed explains clone. async fn prepare_explain_view(&self, db: &GraphDb) -> Result { let mut view = self.wrap_policy_defaults(db.clone()).await?; if !view.is_root() { let snapshot = std::sync::Arc::make_mut(&mut view.snapshot); snapshot.stats = None; - snapshot.annotation_index = None; } Ok(view) } diff --git a/fluree-db-api/tests/it_annotation_filter_pushdown.rs b/fluree-db-api/tests/it_annotation_filter_pushdown.rs index 037d510cea..980c67d7f8 100644 --- a/fluree-db-api/tests/it_annotation_filter_pushdown.rs +++ b/fluree-db-api/tests/it_annotation_filter_pushdown.rs @@ -154,8 +154,7 @@ async fn annotation_body_threshold_reduces_scan_work_on_both_surfaces() { .build_memory(); let ledger_id = "it/annotation-filter-pushdown:threshold"; - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); local .run_until(async move { diff --git a/fluree-db-api/tests/it_edge_annotations.rs b/fluree-db-api/tests/it_edge_annotations.rs index 1e689b7c99..b4d955f6fc 100644 --- a/fluree-db-api/tests/it_edge_annotations.rs +++ b/fluree-db-api/tests/it_edge_annotations.rs @@ -2458,20 +2458,10 @@ async fn wildcard_subject_hydration_keeps_explicit_iri_annotations_visible() { #[tokio::test] async fn cascade_retracts_named_graph_annotations_in_their_own_graph() { - // Regression: cascade retract bundles must carry the same - // `g = Some(graph_sid)` as the original named-graph assertion. - // A default-graph retract would not match named-graph - // assertions in Fluree's flake identity model, leaving the - // annotation orphaned in the named graph. - // - // We can't directly inspect the flake graph from the public - // API, but we *can* observe the retract via the - // `AttachmentNovelty` overlay: if the cascade emitted retracts - // in the named graph, the overlay's observer would record - // them, and `current_annotations_for_at` would return zero - // for the edge after the retract. If the retracts went to the - // default graph, the named-graph assertion would still be - // active in the overlay. + // Regression: the cascade's link retract must carry the named graph of + // the original assertion. A default-graph retract would not match it in + // Fluree's flake identity model, leaving the link live in the named + // graph after its edge is gone. let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "it/edge-annotations:cascade-named-graph"; let ledger0 = genesis_ledger(&fluree, ledger_id); @@ -2497,6 +2487,7 @@ async fn cascade_retracts_named_graph_annotations_in_their_own_graph() { // explicit. The named-graph selector tells the transactor to // emit the retract flake in the named graph. let insert_t = after_insert.ledger.t(); + let after_insert_ledger = after_insert.ledger.clone(); let delete = json!({ "@context": ctx(), "delete": { @@ -2526,39 +2517,38 @@ async fn cascade_retracts_named_graph_annotations_in_their_own_graph() { .await .expect("named-graph base re-insert"); - // After the cascade, the AttachmentNovelty observer should - // have recorded both the named-graph assertion AND a matching - // named-graph retract. With the named-graph fix, both events - // share the same `EdgeKey { g: Some(graph_a), ... }` so the - // forward map's latest event for that key is a retract (op=false). - // - // Without the fix, the assertion is keyed by `g=Some(graph_a)` - // but the retract would be keyed by `g=None` (different - // EdgeKey), so the named-graph forward rows would still show - // the annotation as currently asserted. - // - // We don't reconstruct the EdgeKey directly — we walk the - // forward map and assert that *no* named-graph edge has any - // currently-attached annotation. - let attachments = &after_reinsert.ledger.novelty.attachments; - let as_of = after_reinsert.ledger.t(); - let mut leaked_named_graph_attachments: Vec = Vec::new(); - for (edge_key, _rows) in attachments.iter_forward() { - if edge_key.g.is_none() { - continue; // default-graph edge — not what this test guards - } - let live: Vec = attachments - .current_annotations_for_at(edge_key, as_of) - .collect(); - if !live.is_empty() { - leaked_named_graph_attachments.push(format!("{edge_key:?} -> {live:?}")); - } - } - assert!( - leaked_named_graph_attachments.is_empty(), - "after named-graph cascade, no named-graph edge should have currently-attached \ - annotations; got: {leaked_named_graph_attachments:#?}" + let hr_graph = |ledger: &MemoryLedger| { + ledger + .snapshot + .graph_registry + .graph_id_for_iri("http://example.org/hr-graph") + .expect("named graph registered") + }; + let alice = after_reinsert + .ledger + .snapshot + .encode_iri("http://example.org/alice") + .expect("subject sid"); + assert_eq!( + links_for_subject_in( + &after_insert_ledger, + hr_graph(&after_insert_ledger), + &alice, + insert_t + ) + .await + .len(), + 1, + "precondition: the insert linked the named-graph edge in its graph" ); + let ledger = &after_reinsert.ledger; + for g_id in [hr_graph(ledger), 0] { + let live = links_for_subject_in(ledger, g_id, &alice, ledger.t()).await; + assert!( + live.is_empty(), + "the cascade must retract the named-graph link (graph {g_id}): {live:#?}" + ); + } } #[tokio::test] @@ -5782,10 +5772,9 @@ async fn copy_with_explicit_reifier_reads_scoped_per_graph() { #[tokio::test] async fn count_shapes_read_only_the_reifies_lookups_they_need() { - // Without a sealed arena the wrapper runs the generic `f:reifies*` - // chain, which drops the base-edge check and every lookup whose - // position nothing reads (`elide_redundant_chain`). The answers must - // not move: a plain subject that merely carries the body predicate + // An annotation pattern expands to the link and its term components, + // and lowering drops every component lookup whose position nothing + // reads (`elide_unread_term_binds`). The answers must not move: a plain subject that merely carries the body predicate // (`ex:carol ex:source`) is not a reifier, endpoints that are read // still bind, and a constant endpoint still constrains. let fluree = FlureeBuilder::memory().build_memory(); diff --git a/fluree-db-api/tests/it_edge_annotations_indexed.rs b/fluree-db-api/tests/it_edge_annotations_indexed.rs index 287b4f44ea..0c17961207 100644 --- a/fluree-db-api/tests/it_edge_annotations_indexed.rs +++ b/fluree-db-api/tests/it_edge_annotations_indexed.rs @@ -125,23 +125,12 @@ async fn hydration_reads_indexed_annotations() { #[tokio::test] async fn non_annotation_ledger_skips_inject_annotations() { - // Hydration on a ledger that has never seen an `f:reifies*` - // flake must NOT pay the per-ref-value POST scan that - // `inject_annotations` does on the M2a fallback path. The gate - // (mirror of the cascade fast-path) checks both - // `snapshot.has_annotations` and the overlay's - // `attachments.has_annotations()`. We can't directly observe - // "the scan didn't run," but we can verify three positive - // signals: - // - // 1. `snapshot.has_annotations == false` — sticky bit never - // flipped on an annotation-free ledger. - // 2. The overlay's `attachments.has_annotations()` is also - // false — no novelty-side `f:reifies*` events. - // 3. The hydration query returns the right shape with no - // `@annotation` keys anywhere — the only output the gate - // short-circuits on (the keys would still be absent on the - // scan path, but we'd pay the POST scan to find that out). + // Hydration on a ledger that has never held an annotation must not + // pay `inject_annotations`' per-ref-value link probe. The gate (mirror + // of the cascade fast-path) checks `snapshot.has_annotations` and + // `novelty.has_annotations()`. "The probe didn't run" isn't directly + // observable, so this checks that both bits stay clear, through an + // index build too, and that hydration renders no `@annotation` keys. let fluree = FlureeBuilder::memory() .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build_memory(); @@ -165,18 +154,9 @@ async fn non_annotation_ledger_skips_inject_annotations() { "non-annotation ledger must not have sticky bit set" ); assert!( - !after.ledger.novelty.attachments.has_annotations(), + !after.ledger.novelty.has_annotations(), "novelty overlay must report zero annotations" ); - assert!( - after.ledger.snapshot.annotation_index.is_none(), - "non-annotation ledger must not have an annotation_index" - ); - assert!( - !after.ledger.snapshot.has_arena_reader(), - "non-annotation ledger must not advertise an arena reader \ - (gate guarantees no CAS reads on hydration either)" - ); // Subject hydration that would otherwise call `inject_annotations` // on the worksFor ref value. Confirm output is correct AND has @@ -197,14 +177,7 @@ async fn non_annotation_ledger_skips_inject_annotations() { "non-annotation ledger must not produce any @annotation keys: {json_str}" ); - // Reindex with provider attached. Even with the provider asking - // for events, an annotation-free ledger must produce a fresh - // root with `annotation_index = None` (no arena artifacts in - // CAS at all). Verifies the indexer's "non-annotation fast - // path" — no CAS writes for branch/leaf blobs that would just - // be empty placeholders. - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); local .run_until(async { let _ = fluree.ledger_cached(ledger_id).await.unwrap(); @@ -222,14 +195,6 @@ async fn non_annotation_ledger_skips_inject_annotations() { !post.snapshot.has_annotations, "indexed root must not flip sticky bit on non-annotation ledger" ); - assert!( - post.snapshot.annotation_index.is_none(), - "indexed root must not carry an annotation_index" - ); - assert!( - !post.snapshot.has_arena_reader(), - "post-reindex snapshot must still skip arena reader" - ); }) .await; } @@ -247,8 +212,7 @@ async fn explain_expands_annotations_as_the_executor_does() { .build_memory(); let ledger_id = "it/edge-annotations-indexed:explain-expansion"; - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); local .run_until(async move { @@ -324,8 +288,7 @@ async fn transfer_named_to_named_survives_reindex() { let g1 = "http://example.org/g1"; let g2 = "http://example.org/g2"; - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); local .run_until(async move { @@ -369,14 +332,6 @@ async fn transfer_named_to_named_survives_reindex() { .await .expect("stage COPY"); - // Force the manager to cache the running ledger so the indexer's - // attachment-events provider finds it (mirrors the pattern in - // `incremental_arena_seal_then_arena_backed_query`). - let _ = fluree - .ledger_cached(ledger_id) - .await - .expect("cached load before reindex"); - support::trigger_index_and_wait(&handle, ledger_id, after_copy.receipt.t).await; support::wait_for_index_application(&fluree, ledger_id, after_copy.receipt.t).await; @@ -414,8 +369,7 @@ async fn transfer_default_to_named_synthesized_anchor_survives_reindex() { let ledger_id = "it/edge-annotations-indexed:xfer-default-to-named"; let g2 = "http://example.org/g2"; - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); local .run_until(async move { @@ -499,8 +453,7 @@ async fn indexed_literal_object_annotation_matches() { .build_memory(); let ledger_id = "it/edge-annotations-indexed:literal-object"; - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); local .run_until(async move { diff --git a/fluree-db-api/tests/it_indexing_stats.rs b/fluree-db-api/tests/it_indexing_stats.rs index 577008e515..16f027b78a 100644 --- a/fluree-db-api/tests/it_indexing_stats.rs +++ b/fluree-db-api/tests/it_indexing_stats.rs @@ -73,8 +73,6 @@ async fn apply_index( string_watermark: root.string_watermark, graph_iris: root.graph_iris, has_annotations: root.has_annotations, - annotation_index: root.annotation_index.clone(), - had_annotation_arena: root.had_annotation_arena, has_list_meta: root.has_list_meta, }; let mut db = LedgerSnapshot::new_meta(meta).expect("seed graph registry from root"); diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 083f41591d..0f1b836e63 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -175,8 +175,7 @@ async fn incremental_index_appends_new_terms_and_reuses_existing_handles() { .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build_memory(); let ledger_id = "it/triple-term-links:incremental"; - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); let ctx = json!({ "ex": "http://example.org/" }); local @@ -554,8 +553,7 @@ async fn incremental_index_follows_a_partial_repoint() { .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build_memory(); let ledger_id = "it/triple-term-links:incremental-repoint"; - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); let turtle = |body: &str| format!("VERSION \"1.2\"\n@prefix ex: .\n{body}\n"); @@ -791,8 +789,7 @@ async fn incremental_term_packs_are_compacted() { .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build_memory(); let ledger_id = "it/triple-term-links:term-pack-compaction"; - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); local .run_until(async { @@ -1182,8 +1179,7 @@ async fn novelty_links_follow_the_index_they_were_published_over() { .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build_memory(); let ledger_id = "it/triple-term-links:novelty-over-publish"; - let (local, indexer) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, indexer) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); let claim = |age: u32| { format!( "VERSION \"1.2\"\n@prefix ex: .\n\ @@ -1418,8 +1414,7 @@ async fn arena_kind_links_follow_incremental_repoints_in_every_graph() { .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build_memory(); let ledger_id = "it/triple-term-links:arena-incremental"; - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); let trig = |body: &str| { format!( "VERSION \"1.2\"\n@prefix ex: .\n\ @@ -1581,8 +1576,7 @@ async fn link_counts_follow_every_build() { .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build_memory(); let ledger_id = "it/triple-term-links:link-counts-incremental"; - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); local .run_until(async { let ledger = fluree @@ -1619,8 +1613,7 @@ async fn first_annotation_after_an_index_without_terms() { .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build_memory(); let ledger_id = "it/triple-term-links:first-annotation-incremental"; - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); let (store, _guard) = support::span_capture::init_test_tracing(); local .run_until(async { @@ -1785,8 +1778,7 @@ async fn object_bound_terms_read_the_object_tree() { .with_ledger_cache_config(fluree_db_api::LedgerManagerConfig::default()) .build_memory(); let ledger_id = "it/triple-term-links:object-tree-incremental"; - let (local, handle) = - support::start_background_indexer_with_attachments(&fluree, IndexerConfig::small()); + let (local, handle) = support::start_background_indexer_for(&fluree, IndexerConfig::small()); local .run_until(async { // The first build sees no annotation, so the next one writes the diff --git a/fluree-db-api/tests/support/mod.rs b/fluree-db-api/tests/support/mod.rs index 859d234f3e..a13d55398c 100644 --- a/fluree-db-api/tests/support/mod.rs +++ b/fluree-db-api/tests/support/mod.rs @@ -365,58 +365,13 @@ pub fn start_background_indexer_local( (local, handle) } -/// Variant that wires an `AttachmentEventsProvider` against a -/// running `Fluree`'s `LedgerManager`. Tests that exercise the M2b -/// arena-seal path use this so the worker resolves per-job -/// attachment events from the live overlay. -/// -/// The provider returns `Augment(events)` — the safe default -/// matching the api's production behavior. +/// [`start_background_indexer_local`] over a running `Fluree`'s backend and +/// nameservice. #[cfg(feature = "native")] -pub fn start_background_indexer_with_attachments( +pub fn start_background_indexer_for( fluree: &fluree_db_api::Fluree, config: fluree_db_indexer::IndexerConfig, ) -> (LocalSet, fluree_db_indexer::IndexerHandle) { - use async_trait::async_trait; - use fluree_db_indexer::{AttachmentEventCoverage, AttachmentEventsProvider}; - - struct TestProvider { - manager: Arc, - } - - impl std::fmt::Debug for TestProvider { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.debug_struct("TestProvider").finish() - } - } - - #[async_trait] - impl AttachmentEventsProvider for TestProvider { - async fn attachment_events( - &self, - ledger_id: &fluree_db_api::LedgerId, - ) -> Option { - use fluree_db_api::ledger_manager::RunningCoverage; - let result = self - .manager - .try_running_attachment_events(ledger_id) - .await?; - Some(match result.coverage { - RunningCoverage::Authoritative => { - AttachmentEventCoverage::Authoritative(result.events) - } - RunningCoverage::Augment => AttachmentEventCoverage::Augment(result.events), - }) - } - } - - let manager = fluree - .ledger_manager() - .expect("test must be built with with_ledger_cache_config") - .clone(); - let provider: Arc = Arc::new(TestProvider { manager }); - let config = config.with_attachment_events_provider(provider); - start_background_indexer_local( fluree.backend().clone(), fluree diff --git a/fluree-db-binary-index/Cargo.toml b/fluree-db-binary-index/Cargo.toml index 035c1e37c5..2f4f65a852 100644 --- a/fluree-db-binary-index/Cargo.toml +++ b/fluree-db-binary-index/Cargo.toml @@ -57,7 +57,6 @@ rust-stemmers = "1.2" [dev-dependencies] async-trait.workspace = true -fluree-db-novelty = { path = "../fluree-db-novelty" } [lints] workspace = true diff --git a/fluree-db-binary-index/src/annotation_arena/builder.rs b/fluree-db-binary-index/src/annotation_arena/builder.rs deleted file mode 100644 index a296da123d..0000000000 --- a/fluree-db-binary-index/src/annotation_arena/builder.rs +++ /dev/null @@ -1,757 +0,0 @@ -//! Pure builder helpers for the edge-annotation arenas. -//! -//! The builder is split into two stages so callers can interleave the -//! CAS write between them: -//! -//! 1. **`build_*_leaves`** — chunk pre-sorted rows into leaves and -//! return one encoded blob per chunk along with its routing summary. -//! The caller writes each blob to CAS and collects the resulting -//! [`ContentId`] for that leaf. -//! 2. **`build_*_branch`** — given leaf summaries paired with their CIDs, -//! encode the branch manifest. The caller writes that blob to CAS -//! too and stores the branch CID in [`AnnotationIndexRoot`]. -//! -//! This module contains no I/O — it owns sort-respect, chunking, and -//! aggregate stats computation only. The indexer-side glue -//! (`fluree-db-indexer/src/build/annotation_arena.rs` in slice 3b) -//! handles bundle reconstruction from `f:reifies*` flakes and CAS -//! writes. -//! -//! ## Sort invariants -//! -//! Callers MUST pass rows sorted by: -//! - Forward: `(edge, ann, t, op)` ascending. -//! - Reverse: `(ann, edge, t, op)` ascending. -//! -//! Debug builds assert this; release builds trust the caller. - -use super::format::{ - AnnotationForwardBranch, AnnotationForwardBranchEntry, AnnotationForwardLeaf, - AnnotationForwardRow, AnnotationReverseBranch, AnnotationReverseBranchEntry, - AnnotationReverseLeaf, AnnotationReverseRow, -}; -use fluree_db_core::{AnnotationStats, ContentId, EdgeKey, Sid}; -use std::collections::HashSet; - -/// Default target rows per leaf. Picked to keep the postcard-encoded -/// leaf blob in the low-MB range for typical 100-byte rows. Builders -/// can override via [`build_forward_leaves`] / [`build_reverse_leaves`] -/// when sizing for a known workload. -pub const DEFAULT_TARGET_ROWS_PER_LEAF: usize = 4096; - -/// Routing key bounds for a single forward-arena leaf. -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct ForwardLeafSummary { - pub first_edge: EdgeKey, - pub first_ann: Sid, - pub last_edge: EdgeKey, - pub last_ann: Sid, - pub row_count: u64, -} - -/// Routing key bounds for a single reverse-arena leaf. -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct ReverseLeafSummary { - pub first_ann: Sid, - pub first_edge: EdgeKey, - pub last_ann: Sid, - pub last_edge: EdgeKey, - pub row_count: u64, -} - -/// Compute aggregate stats over a sorted forward-row slice. -/// -/// Returns `(max_t, stats)`. `max_t` is `0` for empty input — the -/// caller (typically the indexer) is responsible for substituting the -/// snapshot's `index_t` if it wants the arena root to advertise the -/// snapshot's `t` even with zero rows. -/// -/// `distinct_edges` and `distinct_annotations` count **live** -/// attachments only: a `(edge, ann)` pair contributes to the totals -/// iff the final row in its history group has `op = true`. This -/// matches the field documentation on [`AnnotationStats`] and gives -/// the cost-based planner a true "currently attached" snapshot -/// rather than an over-count of historical churn. Implementation -/// relies on the `(edge, ann, t, op)` sort: the last row of each -/// `(edge, ann)` run is the latest event for that pair. -pub fn forward_arena_stats(rows: &[AnnotationForwardRow]) -> (i64, AnnotationStats) { - use fluree_db_core::FlakeValue; - - if rows.is_empty() { - return (0, AnnotationStats::default()); - } - - // Single pass: identify each live `(edge, ann)` pair (last-in- - // group + `op == true`) and accumulate distinct-value counters - // and per-pair row counts in lockstep. Mirrors - // `bundle::compute_stats` — keep both paths in sync. - let mut max_t: i64 = i64::MIN; - let mut live_edges: HashSet = HashSet::new(); - let mut live_anns: HashSet = HashSet::new(); - let mut subjects: HashSet = HashSet::new(); - let mut predicates: HashSet = HashSet::new(); - let mut objects: HashSet = HashSet::new(); - let mut graphs: HashSet = HashSet::new(); - let mut langs: HashSet = HashSet::new(); - let mut list_indices: HashSet = HashSet::new(); - let mut graph_rows: u64 = 0; - let mut lang_rows: u64 = 0; - let mut list_index_rows: u64 = 0; - let mut live_pairs: u64 = 0; - let mut graph_anns: HashSet = HashSet::new(); - let mut lang_anns: HashSet = HashSet::new(); - let mut list_index_anns: HashSet = HashSet::new(); - - for i in 0..rows.len() { - if rows[i].t > max_t { - max_t = rows[i].t; - } - let last_in_group = i + 1 == rows.len() - || rows[i].edge != rows[i + 1].edge - || rows[i].ann != rows[i + 1].ann; - // The final event in each (edge, ann) run determines whether - // the pair is currently live. Sort tie-breaker on `op` is - // `false < true`, so an assert at the same `t` as a retract - // correctly wins. - if !(last_in_group && rows[i].op) { - continue; - } - let edge = &rows[i].edge; - let ann = &rows[i].ann; - live_edges.insert(edge.clone()); - live_anns.insert(ann.clone()); - live_pairs += 1; - - subjects.insert(edge.s.clone()); - predicates.insert(edge.p.clone()); - objects.insert(edge.o.clone()); - if let Some(g) = &edge.g { - graphs.insert(g.clone()); - graph_rows += 1; - graph_anns.insert(ann.clone()); - } - if let Some(lang) = &edge.lang { - langs.insert(lang.clone()); - lang_rows += 1; - lang_anns.insert(ann.clone()); - } - if let Some(idx) = edge.list_i { - list_indices.insert(idx); - list_index_rows += 1; - list_index_anns.insert(ann.clone()); - } - } - - // `f:reifiesDatatype` rows are not synthesized from the arena — - // the on-wire predicate may be absent (JSON-LD-compatible - // cascade) and we cannot tell from the reconstructed - // `EdgeKey.dt`. See `bundle::compute_stats` for the rationale. - let stats = AnnotationStats { - forward_rows: rows.len() as u64, - // Reverse rows mirror forward rows when the indexer emits both - // arenas from the same source set. The caller fills in its own - // count if the two are not symmetric. - reverse_rows: rows.len() as u64, - distinct_edges: live_edges.len() as u64, - distinct_annotations: live_anns.len() as u64, - live_attachment_pairs: live_pairs, - distinct_reified_subjects: subjects.len() as u64, - distinct_reified_predicates: predicates.len() as u64, - distinct_reified_objects: objects.len() as u64, - reifies_graph_rows: graph_rows, - distinct_reified_graphs: graphs.len() as u64, - distinct_graph_anns: graph_anns.len() as u64, - reifies_datatype_rows: 0, - distinct_reified_datatypes: 0, - reifies_lang_rows: lang_rows, - distinct_reified_langs: langs.len() as u64, - distinct_lang_anns: lang_anns.len() as u64, - reifies_list_index_rows: list_index_rows, - distinct_reified_list_indices: list_indices.len() as u64, - distinct_list_index_anns: list_index_anns.len() as u64, - }; - (max_t, stats) -} - -/// Encode the forward-arena leaves from a sorted row slice. -/// -/// Output is one `(summary, blob)` tuple per leaf. The blobs carry the -/// `EAFL1` magic and are ready to write to CAS as -/// [`fluree_db_core::ContentKind::AnnotationForwardLeaf`]. -/// -/// Empty input produces an empty Vec — the caller decides whether to -/// emit a zero-leaf branch or omit the section entirely (per -/// `docs/design/edge-annotations.md` Sidecar arena layout, omission -/// is only legal when the snapshot has zero `f:reifies*` bundles). -/// -/// **Routing-key cohesion.** Chunks are extended past -/// `target_rows_per_leaf` whenever splitting would cut a `(edge, ann)` -/// group across two leaves. The branch holds inclusive `[first, last]` -/// `(edge, ann)` bounds per leaf; if a single hot routing key spilled -/// into two leaves, both leaves would advertise the same `(edge, ann)` -/// in their bounds and a `partition_point` lookup would only see the -/// first one — silently dropping the rest of the history. The -/// post-`target` overshoot keeps every history row for one -/// `(edge, ann)` co-located. -pub fn build_forward_leaves( - rows: &[AnnotationForwardRow], - target_rows_per_leaf: usize, -) -> Vec<(ForwardLeafSummary, Vec)> { - let target = target_rows_per_leaf.max(1); - debug_assert!( - rows.windows(2) - .all(|w| (&w[0].edge, &w[0].ann, w[0].t, w[0].op) - <= (&w[1].edge, &w[1].ann, w[1].t, w[1].op)), - "build_forward_leaves: rows must be sorted by (edge, ann, t, op)" - ); - - let mut out: Vec<(ForwardLeafSummary, Vec)> = Vec::new(); - let mut start = 0usize; - while start < rows.len() { - let mut end = (start + target).min(rows.len()); - // Extend `end` so we never split a `(edge, ann)` group across - // two leaves. The routing key for forward leaves is - // `(edge, ann)`; identical values must live in one leaf. - while end < rows.len() { - let prev = &rows[end - 1]; - let next = &rows[end]; - if prev.edge == next.edge && prev.ann == next.ann { - end += 1; - } else { - break; - } - } - let chunk = &rows[start..end]; - let first = chunk.first().expect("non-empty chunk"); - let last = chunk.last().expect("non-empty chunk"); - let summary = ForwardLeafSummary { - first_edge: first.edge.clone(), - first_ann: first.ann.clone(), - last_edge: last.edge.clone(), - last_ann: last.ann.clone(), - row_count: chunk.len() as u64, - }; - let leaf = AnnotationForwardLeaf { - rows: chunk.to_vec(), - }; - out.push((summary, leaf.encode())); - start = end; - } - out -} - -/// Encode the forward-arena branch from leaf summaries paired with -/// their CAS-written CIDs. Order must match `build_forward_leaves`' -/// output (which preserves the input row order). -pub fn build_forward_branch(leaves: &[(ForwardLeafSummary, ContentId)]) -> Vec { - let entries = leaves - .iter() - .map(|(s, cid)| AnnotationForwardBranchEntry { - first_edge: s.first_edge.clone(), - first_ann: s.first_ann.clone(), - last_edge: s.last_edge.clone(), - last_ann: s.last_ann.clone(), - row_count: s.row_count, - leaf_cid: cid.clone(), - }) - .collect(); - AnnotationForwardBranch { leaves: entries }.encode() -} - -/// Encode the reverse-arena leaves from a sorted row slice. -/// -/// Same routing-key cohesion guarantee as -/// [`build_forward_leaves`]: a single `(ann, edge)` group never -/// straddles two leaves. -pub fn build_reverse_leaves( - rows: &[AnnotationReverseRow], - target_rows_per_leaf: usize, -) -> Vec<(ReverseLeafSummary, Vec)> { - let target = target_rows_per_leaf.max(1); - debug_assert!( - rows.windows(2) - .all(|w| (&w[0].ann, &w[0].edge, w[0].t, w[0].op) - <= (&w[1].ann, &w[1].edge, w[1].t, w[1].op)), - "build_reverse_leaves: rows must be sorted by (ann, edge, t, op)" - ); - - let mut out: Vec<(ReverseLeafSummary, Vec)> = Vec::new(); - let mut start = 0usize; - while start < rows.len() { - let mut end = (start + target).min(rows.len()); - while end < rows.len() { - let prev = &rows[end - 1]; - let next = &rows[end]; - if prev.ann == next.ann && prev.edge == next.edge { - end += 1; - } else { - break; - } - } - let chunk = &rows[start..end]; - let first = chunk.first().expect("non-empty chunk"); - let last = chunk.last().expect("non-empty chunk"); - let summary = ReverseLeafSummary { - first_ann: first.ann.clone(), - first_edge: first.edge.clone(), - last_ann: last.ann.clone(), - last_edge: last.edge.clone(), - row_count: chunk.len() as u64, - }; - let leaf = AnnotationReverseLeaf { - rows: chunk.to_vec(), - }; - out.push((summary, leaf.encode())); - start = end; - } - out -} - -/// Encode the reverse-arena branch from leaf summaries + CIDs. -pub fn build_reverse_branch(leaves: &[(ReverseLeafSummary, ContentId)]) -> Vec { - let entries = leaves - .iter() - .map(|(s, cid)| AnnotationReverseBranchEntry { - first_ann: s.first_ann.clone(), - first_edge: s.first_edge.clone(), - last_ann: s.last_ann.clone(), - last_edge: s.last_edge.clone(), - row_count: s.row_count, - leaf_cid: cid.clone(), - }) - .collect(); - AnnotationReverseBranch { leaves: entries }.encode() -} - -#[cfg(test)] -mod tests { - use super::*; - use fluree_db_core::{ContentKind, FlakeValue}; - use fluree_vocab::xsd; - - fn sid(ns: u16, name: &str) -> Sid { - Sid::new(ns, name) - } - - fn edge(idx: u8) -> EdgeKey { - EdgeKey { - g: None, - s: sid(11, &format!("s{idx}")), - p: sid(12, &format!("p{idx}")), - o: FlakeValue::Ref(sid(11, &format!("o{idx}"))), - dt: Sid::new(0, xsd::ANY_URI), - lang: None, - list_i: None, - } - } - - fn fwd_row(edge_idx: u8, ann: &str, t: i64, op: bool) -> AnnotationForwardRow { - AnnotationForwardRow { - edge: edge(edge_idx), - ann: sid(20, ann), - t, - op, - } - } - - fn rev_row(ann: &str, edge_idx: u8, t: i64, op: bool) -> AnnotationReverseRow { - AnnotationReverseRow { - ann: sid(20, ann), - edge: edge(edge_idx), - t, - op, - } - } - - fn cid_for(blob: &[u8], kind: ContentKind) -> ContentId { - ContentId::new(kind, blob) - } - - #[test] - fn forward_stats_empty_returns_default() { - let (max_t, stats) = forward_arena_stats(&[]); - assert_eq!(max_t, 0); - assert_eq!(stats, AnnotationStats::default()); - } - - #[test] - fn forward_stats_distinct_excludes_retracted_pairs() { - // (edge_0, ann_a) attached then retracted → not live; - // (edge_1, ann_b) attached → live. Stats should reflect only - // currently-attached pairs, not historical churn. - let rows = vec![ - fwd_row(0, "ann_a", 1, true), - fwd_row(0, "ann_a", 2, false), - fwd_row(1, "ann_b", 3, true), - ]; - let (max_t, stats) = forward_arena_stats(&rows); - assert_eq!(max_t, 3); - assert_eq!(stats.forward_rows, 3, "all events still counted"); - assert_eq!( - stats.distinct_edges, 1, - "edge_0 was retracted; only edge_1 is live" - ); - assert_eq!(stats.distinct_annotations, 1, "only ann_b is live"); - } - - #[test] - fn forward_stats_assert_at_same_t_as_retract_wins() { - // Same `t`, both ops on the same pair. Sort tie-break on - // `op` puts `false` before `true`, so the assert is the - // final row → pair is live. - let rows = vec![fwd_row(0, "ann_a", 5, false), fwd_row(0, "ann_a", 5, true)]; - let (_, stats) = forward_arena_stats(&rows); - assert_eq!(stats.distinct_edges, 1); - assert_eq!(stats.distinct_annotations, 1); - } - - #[test] - fn forward_stats_counts_distinct_edges_and_annotations() { - // Two edges; first has two annotations and a retract event; - // second shares one annotation with the first. - let rows = vec![ - fwd_row(0, "ann_a", 1, true), - fwd_row(0, "ann_a", 2, false), - fwd_row(0, "ann_b", 3, true), - fwd_row(1, "ann_a", 4, true), - ]; - let (max_t, stats) = forward_arena_stats(&rows); - assert_eq!(max_t, 4); - assert_eq!(stats.forward_rows, 4); - assert_eq!(stats.distinct_edges, 2); - assert_eq!(stats.distinct_annotations, 2, "ann_a + ann_b across edges"); - } - - #[test] - fn forward_stats_live_attachment_pairs_counts_actual_rows() { - // Healthy v1 shape: each (edge, ann) pair is unique → the - // pair count equals distinct_annotations. - let healthy = vec![ - fwd_row(0, "ann_a", 1, true), - fwd_row(1, "ann_b", 2, true), - fwd_row(2, "ann_c", 3, true), - ]; - let (_, stats) = forward_arena_stats(&healthy); - assert_eq!(stats.distinct_annotations, 3); - assert_eq!( - stats.live_attachment_pairs, 3, - "v1 invariant: pairs == anns" - ); - - // Multi-target anomaly: ann_a is attached to two edges. The - // v1 stage-time invariant should prevent this on healthy - // ledgers, but legacy / replayed-from-corrupt-history data - // can produce it. live_attachment_pairs surfaces the actual - // row count so the planner stays accurate. - let anomaly = vec![ - fwd_row(0, "ann_a", 1, true), - fwd_row(1, "ann_a", 2, true), // same ann SID, different edge - fwd_row(2, "ann_b", 3, true), - ]; - let (_, stats) = forward_arena_stats(&anomaly); - assert_eq!(stats.distinct_annotations, 2, "two distinct ann SIDs"); - assert_eq!( - stats.live_attachment_pairs, 3, - "three live (edge, ann) pairs even though only two distinct anns" - ); - } - - #[test] - fn forward_stats_counts_optional_slot_rows_per_attachment() { - // Two annotations on the same named-graph edge must produce - // two `f:reifiesGraph` rows in the live state — counting - // per distinct edge would under-report. Same shape for - // `f:reifiesLang` and `f:reifiesListIndex`. - let g_sid = sid(99, "g1"); - let mk = |s: u8, ann: &str, t: i64, op: bool| AnnotationForwardRow { - edge: EdgeKey { - g: Some(g_sid.clone()), - s: sid(11, &format!("s{s}")), - p: sid(12, "worksFor"), - o: FlakeValue::Ref(sid(11, "acme")), - dt: Sid::new(0, xsd::ANY_URI), - lang: None, - list_i: None, - }, - ann: sid(20, ann), - t, - op, - }; - let rows = vec![ - mk(0, "ann_a", 1, true), - mk(0, "ann_b", 2, true), - mk(1, "ann_c", 3, true), - ]; - let (_, stats) = forward_arena_stats(&rows); - assert_eq!(stats.distinct_edges, 2); - assert_eq!(stats.distinct_annotations, 3); - assert_eq!(stats.distinct_reified_graphs, 1, "single distinct graph"); - assert_eq!( - stats.reifies_graph_rows, 3, - "three live (edge, ann) pairs each contribute one f:reifiesGraph row" - ); - assert_eq!( - stats.distinct_graph_anns, 3, - "three distinct ann SIDs hold f:reifiesGraph rows under v1 invariant" - ); - - // datatype rows always 0 from the arena (see compute_stats - // comment): the arena reconstructs `EdgeKey.dt` from the - // f:reifiesObject flake-level dt, so it can't observe the - // separate f:reifiesDatatype predicate's actual presence. - assert_eq!(stats.reifies_datatype_rows, 0); - assert_eq!(stats.distinct_reified_datatypes, 0); - } - - #[test] - fn forward_stats_distinct_slot_anns_dedupes_under_multi_target() { - // Multi-target anomaly: ann_a has graph slots on three - // different edges. The arena reports `reifies_graph_rows = 3` - // but `distinct_graph_anns = 1` — only one ann SID actually - // holds graph slots. ann_b has one graph row; ann_c has no - // graph (default-graph edge) so doesn't appear in the - // graph-anns set. - let g = sid(99, "g1"); - let mk_with_g = |s: u8, ann: &str, t: i64| AnnotationForwardRow { - edge: EdgeKey { - g: Some(g.clone()), - s: sid(11, &format!("s{s}")), - p: sid(12, "worksFor"), - o: FlakeValue::Ref(sid(11, "acme")), - dt: Sid::new(0, xsd::ANY_URI), - lang: None, - list_i: None, - }, - ann: sid(20, ann), - t, - op: true, - }; - let mk_default = |s: u8, ann: &str, t: i64| AnnotationForwardRow { - edge: EdgeKey { - g: None, - s: sid(11, &format!("s{s}")), - p: sid(12, "worksFor"), - o: FlakeValue::Ref(sid(11, "acme")), - dt: Sid::new(0, xsd::ANY_URI), - lang: None, - list_i: None, - }, - ann: sid(20, ann), - t, - op: true, - }; - let rows = vec![ - mk_with_g(0, "ann_a", 1), - mk_with_g(1, "ann_a", 2), - mk_with_g(2, "ann_a", 3), - mk_with_g(3, "ann_b", 4), - mk_default(4, "ann_c", 5), - ]; - let (_, stats) = forward_arena_stats(&rows); - assert_eq!(stats.distinct_annotations, 3); - assert_eq!(stats.live_attachment_pairs, 5); - assert_eq!(stats.reifies_graph_rows, 4); - assert_eq!( - stats.distinct_graph_anns, 2, - "ann_a + ann_b carry graph rows; ann_c does not" - ); - } - - #[test] - fn forward_round_trip_single_leaf() { - let rows = vec![ - fwd_row(0, "ann_a", 1, true), - fwd_row(0, "ann_b", 2, true), - fwd_row(1, "ann_c", 3, true), - ]; - let leaves = build_forward_leaves(&rows, DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!(leaves.len(), 1); - let (summary, blob) = &leaves[0]; - assert_eq!(summary.row_count, 3); - assert_eq!(summary.first_edge, edge(0)); - assert_eq!(summary.last_edge, edge(1)); - - let decoded = AnnotationForwardLeaf::decode(blob).unwrap(); - assert_eq!(decoded.rows, rows); - - let cid = cid_for(blob, ContentKind::AnnotationForwardLeaf); - let branch_bytes = build_forward_branch(&[(summary.clone(), cid.clone())]); - let branch = AnnotationForwardBranch::decode(&branch_bytes).unwrap(); - assert_eq!(branch.leaves.len(), 1); - assert_eq!(branch.leaves[0].leaf_cid, cid); - assert_eq!(branch.leaves[0].first_edge, edge(0)); - assert_eq!(branch.leaves[0].last_edge, edge(1)); - assert_eq!(branch.leaves[0].row_count, 3); - } - - #[test] - fn forward_chunking_keeps_routing_key_groups_together() { - // 5 events for the same `(edge_0, ann_a)` history followed by - // a singleton `(edge_1, ann_b)`. With `target = 2`, naive - // row-count chunking would split `(edge_0, ann_a)` across two - // leaves, leaving overlapping inclusive bounds in the branch. - // Routing-key cohesion must extend the first chunk past the - // target so the hot key stays in one leaf — otherwise a - // `partition_point` lookup in the branch would only find the - // first leaf and silently drop the rest of the history. - let rows: Vec = (1..=5) - .map(|t| fwd_row(0, "ann_a", t, true)) - .chain(std::iter::once(fwd_row(1, "ann_b", 6, true))) - .collect(); - let leaves = build_forward_leaves(&rows, 2); - assert_eq!(leaves.len(), 2); - assert_eq!( - leaves[0].0.row_count, 5, - "all 5 (edge_0, ann_a) rows colocated" - ); - assert_eq!(leaves[0].0.first_ann, sid(20, "ann_a")); - assert_eq!(leaves[0].0.last_ann, sid(20, "ann_a")); - assert_eq!(leaves[1].0.row_count, 1); - // Branch entries must have non-overlapping `(edge, ann)` bounds. - let entries: Vec<_> = leaves - .iter() - .map(|(s, b)| (s.clone(), cid_for(b, ContentKind::AnnotationForwardLeaf))) - .collect(); - let branch_bytes = build_forward_branch(&entries); - let branch = AnnotationForwardBranch::decode(&branch_bytes).unwrap(); - for w in branch.leaves.windows(2) { - assert!( - (&w[0].last_edge, &w[0].last_ann) < (&w[1].first_edge, &w[1].first_ann), - "leaf bounds must not overlap on routing key" - ); - } - } - - #[test] - fn reverse_chunking_keeps_routing_key_groups_together() { - // Same scenario for the reverse arena: 5 events for the same - // `(ann_a, edge_0)` then a singleton. - let rows: Vec = (1..=5) - .map(|t| rev_row("ann_a", 0, t, true)) - .chain(std::iter::once(rev_row("ann_b", 1, 6, true))) - .collect(); - let leaves = build_reverse_leaves(&rows, 2); - assert_eq!(leaves.len(), 2); - assert_eq!(leaves[0].0.row_count, 5); - assert_eq!(leaves[0].0.first_ann, sid(20, "ann_a")); - assert_eq!(leaves[0].0.last_ann, sid(20, "ann_a")); - } - - #[test] - fn forward_round_trip_chunked_into_multiple_leaves() { - // 5 rows with target=2 should produce 3 leaves of sizes 2, 2, 1. - let rows = vec![ - fwd_row(0, "ann_a", 1, true), - fwd_row(0, "ann_b", 2, true), - fwd_row(1, "ann_a", 3, true), - fwd_row(1, "ann_b", 4, true), - fwd_row(2, "ann_a", 5, true), - ]; - let leaves = build_forward_leaves(&rows, 2); - assert_eq!(leaves.len(), 3); - assert_eq!(leaves[0].0.row_count, 2); - assert_eq!(leaves[1].0.row_count, 2); - assert_eq!(leaves[2].0.row_count, 1); - // Boundary keys must be inclusive on both sides. - assert_eq!(leaves[0].0.first_ann, sid(20, "ann_a")); - assert_eq!(leaves[0].0.last_ann, sid(20, "ann_b")); - assert_eq!(leaves[1].0.first_edge, edge(1)); - assert_eq!(leaves[2].0.first_edge, edge(2)); - - // Concatenating decoded leaves must equal the original input. - let mut roundtrip: Vec = Vec::new(); - let mut entries: Vec<(ForwardLeafSummary, ContentId)> = Vec::new(); - for (summary, blob) in &leaves { - let decoded = AnnotationForwardLeaf::decode(blob).unwrap(); - roundtrip.extend(decoded.rows); - entries.push(( - summary.clone(), - cid_for(blob, ContentKind::AnnotationForwardLeaf), - )); - } - assert_eq!(roundtrip, rows); - - // Branch entries must be in the same order as the leaves. - let branch_bytes = build_forward_branch(&entries); - let branch = AnnotationForwardBranch::decode(&branch_bytes).unwrap(); - assert_eq!(branch.leaves.len(), 3); - for (i, entry) in branch.leaves.iter().enumerate() { - assert_eq!(entry.row_count, entries[i].0.row_count); - assert_eq!(entry.leaf_cid, entries[i].1); - } - } - - #[test] - fn forward_empty_input_produces_zero_leaves() { - let leaves = build_forward_leaves(&[], DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!(leaves.len(), 0); - // The caller can still emit an empty branch. - let branch_bytes = build_forward_branch(&[]); - let branch = AnnotationForwardBranch::decode(&branch_bytes).unwrap(); - assert!(branch.leaves.is_empty()); - } - - #[test] - fn reverse_round_trip_chunked() { - let rows = vec![ - rev_row("ann_a", 0, 1, true), - rev_row("ann_a", 1, 2, true), - rev_row("ann_b", 0, 3, true), - ]; - let leaves = build_reverse_leaves(&rows, 2); - assert_eq!(leaves.len(), 2); - assert_eq!(leaves[0].0.first_ann, sid(20, "ann_a")); - assert_eq!(leaves[0].0.last_ann, sid(20, "ann_a")); - assert_eq!(leaves[1].0.first_ann, sid(20, "ann_b")); - assert_eq!(leaves[1].0.last_ann, sid(20, "ann_b")); - - let mut entries: Vec<(ReverseLeafSummary, ContentId)> = Vec::new(); - let mut roundtrip = Vec::new(); - for (summary, blob) in &leaves { - let decoded = AnnotationReverseLeaf::decode(blob).unwrap(); - roundtrip.extend(decoded.rows); - entries.push(( - summary.clone(), - cid_for(blob, ContentKind::AnnotationReverseLeaf), - )); - } - assert_eq!(roundtrip, rows); - - let branch_bytes = build_reverse_branch(&entries); - let branch = AnnotationReverseBranch::decode(&branch_bytes).unwrap(); - assert_eq!(branch.leaves.len(), 2); - assert_eq!(branch.leaves[0].first_ann, sid(20, "ann_a")); - assert_eq!(branch.leaves[1].last_edge, edge(0)); - } - - // debug_assert!-only invariant: not compiled in release/`bench` profile - // builds (where bench-gate runs the unit tests), so gate on debug_assertions - // — otherwise `#[should_panic]` fails because the assert is stripped. - #[cfg(debug_assertions)] - #[test] - #[should_panic(expected = "rows must be sorted")] - fn forward_debug_assert_catches_unsorted() { - // Out of order on `(edge, ann)` — debug builds must catch. - let rows = vec![fwd_row(1, "ann_a", 1, true), fwd_row(0, "ann_a", 2, true)]; - let _ = build_forward_leaves(&rows, DEFAULT_TARGET_ROWS_PER_LEAF); - } - - #[cfg(debug_assertions)] - #[test] - #[should_panic(expected = "rows must be sorted")] - fn reverse_debug_assert_catches_unsorted() { - let rows = vec![rev_row("ann_b", 0, 1, true), rev_row("ann_a", 0, 2, true)]; - let _ = build_reverse_leaves(&rows, DEFAULT_TARGET_ROWS_PER_LEAF); - } - - #[test] - fn target_rows_zero_is_treated_as_one() { - // Defensive: zero target would otherwise infinite-loop on - // chunks. We clamp to 1. - let rows = vec![fwd_row(0, "ann_a", 1, true), fwd_row(1, "ann_a", 2, true)]; - let leaves = build_forward_leaves(&rows, 0); - assert_eq!(leaves.len(), 2); - assert_eq!(leaves[0].0.row_count, 1); - assert_eq!(leaves[1].0.row_count, 1); - } -} diff --git a/fluree-db-binary-index/src/annotation_arena/bundle.rs b/fluree-db-binary-index/src/annotation_arena/bundle.rs deleted file mode 100644 index 22825cab37..0000000000 --- a/fluree-db-binary-index/src/annotation_arena/bundle.rs +++ /dev/null @@ -1,970 +0,0 @@ -//! Reconstruct edge-annotation rows from a slab of `f:reifies*` flakes. -//! -//! The indexer collects every `f:reifies*` fact reachable from the -//! snapshot's commit chain and hands them to [`build_arenas_from_flakes`]. -//! This module: -//! -//! 1. Groups the flakes by `(ann_subject, t, op)` so each group is a -//! single attachment-event bundle. -//! 2. Calls [`EdgeKey::from_reifies_facts`] to validate the bundle and -//! materialize an [`EdgeKey`]. -//! 3. Emits one forward and one reverse row per validated bundle. -//! 4. Sorts each list, runs the pure builder -//! ([`super::builder`]), and returns blobs ready for CAS writes -//! plus the [`AnnotationStats`] / `max_t` to seal into -//! [`AnnotationIndexRoot`]. -//! -//! Malformed bundles (missing `f:reifiesSubject`, datatype mismatch, -//! list-index facts in v1, …) are skipped with a `tracing::warn` and -//! counted; the rest of the snapshot indexes normally. This mirrors -//! the replay-validator behavior described in -//! `docs/design/edge-annotations.md` ("the on-disk arena never -//! contains rows from malformed bundles"). -//! -//! ## Inputs -//! -//! Callers pass `&[Flake]` containing **only** `f:reifies*` flakes — -//! the filter is the caller's responsibility because they have direct -//! access to predicate-SID information (the global predicate dict in -//! the indexer, or `is_reserved_reifies_predicate` for ad-hoc cases). -//! Passing non-`f:reifies*` flakes is a no-op (they're ignored by the -//! per-bundle decoder), but increases bundle-grouping cost. -//! -//! ## What this module is not -//! -//! - It does not write to CAS. The caller takes the leaf blobs from -//! the [`ArenaBuildOutput`], writes each one, then calls -//! [`super::build_forward_branch`] / [`super::build_reverse_branch`] -//! with `(summary, cid)` pairs. -//! - It does not build [`AnnotationIndexRoot`]. The caller fills in -//! `forward_branch_cid` / `reverse_branch_cid` after writing the -//! branches. - -use super::builder::{ - build_forward_leaves, build_reverse_leaves, ForwardLeafSummary, ReverseLeafSummary, -}; -use super::format::{AnnotationForwardRow, AnnotationReverseRow}; -use fluree_db_core::{AnnotationStats, EdgeKey, Flake, Sid}; -use std::collections::HashMap; - -/// Output of [`build_arenas_from_flakes`]. -/// -/// `forward_leaves` and `reverse_leaves` carry encoded blobs ready -/// for CAS writes. The caller writes each blob, collects the resulting -/// `ContentId`, then calls -/// [`super::build_forward_branch`] / [`super::build_reverse_branch`] -/// with the `(summary, cid)` pairs to encode the branch manifest. -#[derive(Debug)] -pub struct ArenaBuildOutput { - pub forward_leaves: Vec<(ForwardLeafSummary, Vec)>, - pub reverse_leaves: Vec<(ReverseLeafSummary, Vec)>, - /// Highest `t` observed across all valid bundles. `0` when zero - /// valid bundles were produced. - pub max_t: i64, - /// Aggregate stats over the validated rows. Independent counters - /// for forward / reverse (always equal in the current builder, but - /// kept distinct so future filters can diverge). - pub stats: AnnotationStats, - /// Number of malformed bundles skipped. Surfaced so callers can - /// emit a single rolled-up telemetry counter rather than one - /// per-bundle. - pub skipped_bundles: u64, - /// Number of annotation SIDs that reify more than one **live** edge - /// — a single-target-invariant violation. Zero for any data Fluree - /// itself wrote (the transaction path rejects it at stage time); - /// non-zero only for malformed `f:reifies*` data brought in by bulk - /// import (which intentionally does not validate input). Surfaced so - /// the import / reindex caller can warn without re-walking the arena. - pub multi_target_annotations: u64, -} - -/// Reconstruct + sort + chunk forward / reverse arena rows from a slab -/// of `f:reifies*` flakes. -/// -/// `target_rows_per_leaf` controls leaf chunking; pass -/// [`DEFAULT_TARGET_ROWS_PER_LEAF`] when in doubt. -pub fn build_arenas_from_flakes(flakes: &[Flake], target_rows_per_leaf: usize) -> ArenaBuildOutput { - // Group by (flake_graph, ann_sid, t, op) — each group is one - // bundle. The flake-level graph is part of the key because the - // writer convention is that `f:reifies*` flakes for an edge in - // graph G are themselves asserted in graph G; folding two graphs - // together at this stage would let a pathological cross-graph - // collision merge into a single (wrong-graph) bundle. `ann_sid` - // is the *subject* of every `f:reifies*` flake — the annotation's - // id, not the reified edge's subject. - let mut groups: HashMap<(Option, Sid, i64, bool), Vec> = HashMap::new(); - for f in flakes { - groups - .entry((f.g.clone(), f.s.clone(), f.t, f.op)) - .or_default() - .push(f.clone()); - } - - let mut forward_rows: Vec = Vec::with_capacity(groups.len()); - let mut reverse_rows: Vec = Vec::with_capacity(groups.len()); - let mut skipped: u64 = 0; - let mut max_t: i64 = 0; - - for ((bundle_g, ann_sid, t, op), bundle) in groups { - match EdgeKey::from_reifies_facts(&bundle) { - Ok(edge) => { - // Cross-check: the graph the bundle was *asserted in* - // (flake-level `g`) must match the graph the bundle - // *reifies* (`EdgeKey.g`, derived from the optional - // `f:reifiesGraph` flake). Mismatches indicate either - // a malformed bundle (e.g. `f:reifiesGraph` missing on - // a named-graph edge) or tampered history. Either way, - // the safe move is the replay-validator pattern: skip - // + count, never silently file the bundle under the - // wrong graph in the arena. - if edge.g != bundle_g { - skipped += 1; - tracing::warn!( - ann_sid = ?ann_sid, - t, - op, - bundle_graph = ?bundle_g, - edge_graph = ?edge.g, - "skipping bundle: f:reifiesGraph disagrees with flake-level graph" - ); - continue; - } - if t > max_t { - max_t = t; - } - forward_rows.push(AnnotationForwardRow { - edge: edge.clone(), - ann: ann_sid.clone(), - t, - op, - }); - reverse_rows.push(AnnotationReverseRow { - ann: ann_sid, - edge, - t, - op, - }); - } - Err(err) => { - skipped += 1; - // The replay validator pattern: log + count, never fail - // the index build over a single malformed bundle. - // Surrounding non-`f:reifies` metadata stays visible as - // ordinary RDF (just without the attachment binding). - tracing::warn!( - ?err, - ?ann_sid, - t, - op, - "skipping malformed f:reifies* bundle during arena build" - ); - } - } - } - - forward_rows.sort_unstable_by(|a, b| { - a.edge - .cmp(&b.edge) - .then_with(|| a.ann.cmp(&b.ann)) - .then_with(|| a.t.cmp(&b.t)) - .then_with(|| a.op.cmp(&b.op)) - }); - reverse_rows.sort_unstable_by(|a, b| { - a.ann - .cmp(&b.ann) - .then_with(|| a.edge.cmp(&b.edge)) - .then_with(|| a.t.cmp(&b.t)) - .then_with(|| a.op.cmp(&b.op)) - }); - - let stats = compute_stats(&forward_rows, &reverse_rows); - let multi_target_annotations = report_multi_target(&forward_rows); - - let forward_leaves = build_forward_leaves(&forward_rows, target_rows_per_leaf); - let reverse_leaves = build_reverse_leaves(&reverse_rows, target_rows_per_leaf); - - ArenaBuildOutput { - forward_leaves, - reverse_leaves, - max_t, - stats, - skipped_bundles: skipped, - multi_target_annotations, - } -} - -/// `distinct_edges` and `distinct_annotations` count **live** -/// attachments only — `(edge, ann)` pairs whose latest event is -/// `op = true`. The forward slice is already sorted by -/// `(edge, ann, t, op)` (caller of [`build_arenas_from_flakes`] -/// guarantees this), so the last row in each `(edge, ann)` run is -/// the latest event for that pair. Counting "any row" would -/// overstate live state after retractions; see the field docs on -/// [`AnnotationStats`]. -/// Build forward + reverse arenas from a stream of pre-decoded -/// attachment events. -/// -/// Bypasses the bundle-reconstruction step that -/// [`build_arenas_from_flakes`] performs — callers that already -/// hold `(EdgeKey, ann_sid, t, op)` tuples (e.g. -/// `AttachmentNovelty.iter_*`, or any source that has already -/// validated `f:reifiesGraph` agreement) hand them in directly. Each -/// input event maps to one forward row + one reverse row. -/// -/// Sort, chunk, and stats handling match the [`build_arenas_from_flakes`] -/// path. The output is ready for the same CAS-write + branch-encode -/// flow. -pub fn build_arenas_from_event_pairs( - events: impl IntoIterator, - target_rows_per_leaf: usize, -) -> ArenaBuildOutput { - let mut forward_rows: Vec = Vec::new(); - let mut reverse_rows: Vec = Vec::new(); - let mut max_t: i64 = 0; - for (edge, ann, t, op) in events { - if t > max_t { - max_t = t; - } - forward_rows.push(AnnotationForwardRow { - edge: edge.clone(), - ann: ann.clone(), - t, - op, - }); - reverse_rows.push(AnnotationReverseRow { ann, edge, t, op }); - } - - forward_rows.sort_unstable_by(|a, b| { - a.edge - .cmp(&b.edge) - .then_with(|| a.ann.cmp(&b.ann)) - .then_with(|| a.t.cmp(&b.t)) - .then_with(|| a.op.cmp(&b.op)) - }); - reverse_rows.sort_unstable_by(|a, b| { - a.ann - .cmp(&b.ann) - .then_with(|| a.edge.cmp(&b.edge)) - .then_with(|| a.t.cmp(&b.t)) - .then_with(|| a.op.cmp(&b.op)) - }); - - let stats = compute_stats(&forward_rows, &reverse_rows); - let multi_target_annotations = report_multi_target(&forward_rows); - let forward_leaves = build_forward_leaves(&forward_rows, target_rows_per_leaf); - let reverse_leaves = build_reverse_leaves(&reverse_rows, target_rows_per_leaf); - - ArenaBuildOutput { - forward_leaves, - reverse_leaves, - max_t, - stats, - skipped_bundles: 0, - multi_target_annotations, - } -} - -/// Count annotation SIDs that reify more than one **live** edge — the -/// single-target invariant violation. "Live" means the latest event for -/// an `(edge, ann)` pair is an assert (`op = true`); `forward` must be -/// sorted by `(edge, ann, t, op)` as both builders leave it. Returns the -/// number of offending annotations plus the smallest such SID for the -/// diagnostic message. Retract-based re-points and idempotent -/// re-asserts do not count (they net to one live edge). Cheap: one pass -/// over the annotation-only forward rows. -fn detect_multi_target_annotations(forward: &[AnnotationForwardRow]) -> (u64, Option) { - use std::collections::HashMap; - // Counted per `(graph, annotation)`, not per annotation. The invariant is - // one edge per reifier *within a graph*: the same reifier IRI describing - // an edge in two named graphs is a state graph management produces on - // purpose, since `COPY TO ` duplicates annotated edges with their - // reifier IRIs. Counting across graphs flagged that as a violation, so a - // correct ledger warned about itself after any copy of an annotated graph. - // - // This is the read-side twin of the staging check: the same `(graph, - // reifier)` key, for the same reason, is derived at length in - // `fluree-db-transact`'s `enforce_single_target_reifiers`. Change one and - // the other has to move with it, or the arena reports as malformed a - // ledger the transaction path accepted. - let mut live_edges_per_ann: HashMap<(Option<&Sid>, &Sid), u32> = HashMap::new(); - for i in 0..forward.len() { - let last_in_group = i + 1 == forward.len() - || forward[i].edge != forward[i + 1].edge - || forward[i].ann != forward[i + 1].ann; - if last_in_group && forward[i].op { - *live_edges_per_ann - .entry((forward[i].edge.g.as_ref(), &forward[i].ann)) - .or_insert(0) += 1; - } - } - let mut offenders: Vec<&Sid> = live_edges_per_ann - .iter() - .filter(|(_, &c)| c > 1) - .map(|((_, a), _)| *a) - .collect(); - // Sort BEFORE dedup. The map is keyed by `(graph, reifier)` but the - // report names reifiers, so one reifier that violates in two graphs - // arrives here twice. `Vec::dedup` only collapses *adjacent* equals, and - // the source is `HashMap::iter`, whose order is not stable across runs — - // deduping first therefore collapsed those two entries only when the hash - // order happened to put them side by side, and the reported count for one - // fixed input varied run to run. - offenders.sort_unstable(); - offenders.dedup(); - ( - offenders.len() as u64, - offenders.first().map(|s| (*s).clone()), - ) -} - -/// Warn when the arena build observed any multi-target annotation, and -/// return the count for the build output. Keeps the two builders' -/// emission identical. -fn report_multi_target(forward: &[AnnotationForwardRow]) -> u64 { - let (count, example) = detect_multi_target_annotations(forward); - if count > 0 { - tracing::warn!( - multi_target_annotations = count, - example = ?example, - "arena build: annotation @id(s) reify multiple live edges in one graph \ - (single-target invariant violated); the reverse lookup will return \ - multiple edges for these annotations. The transaction paths refuse \ - this, so it points at f:reifies* data that did not come through one \ - — bulk import, or a commit replayed from an older build" - ); - } - count -} - -fn compute_stats( - forward: &[AnnotationForwardRow], - reverse: &[AnnotationReverseRow], -) -> AnnotationStats { - use fluree_db_core::FlakeValue; - use std::collections::HashSet; - - // Walk the rows once. For each live `(edge, ann)` pair (the - // last-in-group row where `op == true`), count one row per - // optional-slot the edge carries — multiple annotations on the - // same named-graph edge contribute separate `f:reifiesGraph` - // rows. Distinct-value counters dedupe via `HashSet`. - let mut live_edges: HashSet<&EdgeKey> = HashSet::new(); - let mut live_anns: HashSet<&Sid> = HashSet::new(); - let mut subjects: HashSet<&Sid> = HashSet::new(); - let mut predicates: HashSet<&Sid> = HashSet::new(); - let mut objects: HashSet<&FlakeValue> = HashSet::new(); - let mut graphs: HashSet<&Sid> = HashSet::new(); - let mut langs: HashSet<&str> = HashSet::new(); - let mut list_indices: HashSet = HashSet::new(); - let mut graph_rows: u64 = 0; - let mut lang_rows: u64 = 0; - let mut list_index_rows: u64 = 0; - let mut live_pairs: u64 = 0; - // Per-optional-slot ann-SID sets — the right denominator for - // ` f:reifies ?v` BoundSubject estimates. - // Under the v1 invariant each set's size equals the slot's row - // count; under multi-target it can be smaller. - let mut graph_anns: HashSet<&Sid> = HashSet::new(); - let mut lang_anns: HashSet<&Sid> = HashSet::new(); - let mut list_index_anns: HashSet<&Sid> = HashSet::new(); - - for i in 0..forward.len() { - let last_in_group = i + 1 == forward.len() - || forward[i].edge != forward[i + 1].edge - || forward[i].ann != forward[i + 1].ann; - if !(last_in_group && forward[i].op) { - continue; - } - let edge = &forward[i].edge; - let ann = &forward[i].ann; - live_edges.insert(edge); - live_anns.insert(ann); - live_pairs += 1; - - // Distinct values are per-edge (HashSet dedupes natural - // duplicates from parallel annotations). - subjects.insert(&edge.s); - predicates.insert(&edge.p); - objects.insert(&edge.o); - if let Some(g) = &edge.g { - graphs.insert(g); - } - if let Some(lang) = &edge.lang { - langs.insert(lang.as_str()); - } - if let Some(i) = edge.list_i { - list_indices.insert(i); - } - - // Row counts for optional slots are per `(edge, ann)` pair - // — each annotation on a named-graph edge contributes its - // own `f:reifiesGraph` row. Counting per distinct edge would - // under-report when parallel annotations share an endpoint. - // The per-slot ann-SID sets dedupe naturally for the - // multi-target anomaly case. - if edge.g.is_some() { - graph_rows += 1; - graph_anns.insert(ann); - } - if edge.lang.is_some() { - lang_rows += 1; - lang_anns.insert(ann); - } - if edge.list_i.is_some() { - list_index_rows += 1; - list_index_anns.insert(ann); - } - } - - // `f:reifiesDatatype` is *not* synthesized from the arena. The - // arena reconstructs `EdgeKey.dt` from the flake-level dt of - // the `f:reifiesObject` row, so the on-wire `f:reifiesDatatype` - // predicate may have zero rows (JSON-LD-compatible cascade) or - // one-per-annotation (full bundle path) and we cannot tell - // which from the arena state alone. Reporting a synthesized - // count would let `merge_annotation_stats` overwrite the real - // `IndexStats.properties` HLL with a phantom value. Leave the - // datatype counters at zero so the planner falls back to the - // HLL. - - AnnotationStats { - forward_rows: forward.len() as u64, - reverse_rows: reverse.len() as u64, - distinct_edges: live_edges.len() as u64, - distinct_annotations: live_anns.len() as u64, - live_attachment_pairs: live_pairs, - distinct_reified_subjects: subjects.len() as u64, - distinct_reified_predicates: predicates.len() as u64, - distinct_reified_objects: objects.len() as u64, - reifies_graph_rows: graph_rows, - distinct_reified_graphs: graphs.len() as u64, - distinct_graph_anns: graph_anns.len() as u64, - reifies_datatype_rows: 0, - distinct_reified_datatypes: 0, - reifies_lang_rows: lang_rows, - distinct_reified_langs: langs.len() as u64, - distinct_lang_anns: lang_anns.len() as u64, - reifies_list_index_rows: list_index_rows, - distinct_reified_list_indices: list_indices.len() as u64, - distinct_list_index_anns: list_index_anns.len() as u64, - } -} - -#[cfg(test)] -mod tests { - use super::super::builder::DEFAULT_TARGET_ROWS_PER_LEAF; - use super::*; - use fluree_db_core::{FlakeValue, Sid}; - use fluree_vocab::db as db_predicates; - - fn ns_fluree_db() -> u16 { - fluree_vocab::namespaces::FLUREE_DB - } - - fn ann_sid(name: &str) -> Sid { - Sid::new(20, name) - } - - fn ref_sid(name: &str) -> Sid { - Sid::new(11, name) - } - - fn p_reifies(suffix: &str) -> Sid { - Sid::new(ns_fluree_db(), suffix) - } - - fn id_dt() -> Sid { - fluree_db_core::id_datatype_sid() - } - - /// Build a JSON-LD-compatible 3-flake bundle (subject, predicate, - /// object) for ann `ann_name` reifying edge `(ref_sid("alice"), - /// ref_sid("worksFor"), ref_sid("acme"))` at `(t, op)`. - fn make_bundle(ann_name: &str, t: i64, op: bool) -> Vec { - let ann = ann_sid(ann_name); - vec![ - Flake::new( - ann.clone(), - p_reifies(db_predicates::REIFIES_SUBJECT), - FlakeValue::Ref(ref_sid("alice")), - id_dt(), - t, - op, - None, - ), - Flake::new( - ann.clone(), - p_reifies(db_predicates::REIFIES_PREDICATE), - FlakeValue::Ref(ref_sid("worksFor")), - id_dt(), - t, - op, - None, - ), - Flake::new( - ann, - p_reifies(db_predicates::REIFIES_OBJECT), - FlakeValue::Ref(ref_sid("acme")), - id_dt(), - t, - op, - None, - ), - ] - } - - #[test] - fn empty_input_produces_zero_rows() { - let out = build_arenas_from_flakes(&[], DEFAULT_TARGET_ROWS_PER_LEAF); - assert!(out.forward_leaves.is_empty()); - assert!(out.reverse_leaves.is_empty()); - assert_eq!(out.max_t, 0); - assert_eq!(out.stats, AnnotationStats::default()); - assert_eq!(out.skipped_bundles, 0); - } - - #[test] - fn single_assert_bundle_produces_one_forward_and_one_reverse_row() { - let flakes = make_bundle("ann_1", 5, true); - let out = build_arenas_from_flakes(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!(out.forward_leaves.len(), 1); - assert_eq!(out.reverse_leaves.len(), 1); - assert_eq!(out.forward_leaves[0].0.row_count, 1); - assert_eq!(out.reverse_leaves[0].0.row_count, 1); - assert_eq!(out.max_t, 5); - assert_eq!(out.stats.forward_rows, 1); - assert_eq!(out.stats.distinct_edges, 1); - assert_eq!(out.stats.distinct_annotations, 1); - assert_eq!(out.skipped_bundles, 0); - } - - #[test] - fn assert_then_retract_same_ann_emits_two_rows_but_zero_live() { - // Two events on the same (edge, ann): attached then retracted. - // Both rows survive (history queries need them), but the pair - // is no longer live and must not contribute to live-state - // stats. - let mut flakes = make_bundle("ann_1", 5, true); - flakes.extend(make_bundle("ann_1", 7, false)); - let out = build_arenas_from_flakes(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!(out.stats.forward_rows, 2); - assert_eq!(out.stats.reverse_rows, 2); - assert_eq!( - out.stats.distinct_edges, 0, - "edge attached then retracted is no longer live" - ); - assert_eq!( - out.stats.distinct_annotations, 0, - "ann_1 attached then retracted is no longer live" - ); - assert_eq!(out.max_t, 7); - } - - #[test] - fn distinct_stats_count_only_currently_live_pairs() { - // ann_a attached then retracted → not live; - // ann_b attached → live. Stats reflect live state, not history. - let mut flakes = Vec::new(); - flakes.extend(make_bundle("ann_a", 1, true)); - flakes.extend(make_bundle("ann_a", 2, false)); - flakes.extend(make_bundle("ann_b", 3, true)); - let out = build_arenas_from_flakes(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!(out.stats.forward_rows, 3, "all events kept for history"); - assert_eq!(out.stats.distinct_edges, 1, "edge with ann_b is live"); - assert_eq!(out.stats.distinct_annotations, 1, "only ann_b is live"); - } - - /// Same as [`make_bundle`] but the reified object is `obj`, so the - /// edge identity differs while the annotation SID can be reused. - fn make_bundle_obj(ann_name: &str, obj: &str, t: i64, op: bool) -> Vec { - let ann = ann_sid(ann_name); - vec![ - Flake::new( - ann.clone(), - p_reifies(db_predicates::REIFIES_SUBJECT), - FlakeValue::Ref(ref_sid("alice")), - id_dt(), - t, - op, - None, - ), - Flake::new( - ann.clone(), - p_reifies(db_predicates::REIFIES_PREDICATE), - FlakeValue::Ref(ref_sid("worksFor")), - id_dt(), - t, - op, - None, - ), - Flake::new( - ann, - p_reifies(db_predicates::REIFIES_OBJECT), - FlakeValue::Ref(ref_sid(obj)), - id_dt(), - t, - op, - None, - ), - ] - } - - #[test] - fn single_target_data_reports_zero_multi_target() { - let flakes = make_bundle("ann_1", 5, true); - let out = build_arenas_from_flakes(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!(out.multi_target_annotations, 0); - } - - #[test] - fn same_ann_two_live_edges_across_t_is_flagged_multi_target() { - // ann_1 reifies (alice, worksFor, acme) at t=1 AND - // (alice, worksFor, bob) at t=2 — both asserted, neither - // retracted. Each is a distinct, individually-valid bundle - // (the per-(ann,t) decode can't see the conflict), so this is - // exactly the cross-commit multi-target a bulk import can - // introduce. The arena builder must flag it. - let mut flakes = make_bundle_obj("ann_1", "acme", 1, true); - flakes.extend(make_bundle_obj("ann_1", "bob", 2, true)); - let out = build_arenas_from_flakes(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!( - out.multi_target_annotations, 1, - "one annotation reifies two live edges" - ); - } - - #[test] - fn retract_then_repoint_is_not_flagged() { - // Legitimate re-point: attach to acme, retract it, attach to bob. - // Net live state is a single edge, so it must not be flagged. - let mut flakes = make_bundle_obj("ann_1", "acme", 1, true); - flakes.extend(make_bundle_obj("ann_1", "acme", 2, false)); - flakes.extend(make_bundle_obj("ann_1", "bob", 3, true)); - let out = build_arenas_from_flakes(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!(out.multi_target_annotations, 0); - } - - #[test] - fn idempotent_reassert_is_not_flagged() { - // Re-asserting the identical attachment at a later t nets to one - // live edge. - let mut flakes = make_bundle_obj("ann_1", "acme", 1, true); - flakes.extend(make_bundle_obj("ann_1", "acme", 2, true)); - let out = build_arenas_from_flakes(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!(out.multi_target_annotations, 0); - } - - #[test] - fn bundle_with_mismatched_flake_graph_is_skipped() { - // Manually build a bundle whose flakes are asserted in a - // named graph but whose `f:reifiesGraph` is omitted (so - // `EdgeKey::from_reifies_facts` infers default-graph). This - // is the malformed shape the cross-check guards against — - // accepting it would file the annotation under the wrong - // graph in the arena. - let ann = ann_sid("ann_x"); - let bundle_graph = Some(ref_sid("graph_a")); - let mk = |p: &str, o: FlakeValue, dt: Sid| { - // `Flake::new_in_graph` to mark these as living in - // graph_a. - Flake::new_in_graph( - bundle_graph.clone().unwrap(), - ann.clone(), - p_reifies(p), - o, - dt, - 1, - true, - None, - ) - }; - let flakes = vec![ - mk( - db_predicates::REIFIES_SUBJECT, - FlakeValue::Ref(ref_sid("alice")), - id_dt(), - ), - mk( - db_predicates::REIFIES_PREDICATE, - FlakeValue::Ref(ref_sid("worksFor")), - id_dt(), - ), - mk( - db_predicates::REIFIES_OBJECT, - FlakeValue::Ref(ref_sid("acme")), - id_dt(), - ), - // Note: NO `f:reifiesGraph` flake. EdgeKey.g will decode - // as None (default graph), but the bundle was asserted in - // graph_a. Cross-check must reject. - ]; - let out = build_arenas_from_flakes(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!(out.skipped_bundles, 1); - assert_eq!(out.stats.forward_rows, 0); - assert!(out.forward_leaves.is_empty()); - assert!(out.reverse_leaves.is_empty()); - } - - #[test] - fn bundle_with_matching_named_graph_is_accepted() { - // Same as above, but with the `f:reifiesGraph` flake present - // and pointing to graph_a. Cross-check passes; the row lands - // in the arena with `EdgeKey.g = Some(graph_a)`. - let ann = ann_sid("ann_y"); - let g = ref_sid("graph_a"); - let mk = |p: &str, o: FlakeValue, dt: Sid| { - Flake::new_in_graph(g.clone(), ann.clone(), p_reifies(p), o, dt, 1, true, None) - }; - let flakes = vec![ - mk( - db_predicates::REIFIES_GRAPH, - FlakeValue::Ref(g.clone()), - id_dt(), - ), - mk( - db_predicates::REIFIES_SUBJECT, - FlakeValue::Ref(ref_sid("alice")), - id_dt(), - ), - mk( - db_predicates::REIFIES_PREDICATE, - FlakeValue::Ref(ref_sid("worksFor")), - id_dt(), - ), - mk( - db_predicates::REIFIES_OBJECT, - FlakeValue::Ref(ref_sid("acme")), - id_dt(), - ), - ]; - let out = build_arenas_from_flakes(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!(out.skipped_bundles, 0); - assert_eq!(out.stats.forward_rows, 1); - assert_eq!(out.stats.distinct_edges, 1); - // Recovered EdgeKey carries the named graph. - let leaf_blob = &out.forward_leaves[0].1; - let leaf = - crate::annotation_arena::format::AnnotationForwardLeaf::decode(leaf_blob).unwrap(); - assert_eq!(leaf.rows[0].edge.g, Some(g)); - } - - #[test] - fn malformed_bundle_skipped_with_counter() { - // Bundle missing f:reifiesSubject. Surrounding good bundle - // must still produce a row. - let good = make_bundle("ann_good", 1, true); - let bad = vec![Flake::new( - ann_sid("ann_bad"), - p_reifies(db_predicates::REIFIES_PREDICATE), - FlakeValue::Ref(ref_sid("worksFor")), - id_dt(), - 1, - true, - None, - )]; - let mut all = good; - all.extend(bad); - let out = build_arenas_from_flakes(&all, DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!(out.skipped_bundles, 1); - assert_eq!(out.stats.forward_rows, 1, "good bundle still emitted"); - } - - #[test] - fn multiple_bundles_sort_into_arena_order() { - // Three bundles, distinct edges. The forward arena sort key is - // (edge, ann, t, op); reverse is (ann, edge, t, op). We assert - // the sort by inspecting the first/last keys of the produced - // single leaf each. - let mut flakes = Vec::new(); - flakes.extend(make_bundle("ann_3", 1, true)); - flakes.extend(make_bundle("ann_1", 2, true)); - flakes.extend(make_bundle("ann_2", 3, true)); - // make_bundle reuses the same edge for every annotation, so - // forward arena rows differ only in the annotation Sid. - let out = build_arenas_from_flakes(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - assert_eq!(out.forward_leaves.len(), 1); - let summary = &out.forward_leaves[0].0; - // Sorted ascending by ann_sid: ann_1 < ann_2 < ann_3. - assert_eq!(summary.first_ann, ann_sid("ann_1")); - assert_eq!(summary.last_ann, ann_sid("ann_3")); - - let rev = &out.reverse_leaves[0].0; - assert_eq!(rev.first_ann, ann_sid("ann_1")); - assert_eq!(rev.last_ann, ann_sid("ann_3")); - } - - #[test] - fn rows_chunk_into_multiple_leaves_when_exceeding_target() { - // 5 distinct annotations against the same edge → 5 forward - // rows; with target=2 → 3 leaves. - let mut flakes = Vec::new(); - for i in 0..5 { - flakes.extend(make_bundle(&format!("ann_{i}"), i64::from(i) + 1, true)); - } - let out = build_arenas_from_flakes(&flakes, 2); - assert_eq!(out.forward_leaves.len(), 3); - let total: u64 = out.forward_leaves.iter().map(|(s, _)| s.row_count).sum(); - assert_eq!(total, 5); - } - - #[test] - fn forward_and_reverse_blobs_decode_back_to_original_rows() { - use crate::annotation_arena::format::{AnnotationForwardLeaf, AnnotationReverseLeaf}; - - let mut flakes = Vec::new(); - flakes.extend(make_bundle("ann_a", 1, true)); - flakes.extend(make_bundle("ann_b", 2, true)); - flakes.extend(make_bundle("ann_a", 3, false)); - let out = build_arenas_from_flakes(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - - let mut decoded_forward = Vec::new(); - for (_, blob) in &out.forward_leaves { - decoded_forward.extend(AnnotationForwardLeaf::decode(blob).unwrap().rows); - } - let mut decoded_reverse = Vec::new(); - for (_, blob) in &out.reverse_leaves { - decoded_reverse.extend(AnnotationReverseLeaf::decode(blob).unwrap().rows); - } - assert_eq!(decoded_forward.len(), 3); - assert_eq!(decoded_reverse.len(), 3); - - // Cross-check: every forward (edge, ann, t, op) appears in - // reverse with the same fields. - for fwd in &decoded_forward { - let found = decoded_reverse.iter().any(|rev| { - rev.edge == fwd.edge && rev.ann == fwd.ann && rev.t == fwd.t && rev.op == fwd.op - }); - assert!(found, "forward row missing from reverse: {fwd:?}"); - } - } - - #[test] - fn event_pairs_path_matches_flakes_path() { - // Building from pre-decoded events must produce the same - // arena rows as the bundle-driven path. Same input data, two - // forms. - let mut flakes = Vec::new(); - flakes.extend(make_bundle("ann_a", 1, true)); - flakes.extend(make_bundle("ann_b", 2, true)); - let from_flakes = build_arenas_from_flakes(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - - // Equivalent pre-decoded events. `make_bundle` reuses the - // same edge for each annotation, so we reconstruct it from - // the bundle once. - let edge = EdgeKey::from_reifies_facts(&make_bundle("ann_a", 1, true)).unwrap(); - let events = vec![ - (edge.clone(), ann_sid("ann_a"), 1, true), - (edge, ann_sid("ann_b"), 2, true), - ]; - let from_events = build_arenas_from_event_pairs(events, DEFAULT_TARGET_ROWS_PER_LEAF); - - assert_eq!(from_flakes.stats, from_events.stats); - assert_eq!(from_flakes.max_t, from_events.max_t); - assert_eq!( - from_flakes.forward_leaves.len(), - from_events.forward_leaves.len() - ); - } - - #[test] - fn event_pairs_empty_input_produces_empty_arena() { - let out = build_arenas_from_event_pairs(std::iter::empty(), DEFAULT_TARGET_ROWS_PER_LEAF); - assert!(out.forward_leaves.is_empty()); - assert!(out.reverse_leaves.is_empty()); - assert_eq!(out.max_t, 0); - assert_eq!(out.skipped_bundles, 0); - } -} - -#[cfg(test)] -mod multi_target_graph_tests { - use super::*; - use fluree_db_core::{FlakeValue, Sid}; - - fn row(graph: Option<&str>, ann: &str, obj: &str) -> AnnotationForwardRow { - AnnotationForwardRow { - edge: fluree_db_core::edge::EdgeKey { - g: graph.map(|g| Sid::new(14, g)), - s: Sid::new(11, "alice"), - p: Sid::new(11, "knows"), - o: FlakeValue::Ref(Sid::new(11, obj)), - dt: fluree_db_core::id_datatype_sid(), - lang: None, - list_i: None, - }, - ann: Sid::new(20, ann), - t: 1, - op: true, - } - } - - /// One reifier on the same edge in two graphs is not a violation. - /// - /// `COPY TO ` duplicates annotated edges with their reifier IRIs, - /// so this is a state graph management produces on purpose. Counting live - /// edges per annotation across graphs flagged it, which meant a correct - /// ledger warned about itself after any copy of an annotated graph — and - /// the warning said the transaction path rejects this, which it does not. - #[test] - fn the_same_reifier_in_two_graphs_is_not_multi_target() { - let mut forward = vec![ - row(Some("g1"), "claim1", "bob"), - row(Some("g2"), "claim1", "bob"), - ]; - forward.sort_by(|a, b| (&a.edge, &a.ann, a.t, a.op).cmp(&(&b.edge, &b.ann, b.t, b.op))); - let (count, _) = detect_multi_target_annotations(&forward); - assert_eq!(count, 0, "one edge per graph is the invariant"); - } - - /// Two different edges in the SAME graph is still a violation. - #[test] - fn two_edges_in_one_graph_is_still_multi_target() { - let mut forward = vec![ - row(Some("g1"), "claim1", "bob"), - row(Some("g1"), "claim1", "carol"), - ]; - forward.sort_by(|a, b| (&a.edge, &a.ann, a.t, a.op).cmp(&(&b.edge, &b.ann, b.t, b.op))); - let (count, example) = detect_multi_target_annotations(&forward); - assert_eq!(count, 1, "two live edges for one reifier in one graph"); - assert_eq!(example, Some(Sid::new(20, "claim1"))); - } - - /// And in the default graph, where `g` is `None` on both sides. - #[test] - fn two_edges_in_the_default_graph_is_still_multi_target() { - let mut forward = vec![row(None, "claim1", "bob"), row(None, "claim1", "carol")]; - forward.sort_by(|a, b| (&a.edge, &a.ann, a.t, a.op).cmp(&(&b.edge, &b.ann, b.t, b.op))); - let (count, _) = detect_multi_target_annotations(&forward); - assert_eq!(count, 1); - } - - /// One reifier that offends in two graphs is still ONE offender, on every - /// run. - /// - /// The counter is keyed by `(graph, reifier)` while the report names - /// reifiers, so `claim1` reaches the offender list once per graph. Because - /// the list is built from `HashMap::iter`, whose order is reseeded per - /// map, deduping before sorting collapsed that pair only when the hash - /// order happened to place the two entries adjacently — the same fixed - /// input reported 2 or 3 depending on the run. A single pass is therefore - /// a coin flip, and the loop is what makes the assertion mean something. - #[test] - fn the_offender_count_does_not_vary_between_runs() { - for _ in 0..500 { - let mut forward = vec![ - row(Some("g1"), "claim1", "bob"), - row(Some("g1"), "claim1", "carol"), - row(Some("g2"), "claim1", "bob"), - row(Some("g2"), "claim1", "carol"), - row(Some("g1"), "claim2", "bob"), - row(Some("g1"), "claim2", "carol"), - ]; - forward.sort_by(|a, b| (&a.edge, &a.ann, a.t, a.op).cmp(&(&b.edge, &b.ann, b.t, b.op))); - let (count, _) = detect_multi_target_annotations(&forward); - assert_eq!( - count, 2, - "claim1 offends in two graphs but is one offender; claim2 is the other" - ); - } - } -} diff --git a/fluree-db-binary-index/src/annotation_arena/format.rs b/fluree-db-binary-index/src/annotation_arena/format.rs deleted file mode 100644 index 4fdc7964bb..0000000000 --- a/fluree-db-binary-index/src/annotation_arena/format.rs +++ /dev/null @@ -1,450 +0,0 @@ -//! Wire format for the edge-annotation forward/reverse arenas. -//! -//! ## Blob shape (forward leaf, reverse leaf, forward branch, reverse branch) -//! -//! ```text -//! [magic: 4B][version: u8][flags: u8][reserved: u16 = 0][body_len: u32 LE] -//! [body: postcard-encoded payload, body_len bytes] -//! ``` -//! -//! Header is fixed at 12 bytes. The payload is a CBOR-encoded -//! [`AnnotationForwardLeaf`] / [`AnnotationReverseLeaf`] / -//! [`AnnotationForwardBranch`] / [`AnnotationReverseBranch`] respectively. -//! -//! CBOR (via `ciborium`) was chosen over postcard because [`FlakeValue`] -//! uses `#[serde(untagged)]` for the polymorphic object value, which -//! non-self-describing formats like postcard cannot deserialize. -//! -//! ## Sort orders -//! -//! - **Forward leaf rows**: `(EdgeKey, ann_sid, t, op)` — ascending. -//! The leaf's first/last keys (held in the branch entry) are -//! `(EdgeKey, ann_sid)` projections. -//! - **Reverse leaf rows**: `(ann_sid, EdgeKey, t, op)` — ascending. -//! The branch routes on `ann_sid`. -//! -//! ## Empty blobs -//! -//! An empty leaf encodes to `body = postcard(empty Vec)` and is valid. -//! An empty branch likewise. Builders that produce no rows should still -//! emit an empty branch + leaf rather than skipping the section, so -//! `IndexRoot.annotation_index = None` keeps its "zero attachments" -//! correctness guarantee. -//! -//! See `docs/design/edge-annotations.md` (Sidecar arena layout) for -//! the design contract. - -use fluree_db_core::{ContentId, EdgeKey, Sid}; -use serde::{Deserialize, Serialize}; -use thiserror::Error; - -// Re-exported from `fluree-db-core` so callers continue to import from -// this module. The structs themselves live in core because -// `LedgerSnapshot` holds `Option`. -pub use fluree_db_core::{AnnotationIndexRoot, AnnotationStats}; - -/// Magic bytes for an edge-annotation forward branch (`EAFB1`). -pub const FORWARD_BRANCH_MAGIC: [u8; 4] = *b"EAFB"; - -/// Magic bytes for an edge-annotation forward leaf (`EAFL1`). -pub const FORWARD_LEAF_MAGIC: [u8; 4] = *b"EAFL"; - -/// Magic bytes for an edge-annotation reverse branch (`EARB1`). -pub const REVERSE_BRANCH_MAGIC: [u8; 4] = *b"EARB"; - -/// Magic bytes for an edge-annotation reverse leaf (`EARL1`). -pub const REVERSE_LEAF_MAGIC: [u8; 4] = *b"EARL"; - -/// Wire-format version for all four arena blob kinds. -pub const ARENA_VERSION: u8 = 1; - -/// Header length (magic + version + flags + reserved + body_len). -pub const ARENA_HEADER_LEN: usize = 12; - -// ── Forward arena ─────────────────────────────────────────────────────────── - -/// One row in a forward-arena leaf. -/// -/// `(edge, ann)` are the routing key; `(t, op)` records the event so -/// history queries can replay the timeline. The on-disk sort order is -/// `(edge, ann, t, op)`. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct AnnotationForwardRow { - pub edge: EdgeKey, - pub ann: Sid, - pub t: i64, - /// `true` = assertion (attach), `false` = retraction (detach). - pub op: bool, -} - -/// Forward-arena leaf body. Rows are sorted by `(edge, ann, t, op)`. -#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] -pub struct AnnotationForwardLeaf { - pub rows: Vec, -} - -/// One leaf entry in a forward-arena branch. -/// -/// `first_key` / `last_key` are the inclusive `(edge, ann)` bounds for -/// the leaf, stored explicitly so the branch can binary-search on -/// `EdgeKey` without loading the leaf body. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct AnnotationForwardBranchEntry { - pub first_edge: EdgeKey, - pub first_ann: Sid, - pub last_edge: EdgeKey, - pub last_ann: Sid, - pub row_count: u64, - pub leaf_cid: ContentId, -} - -/// Forward-arena branch body. Entries are sorted by `(first_edge, first_ann)`. -#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] -pub struct AnnotationForwardBranch { - pub leaves: Vec, -} - -// ── Reverse arena ─────────────────────────────────────────────────────────── - -/// One row in a reverse-arena leaf. -/// -/// On-disk sort order is `(ann, edge, t, op)`. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct AnnotationReverseRow { - pub ann: Sid, - pub edge: EdgeKey, - pub t: i64, - pub op: bool, -} - -/// Reverse-arena leaf body. Rows are sorted by `(ann, edge, t, op)`. -#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] -pub struct AnnotationReverseLeaf { - pub rows: Vec, -} - -/// One leaf entry in a reverse-arena branch. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct AnnotationReverseBranchEntry { - pub first_ann: Sid, - pub first_edge: EdgeKey, - pub last_ann: Sid, - pub last_edge: EdgeKey, - pub row_count: u64, - pub leaf_cid: ContentId, -} - -/// Reverse-arena branch body. Entries are sorted by `(first_ann, first_edge)`. -#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] -pub struct AnnotationReverseBranch { - pub leaves: Vec, -} - -// ── Wire encoding ─────────────────────────────────────────────────────────── - -#[derive(Debug, Error)] -pub enum DecodeError { - #[error("blob shorter than {ARENA_HEADER_LEN}-byte header (got {0})")] - TruncatedHeader(usize), - #[error("magic mismatch: expected {expected:?}, got {got:?}")] - BadMagic { expected: [u8; 4], got: [u8; 4] }, - #[error("unsupported version {0} (this build expects {ARENA_VERSION})")] - UnsupportedVersion(u8), - #[error("body length {declared} exceeds available bytes {available}")] - TruncatedBody { declared: usize, available: usize }, - #[error("cbor decode error: {0}")] - Cbor(String), - #[error("non-zero reserved bytes: {0:#06x}")] - NonZeroReserved(u16), -} - -fn encode_blob(magic: [u8; 4], payload: &T) -> Vec { - let mut body = Vec::new(); - ciborium::ser::into_writer(payload, &mut body) - .expect("ciborium serialization to Vec is infallible"); - let mut out = Vec::with_capacity(ARENA_HEADER_LEN + body.len()); - out.extend_from_slice(&magic); - out.push(ARENA_VERSION); - out.push(0u8); // flags - out.extend_from_slice(&0u16.to_le_bytes()); // reserved - out.extend_from_slice(&(body.len() as u32).to_le_bytes()); - out.extend_from_slice(&body); - out -} - -fn decode_blob Deserialize<'de>>( - bytes: &[u8], - expected_magic: [u8; 4], -) -> Result { - if bytes.len() < ARENA_HEADER_LEN { - return Err(DecodeError::TruncatedHeader(bytes.len())); - } - let magic: [u8; 4] = bytes[0..4].try_into().unwrap(); - if magic != expected_magic { - return Err(DecodeError::BadMagic { - expected: expected_magic, - got: magic, - }); - } - let version = bytes[4]; - if version != ARENA_VERSION { - return Err(DecodeError::UnsupportedVersion(version)); - } - let _flags = bytes[5]; - let reserved = u16::from_le_bytes(bytes[6..8].try_into().unwrap()); - if reserved != 0 { - return Err(DecodeError::NonZeroReserved(reserved)); - } - let body_len = u32::from_le_bytes(bytes[8..12].try_into().unwrap()) as usize; - let body_end = ARENA_HEADER_LEN - .checked_add(body_len) - .ok_or(DecodeError::TruncatedBody { - declared: body_len, - available: bytes.len().saturating_sub(ARENA_HEADER_LEN), - })?; - if body_end > bytes.len() { - return Err(DecodeError::TruncatedBody { - declared: body_len, - available: bytes.len().saturating_sub(ARENA_HEADER_LEN), - }); - } - let payload = ciborium::de::from_reader::(&bytes[ARENA_HEADER_LEN..body_end]) - .map_err(|e| DecodeError::Cbor(e.to_string()))?; - Ok(payload) -} - -impl AnnotationForwardLeaf { - pub fn encode(&self) -> Vec { - encode_blob(FORWARD_LEAF_MAGIC, self) - } - pub fn decode(bytes: &[u8]) -> Result { - decode_blob(bytes, FORWARD_LEAF_MAGIC) - } -} - -impl AnnotationForwardBranch { - pub fn encode(&self) -> Vec { - encode_blob(FORWARD_BRANCH_MAGIC, self) - } - pub fn decode(bytes: &[u8]) -> Result { - decode_blob(bytes, FORWARD_BRANCH_MAGIC) - } -} - -impl AnnotationReverseLeaf { - pub fn encode(&self) -> Vec { - encode_blob(REVERSE_LEAF_MAGIC, self) - } - pub fn decode(bytes: &[u8]) -> Result { - decode_blob(bytes, REVERSE_LEAF_MAGIC) - } -} - -impl AnnotationReverseBranch { - pub fn encode(&self) -> Vec { - encode_blob(REVERSE_BRANCH_MAGIC, self) - } - pub fn decode(bytes: &[u8]) -> Result { - decode_blob(bytes, REVERSE_BRANCH_MAGIC) - } -} - -#[cfg(test)] -mod tests { - use super::*; - use fluree_db_core::{ContentId, ContentKind, FlakeValue, Sid}; - use fluree_vocab::xsd; - - fn sid(ns: u16, name: &str) -> Sid { - Sid::new(ns, name) - } - - fn cid(seed: &str) -> ContentId { - ContentId::new(ContentKind::AnnotationForwardLeaf, seed.as_bytes()) - } - - fn sample_edge(idx: u8) -> EdgeKey { - EdgeKey { - g: if idx.is_multiple_of(2) { - None - } else { - Some(sid(10, &format!("g{idx}"))) - }, - s: sid(11, &format!("s{idx}")), - p: sid(12, &format!("p{idx}")), - o: FlakeValue::Ref(sid(11, &format!("o{idx}"))), - dt: Sid::new(0, xsd::ANY_URI), - lang: None, - list_i: None, - } - } - - #[test] - fn forward_leaf_roundtrip_empty() { - let leaf = AnnotationForwardLeaf::default(); - let bytes = leaf.encode(); - // header-only bodies still carry a CBOR-encoded empty Vec. - assert!(bytes.len() >= ARENA_HEADER_LEN); - assert_eq!(&bytes[..4], &FORWARD_LEAF_MAGIC); - let decoded = AnnotationForwardLeaf::decode(&bytes).unwrap(); - assert_eq!(decoded, leaf); - } - - #[test] - fn forward_leaf_roundtrip_multi() { - let leaf = AnnotationForwardLeaf { - rows: vec![ - AnnotationForwardRow { - edge: sample_edge(0), - ann: sid(20, "ann0"), - t: 1, - op: true, - }, - AnnotationForwardRow { - edge: sample_edge(0), - ann: sid(20, "ann0"), - t: 2, - op: false, - }, - AnnotationForwardRow { - edge: sample_edge(1), - ann: sid(20, "ann1"), - t: 3, - op: true, - }, - ], - }; - let bytes = leaf.encode(); - let decoded = AnnotationForwardLeaf::decode(&bytes).unwrap(); - assert_eq!(decoded, leaf); - } - - #[test] - fn forward_branch_roundtrip() { - let branch = AnnotationForwardBranch { - leaves: vec![AnnotationForwardBranchEntry { - first_edge: sample_edge(0), - first_ann: sid(20, "ann0"), - last_edge: sample_edge(1), - last_ann: sid(20, "ann1"), - row_count: 3, - leaf_cid: cid("forward-leaf-0"), - }], - }; - let bytes = branch.encode(); - assert_eq!(&bytes[..4], &FORWARD_BRANCH_MAGIC); - let decoded = AnnotationForwardBranch::decode(&bytes).unwrap(); - assert_eq!(decoded, branch); - } - - #[test] - fn reverse_leaf_roundtrip() { - let leaf = AnnotationReverseLeaf { - rows: vec![ - AnnotationReverseRow { - ann: sid(20, "ann0"), - edge: sample_edge(0), - t: 1, - op: true, - }, - AnnotationReverseRow { - ann: sid(20, "ann1"), - edge: sample_edge(1), - t: 2, - op: true, - }, - ], - }; - let bytes = leaf.encode(); - assert_eq!(&bytes[..4], &REVERSE_LEAF_MAGIC); - let decoded = AnnotationReverseLeaf::decode(&bytes).unwrap(); - assert_eq!(decoded, leaf); - } - - #[test] - fn reverse_branch_roundtrip() { - let branch = AnnotationReverseBranch { - leaves: vec![AnnotationReverseBranchEntry { - first_ann: sid(20, "ann0"), - first_edge: sample_edge(0), - last_ann: sid(20, "ann1"), - last_edge: sample_edge(1), - row_count: 2, - leaf_cid: cid("reverse-leaf-0"), - }], - }; - let bytes = branch.encode(); - let decoded = AnnotationReverseBranch::decode(&bytes).unwrap(); - assert_eq!(decoded, branch); - } - - #[test] - fn rejects_truncated_header() { - let err = AnnotationForwardLeaf::decode(&[0u8; 4]).unwrap_err(); - assert!(matches!(err, DecodeError::TruncatedHeader(4))); - } - - #[test] - fn rejects_wrong_magic() { - let bytes = AnnotationForwardLeaf::default().encode(); - // Decoding forward bytes as reverse must fail loudly. - let err = AnnotationReverseLeaf::decode(&bytes).unwrap_err(); - assert!(matches!(err, DecodeError::BadMagic { .. })); - } - - #[test] - fn rejects_unsupported_version() { - let mut bytes = AnnotationForwardLeaf::default().encode(); - bytes[4] = ARENA_VERSION + 1; - let err = AnnotationForwardLeaf::decode(&bytes).unwrap_err(); - assert!(matches!(err, DecodeError::UnsupportedVersion(_))); - } - - #[test] - fn rejects_truncated_body() { - let bytes = AnnotationForwardLeaf { - rows: vec![AnnotationForwardRow { - edge: sample_edge(0), - ann: sid(20, "ann0"), - t: 1, - op: true, - }], - } - .encode(); - // Lop off the last 3 bytes of the body. - let truncated = &bytes[..bytes.len() - 3]; - let err = AnnotationForwardLeaf::decode(truncated).unwrap_err(); - assert!(matches!(err, DecodeError::TruncatedBody { .. })); - } - - #[test] - fn rejects_non_zero_reserved() { - let mut bytes = AnnotationForwardLeaf::default().encode(); - bytes[6] = 0xff; - let err = AnnotationForwardLeaf::decode(&bytes).unwrap_err(); - assert!(matches!(err, DecodeError::NonZeroReserved(_))); - } - - #[test] - fn annotation_index_root_construct_and_clone() { - // No on-disk format yet for AnnotationIndexRoot itself — it lives - // inline in IndexRoot. Slice 2 wires it into FIR6 encoding. This - // test pins the struct contract so accidental field renames here - // surface immediately. - let root = AnnotationIndexRoot { - version: ARENA_VERSION, - max_t: 42, - forward_branch_cid: cid("fwd"), - reverse_branch_cid: cid("rev"), - stats: AnnotationStats { - forward_rows: 10, - reverse_rows: 10, - distinct_edges: 4, - distinct_annotations: 6, - ..Default::default() - }, - }; - let cloned = root.clone(); - assert_eq!(root, cloned); - } -} diff --git a/fluree-db-binary-index/src/annotation_arena/mod.rs b/fluree-db-binary-index/src/annotation_arena/mod.rs deleted file mode 100644 index 98b79b936e..0000000000 --- a/fluree-db-binary-index/src/annotation_arena/mod.rs +++ /dev/null @@ -1,53 +0,0 @@ -//! Edge-annotation arenas — secondary indexes over `f:reifies*` system facts. -//! -//! Two arenas, both content-addressed and treated like dictionary trees: -//! -//! - **Forward arena** (`EAFB1` / `EAFL1`): `EdgeKey -> ann_sid` lookup. Used by -//! the read-side hydrator to attach annotation metadata to base-edge rows, -//! and by the retract cascade to find dependent attachments. -//! - **Reverse arena** (`EARB1` / `EARL1`): `ann_sid -> EdgeKey` lookup. Used by -//! `@reifies`-rooted queries and SPARQL-star `?ann rdf:reifies <<...>>` shape. -//! -//! Rows carry `(t, op)` so history queries surface the same attach/detach -//! events as flake history. Visibility filtering is applied at read time; -//! readers return iterators because the underlying store is a multimap. -//! -//! ## Layering -//! -//! - **Format** (`format`) — wire bytes, magic numbers, codec routing, -//! roundtrip-only. No I/O. -//! - **Builder** (`builder`) — chunk pre-sorted rows into leaves and -//! manifest entries; pure (no I/O), so callers interleave the CAS -//! write between leaf encode and branch encode. -//! - Reader / incremental-merge live in sibling modules added in -//! later slices of M2b. -//! -//! ## Empty-vs-absent semantics -//! -//! - `IndexRoot.annotation_index = None` paired with -//! `IndexRoot.has_annotations = false` is a hard guarantee that the -//! indexed snapshot has zero annotation attachments. See the truth -//! table in [`fluree_db_core::annotation_index`]. -//! - Empty branches/leaves are valid and decode to empty row vectors. -//! -//! See `docs/design/edge-annotations.md` for the storage shape, -//! sidecar arena layout, and replay-validation contract. - -pub mod builder; -pub mod bundle; -pub mod format; -pub mod reader; - -pub use builder::{ - build_forward_branch, build_forward_leaves, build_reverse_branch, build_reverse_leaves, - forward_arena_stats, ForwardLeafSummary, ReverseLeafSummary, DEFAULT_TARGET_ROWS_PER_LEAF, -}; -pub use bundle::{build_arenas_from_event_pairs, build_arenas_from_flakes, ArenaBuildOutput}; -pub use format::{ - AnnotationForwardBranch, AnnotationForwardBranchEntry, AnnotationForwardLeaf, - AnnotationForwardRow, AnnotationIndexRoot, AnnotationReverseBranch, - AnnotationReverseBranchEntry, AnnotationReverseLeaf, AnnotationReverseRow, AnnotationStats, - DecodeError, FORWARD_BRANCH_MAGIC, FORWARD_LEAF_MAGIC, REVERSE_BRANCH_MAGIC, - REVERSE_LEAF_MAGIC, -}; -pub use reader::AnnotationArenaReader; diff --git a/fluree-db-binary-index/src/annotation_arena/reader.rs b/fluree-db-binary-index/src/annotation_arena/reader.rs deleted file mode 100644 index e1a9b9ec24..0000000000 --- a/fluree-db-binary-index/src/annotation_arena/reader.rs +++ /dev/null @@ -1,1164 +0,0 @@ -//! Lazy reader for the edge-annotation arenas. -//! -//! Wraps an [`AnnotationIndexRoot`] + a `ContentStore` and provides -//! visibility-filtered lookups in either direction: -//! -//! - [`AnnotationArenaReader::current_annotations_for`] — given an -//! edge, return the annotation subjects whose latest event at or -//! before `as_of_t` is `op = true`. -//! - [`AnnotationArenaReader::current_targets_for`] — given an -//! annotation subject, return the edges it currently reifies. -//! - [`AnnotationArenaReader::target_history_for`] — given an -//! annotation, return every `(edge, t, op)` event (history view). -//! -//! ## Loading strategy -//! -//! Branches are loaded lazily on first lookup and cached for the -//! reader's lifetime. Leaves are loaded only when the branch -//! binary-search resolves to a candidate leaf, and cached by CID -//! (multiple lookups of the same hot leaf hit the cache). -//! -//! ## Visibility model -//! -//! Forward-arena rows are sorted by `(edge, ann, t, op)`. Within a -//! single `(edge, ann)` group: -//! -//! - The latest row whose `t <= as_of_t` is the visible event. -//! - If that event is `op = true`, the annotation is live at that t. -//! - If `op = false` (or no event ≤ as_of_t exists), the annotation -//! is not visible. -//! -//! Same model for reverse, swapping the routing key. -//! -//! ## Merging with novelty -//! -//! The arena holds events committed up to its `max_t`; an in-memory -//! attachment overlay (e.g. `fluree_db_novelty::AttachmentNovelty`) -//! holds events from after that point. The merged variants -//! ([`AnnotationArenaReader::current_annotations_merged`] / -//! [`AnnotationArenaReader::current_targets_merged`]) accept -//! pre-collected `(other, t, op)` slices from the overlay and apply -//! the same visibility filter across the union of events. This keeps -//! `fluree-db-binary-index` decoupled from `fluree-db-novelty`: -//! callers in higher-layer crates do the per-key event collection -//! and hand it down. -//! -//! ## What this module is not -//! -//! - It does not validate the on-disk arena. Slice 5 adds the -//! storage-inspector path. - -use super::format::{ - AnnotationForwardBranch, AnnotationForwardBranchEntry, AnnotationForwardLeaf, - AnnotationReverseBranch, AnnotationReverseBranchEntry, AnnotationReverseLeaf, -}; -use fluree_db_core::{ - storage::ContentStore, AnnotationIndexRoot, ContentId, EdgeKey, Result as CoreResult, Sid, -}; -use parking_lot::Mutex; -use std::collections::HashMap; -use std::sync::Arc; - -/// Lazy reader over a single [`AnnotationIndexRoot`]. -/// -/// Reuse one instance across multiple lookups in the same query — it -/// caches the forward/reverse branches and any loaded leaves. The -/// reader holds borrowed references to the root and store, so its -/// lifetime is tied to the surrounding query / cascade scope. -#[derive(Debug)] -pub struct AnnotationArenaReader<'a, S: ContentStore + ?Sized> { - root: &'a AnnotationIndexRoot, - store: &'a S, - forward_branch: Mutex>>, - reverse_branch: Mutex>>, - forward_leaves: Mutex>>, - reverse_leaves: Mutex>>, -} - -impl<'a, S: ContentStore + ?Sized> AnnotationArenaReader<'a, S> { - pub fn new(root: &'a AnnotationIndexRoot, store: &'a S) -> Self { - Self { - root, - store, - forward_branch: Mutex::new(None), - reverse_branch: Mutex::new(None), - forward_leaves: Mutex::new(HashMap::new()), - reverse_leaves: Mutex::new(HashMap::new()), - } - } - - /// Annotations whose latest event at or before `as_of_t` is - /// `op = true` for the given edge. - /// - /// Returns an empty vec when: - /// - the edge has no rows in the arena; - /// - all rows for the edge are at `t > as_of_t`; - /// - every `(edge, ann)` group's visible event is a retract. - pub async fn current_annotations_for( - &self, - edge: &EdgeKey, - as_of_t: i64, - ) -> CoreResult> { - let branch = self.load_forward_branch().await?; - // Forward branch routes on `(edge, ann)`; we want all leaves - // that could contain rows for this edge, regardless of ann. - // Since rows are sorted by edge first, all rows for one edge - // are in a contiguous span of leaves. We scan branch entries - // whose `[first_edge, last_edge]` covers `edge`. - let mut out: Vec = Vec::new(); - for entry in &branch.leaves { - if entry.last_edge < *edge { - continue; - } - if entry.first_edge > *edge { - break; - } - let leaf = self.load_forward_leaf(&entry.leaf_cid).await?; - // Within a leaf, walk groups by `(edge, ann)`. Only those - // matching our edge contribute. - collect_live_anns_from_forward_leaf(&leaf, edge, as_of_t, &mut out); - } - Ok(out) - } - - /// Batched forward lookup: live annotations for many edges in one - /// merge-scan of the forward arena. Returns a vector index-aligned - /// with `edges` (entry `i` holds the live annotation SIDs for - /// `edges[i]`, empty when the edge has none). - /// - /// This is the physical counterpart to a Cypher relationship binding - /// over a stream of base edges: instead of `N` independent point - /// lookups (each re-walking the branch and re-scanning a leaf), we - /// sort the probe keys and co-iterate them with the - /// `EdgeKey`-sorted branch, touching each covering leaf once while - /// it is hot in the cache. Cost is `O(edges·log edges + covered - /// leaves)` rather than `O(edges · leaf_size)`. - /// - /// Arena-only: novelty is **not** merged here. Callers with a - /// non-empty attachment overlay must fall back to the per-edge - /// [`Self::current_annotations_merged`]; on a freshly indexed - /// snapshot (empty annotation novelty) this result is authoritative. - pub async fn current_annotations_batch( - &self, - edges: &[EdgeKey], - as_of_t: i64, - ) -> CoreResult>> { - let mut out: Vec> = vec![Vec::new(); edges.len()]; - if edges.is_empty() { - return Ok(out); - } - let branch = self.load_forward_branch().await?; - - // Probe keys in `EdgeKey` order; the branch is sorted by - // `first_edge`, so a single monotonic cursor over branch entries - // suffices. - let mut order: Vec = (0..edges.len()).collect(); - order.sort_by(|&a, &b| edges[a].cmp(&edges[b])); - - let leaves = &branch.leaves; - let mut bi = 0usize; - for &idx in &order { - let edge = &edges[idx]; - // Advance past leaves that end before this edge. - while bi < leaves.len() && leaves[bi].last_edge < *edge { - bi += 1; - } - // An edge's annotations can span consecutive leaves (one - // edge with enough annotations to cross a leaf boundary), so - // visit every entry that covers it without disturbing the - // monotonic `bi` cursor. - let mut j = bi; - while j < leaves.len() && leaves[j].first_edge <= *edge { - if leaves[j].last_edge >= *edge { - let leaf = self.load_forward_leaf(&leaves[j].leaf_cid).await?; - collect_live_anns_from_forward_leaf(&leaf, edge, as_of_t, &mut out[idx]); - } - j += 1; - } - } - Ok(out) - } - - /// Edges whose latest event at or before `as_of_t` for the given - /// annotation is `op = true`. Multiple results are possible if the - /// annotation has been re-pointed across history (legitimate or - /// from replayed-corrupt-history anomalies — the reader surfaces - /// what the arena actually contains). - pub async fn current_targets_for(&self, ann: &Sid, as_of_t: i64) -> CoreResult> { - let branch = self.load_reverse_branch().await?; - let mut out: Vec = Vec::new(); - for entry in &branch.leaves { - if entry.last_ann < *ann { - continue; - } - if entry.first_ann > *ann { - break; - } - let leaf = self.load_reverse_leaf(&entry.leaf_cid).await?; - collect_live_edges_from_reverse_leaf(&leaf, ann, as_of_t, &mut out); - } - Ok(out) - } - - /// Merged forward lookup combining indexed arena events with - /// caller-provided novelty events. Returns annotations whose - /// latest event over the union (`t <= as_of_t`) is `op = true`. - /// - /// `novelty_events` should contain every overlay event for this - /// edge — typically collected via - /// `AttachmentNovelty::forward_history(edge)` in the caller's - /// crate. The merge applies one visibility pass over the union, - /// so an arena `op = true` followed by a novelty `op = false` - /// (or vice versa) resolves correctly without the caller doing - /// any pre-merging. - pub async fn current_annotations_merged( - &self, - edge: &EdgeKey, - novelty_events: &[(Sid, i64, bool)], - as_of_t: i64, - ) -> CoreResult> { - let arena_events = self.collect_forward_events(edge, as_of_t).await?; - Ok(merge_live_annotations( - &arena_events, - novelty_events, - as_of_t, - )) - } - - /// Merged reverse lookup combining indexed arena events with - /// caller-provided novelty events. See - /// [`Self::current_annotations_merged`] for the merge semantics. - pub async fn current_targets_merged( - &self, - ann: &Sid, - novelty_events: &[(EdgeKey, i64, bool)], - as_of_t: i64, - ) -> CoreResult> { - let arena_events = self.collect_reverse_events(ann, as_of_t).await?; - Ok(merge_live_edges(&arena_events, novelty_events, as_of_t)) - } - - /// Collect every forward event `(ann, t, op)` for the given edge - /// from the indexed arena, with `t <= as_of_t`. Used internally by - /// the merged path; exposed in case callers want to build their - /// own merge logic. - pub async fn collect_forward_events( - &self, - edge: &EdgeKey, - as_of_t: i64, - ) -> CoreResult> { - let branch = self.load_forward_branch().await?; - let mut out: Vec<(Sid, i64, bool)> = Vec::new(); - for entry in &branch.leaves { - if entry.last_edge < *edge { - continue; - } - if entry.first_edge > *edge { - break; - } - let leaf = self.load_forward_leaf(&entry.leaf_cid).await?; - for row in &leaf.rows { - if row.edge == *edge && row.t <= as_of_t { - out.push((row.ann.clone(), row.t, row.op)); - } - } - } - Ok(out) - } - - /// Collect every reverse event `(edge, t, op)` for the given - /// annotation from the indexed arena, with `t <= as_of_t`. - pub async fn collect_reverse_events( - &self, - ann: &Sid, - as_of_t: i64, - ) -> CoreResult> { - let branch = self.load_reverse_branch().await?; - let mut out: Vec<(EdgeKey, i64, bool)> = Vec::new(); - for entry in &branch.leaves { - if entry.last_ann < *ann { - continue; - } - if entry.first_ann > *ann { - break; - } - let leaf = self.load_reverse_leaf(&entry.leaf_cid).await?; - for row in &leaf.rows { - if row.ann == *ann && row.t <= as_of_t { - out.push((row.edge.clone(), row.t, row.op)); - } - } - } - Ok(out) - } - - /// Collect every leaf CID referenced by the arena's forward and - /// reverse branches. Used by the indexer's GC bookkeeping when - /// replacing an arena: the new root supersedes the old, and - /// `ContentStore::release` needs exact CIDs (not child graphs) - /// to reclaim each old leaf. - pub async fn all_leaf_cids(&self) -> CoreResult> { - let fwd = self.load_forward_branch().await?; - let rev = self.load_reverse_branch().await?; - let mut out: Vec = Vec::with_capacity(fwd.leaves.len() + rev.leaves.len()); - out.extend(fwd.leaves.iter().map(|e| e.leaf_cid.clone())); - out.extend(rev.leaves.iter().map(|e| e.leaf_cid.clone())); - Ok(out) - } - - /// Forward-arena leaf entries in arena order — the itinerary for a - /// streaming walk of every live attachment, one - /// [`Self::live_pairs_in_forward_leaf`] call per entry. - pub async fn forward_leaf_entries(&self) -> CoreResult> { - Ok(self.load_forward_branch().await?.leaves.clone()) - } - - /// Live `(edge, ann)` pairs of one forward leaf at `as_of_t`, in arena - /// order. The leaf is decoded and dropped rather than cached — a - /// whole-arena walk must not retain every leaf — and the builder never - /// splits an `(edge, ann)` group across leaves, so latest-wins resolves - /// within the leaf. - pub async fn live_pairs_in_forward_leaf( - &self, - cid: &ContentId, - as_of_t: i64, - ) -> CoreResult> { - let bytes = self.store.get(cid).await?; - let leaf = AnnotationForwardLeaf::decode(&bytes).map_err(|e| { - fluree_db_core::Error::invalid_index(format!("annotation forward leaf decode: {e}")) - })?; - let rows = &leaf.rows; - let mut out = Vec::new(); - let mut i = 0; - while i < rows.len() { - let group = &rows[i]; - let mut latest_visible: Option = None; - while i < rows.len() && rows[i].edge == group.edge && rows[i].ann == group.ann { - if rows[i].t <= as_of_t { - // Sorted `(t, op)` ascending with `false < true`, so the - // last row at or before `as_of_t` is the visible event. - latest_visible = Some(rows[i].op); - } - i += 1; - } - if latest_visible == Some(true) { - out.push((group.edge.clone(), group.ann.clone())); - } - } - Ok(out) - } - - /// Reverse-arena leaf entries in arena order — the itinerary for a - /// streaming walk of every live attachment in **reifier** order - /// (`(ann, edge)` sorted), one [`Self::live_pairs_in_reverse_leaf`] - /// call per entry. Reifier order is dictionary order, so a consumer - /// that encodes the reifiers touches the subject dictionary - /// sequentially instead of at random. - pub async fn reverse_leaf_entries(&self) -> CoreResult> { - Ok(self.load_reverse_branch().await?.leaves.clone()) - } - - /// Live `(edge, ann)` pairs of one reverse leaf at `as_of_t`, in - /// `(ann, edge)` order. Same contract as - /// [`Self::live_pairs_in_forward_leaf`]: decoded and dropped, not - /// cached, latest-wins within the leaf (the builder never splits an - /// `(ann, edge)` group across leaves). - pub async fn live_pairs_in_reverse_leaf( - &self, - cid: &ContentId, - as_of_t: i64, - ) -> CoreResult> { - let bytes = self.store.get(cid).await?; - let leaf = AnnotationReverseLeaf::decode(&bytes).map_err(|e| { - fluree_db_core::Error::invalid_index(format!("annotation reverse leaf decode: {e}")) - })?; - let rows = &leaf.rows; - let mut out = Vec::new(); - let mut i = 0; - while i < rows.len() { - let group = &rows[i]; - let mut latest_visible: Option = None; - while i < rows.len() && rows[i].ann == group.ann && rows[i].edge == group.edge { - if rows[i].t <= as_of_t { - latest_visible = Some(rows[i].op); - } - i += 1; - } - if latest_visible == Some(true) { - out.push((group.edge.clone(), group.ann.clone())); - } - } - Ok(out) - } - - /// Walk every forward-arena row in sort order and yield it as a - /// `(EdgeKey, ann, t, op)` event tuple. Used by the indexer's - /// arena-rebuild path when merging the previous arena with a - /// novelty bundle delta — the union of both event sets is - /// re-sorted and re-built. - /// - /// Loads every leaf (no skipping); on a healthy ledger this is - /// only called at index-build time, not on hot read paths. - pub async fn collect_all_forward_events(&self) -> CoreResult> { - let branch = self.load_forward_branch().await?; - let total: usize = branch.leaves.iter().map(|e| e.row_count as usize).sum(); - let mut out: Vec<(EdgeKey, Sid, i64, bool)> = Vec::with_capacity(total); - for entry in &branch.leaves { - let leaf = self.load_forward_leaf(&entry.leaf_cid).await?; - for row in &leaf.rows { - out.push((row.edge.clone(), row.ann.clone(), row.t, row.op)); - } - } - Ok(out) - } - - /// Every `(edge, t, op)` event for the given annotation, in arena - /// sort order — `(edge, t, op)` ascending. Used by history queries - /// to surface attach/detach timelines without applying a - /// visibility filter. - pub async fn target_history_for(&self, ann: &Sid) -> CoreResult> { - let branch = self.load_reverse_branch().await?; - let mut out: Vec<(EdgeKey, i64, bool)> = Vec::new(); - for entry in &branch.leaves { - if entry.last_ann < *ann { - continue; - } - if entry.first_ann > *ann { - break; - } - let leaf = self.load_reverse_leaf(&entry.leaf_cid).await?; - for row in &leaf.rows { - if row.ann == *ann { - out.push((row.edge.clone(), row.t, row.op)); - } - } - } - Ok(out) - } - - // ── Loaders / cache ───────────────────────────────────────────── - - async fn load_forward_branch(&self) -> CoreResult> { - if let Some(b) = self.forward_branch.lock().clone() { - return Ok(b); - } - let bytes = self.store.get(&self.root.forward_branch_cid).await?; - let branch = AnnotationForwardBranch::decode(&bytes).map_err(|e| { - fluree_db_core::Error::invalid_index(format!("annotation forward branch decode: {e}")) - })?; - let arc = Arc::new(branch); - *self.forward_branch.lock() = Some(arc.clone()); - Ok(arc) - } - - async fn load_reverse_branch(&self) -> CoreResult> { - if let Some(b) = self.reverse_branch.lock().clone() { - return Ok(b); - } - let bytes = self.store.get(&self.root.reverse_branch_cid).await?; - let branch = AnnotationReverseBranch::decode(&bytes).map_err(|e| { - fluree_db_core::Error::invalid_index(format!("annotation reverse branch decode: {e}")) - })?; - let arc = Arc::new(branch); - *self.reverse_branch.lock() = Some(arc.clone()); - Ok(arc) - } - - async fn load_forward_leaf(&self, cid: &ContentId) -> CoreResult> { - if let Some(l) = self.forward_leaves.lock().get(cid).cloned() { - return Ok(l); - } - let bytes = self.store.get(cid).await?; - let leaf = AnnotationForwardLeaf::decode(&bytes).map_err(|e| { - fluree_db_core::Error::invalid_index(format!("annotation forward leaf decode: {e}")) - })?; - let arc = Arc::new(leaf); - self.forward_leaves.lock().insert(cid.clone(), arc.clone()); - Ok(arc) - } - - async fn load_reverse_leaf(&self, cid: &ContentId) -> CoreResult> { - if let Some(l) = self.reverse_leaves.lock().get(cid).cloned() { - return Ok(l); - } - let bytes = self.store.get(cid).await?; - let leaf = AnnotationReverseLeaf::decode(&bytes).map_err(|e| { - fluree_db_core::Error::invalid_index(format!("annotation reverse leaf decode: {e}")) - })?; - let arc = Arc::new(leaf); - self.reverse_leaves.lock().insert(cid.clone(), arc.clone()); - Ok(arc) - } -} - -// ── Visibility-filtered scanners (pure, no I/O) ───────────────────── - -/// Walk a forward leaf and append every annotation that is live at -/// `as_of_t` for the given edge. Within the leaf, rows are sorted by -/// `(edge, ann, t, op)`, so we advance to the first row matching -/// `edge` and walk groups until the edge differs. -fn collect_live_anns_from_forward_leaf( - leaf: &AnnotationForwardLeaf, - edge: &EdgeKey, - as_of_t: i64, - out: &mut Vec, -) { - let rows = &leaf.rows; - // Skip rows below `edge`. - let start = rows.partition_point(|r| r.edge < *edge); - let mut i = start; - while i < rows.len() && rows[i].edge == *edge { - // Walk this `(edge, ann)` group. - let group_ann = rows[i].ann.clone(); - let mut latest_visible: Option = None; - while i < rows.len() && rows[i].edge == *edge && rows[i].ann == group_ann { - if rows[i].t <= as_of_t { - // Sort within the group is `(t, op)` ascending and - // `false < true`, so the last row at or before - // `as_of_t` is the latest visible event. - latest_visible = Some(rows[i].op); - } - i += 1; - } - if latest_visible == Some(true) { - out.push(group_ann); - } - } -} - -/// Merge indexed + novelty events for one edge under one visibility -/// pass. Per-`(edge, ann)` pair, the latest event with `t <= as_of_t` -/// determines liveness. The arena and novelty event lists are not -/// required to be sorted or deduplicated — duplicate `(ann, t)` -/// entries are tolerated; later occurrences with the same `t` win on -/// the `op` axis (`false < true` so an assert at the same `t` as a -/// retract wins, matching the on-disk sort tie-break). -fn merge_live_annotations( - arena: &[(Sid, i64, bool)], - novelty: &[(Sid, i64, bool)], - as_of_t: i64, -) -> Vec { - use std::collections::BTreeMap; - let mut latest: BTreeMap = BTreeMap::new(); - let consider = |latest: &mut BTreeMap, ann: &Sid, t: i64, op: bool| { - if t > as_of_t { - return; - } - latest - .entry(ann.clone()) - .and_modify(|cur| { - if (t, op) >= (cur.0, cur.1) { - *cur = (t, op); - } - }) - .or_insert((t, op)); - }; - for (ann, t, op) in arena { - consider(&mut latest, ann, *t, *op); - } - for (ann, t, op) in novelty { - consider(&mut latest, ann, *t, *op); - } - latest - .into_iter() - .filter_map(|(ann, (_, op))| if op { Some(ann) } else { None }) - .collect() -} - -fn merge_live_edges( - arena: &[(EdgeKey, i64, bool)], - novelty: &[(EdgeKey, i64, bool)], - as_of_t: i64, -) -> Vec { - use std::collections::BTreeMap; - let mut latest: BTreeMap = BTreeMap::new(); - let consider = - |latest: &mut BTreeMap, edge: &EdgeKey, t: i64, op: bool| { - if t > as_of_t { - return; - } - latest - .entry(edge.clone()) - .and_modify(|cur| { - if (t, op) >= (cur.0, cur.1) { - *cur = (t, op); - } - }) - .or_insert((t, op)); - }; - for (edge, t, op) in arena { - consider(&mut latest, edge, *t, *op); - } - for (edge, t, op) in novelty { - consider(&mut latest, edge, *t, *op); - } - latest - .into_iter() - .filter_map(|(edge, (_, op))| if op { Some(edge) } else { None }) - .collect() -} - -fn collect_live_edges_from_reverse_leaf( - leaf: &AnnotationReverseLeaf, - ann: &Sid, - as_of_t: i64, - out: &mut Vec, -) { - let rows = &leaf.rows; - let start = rows.partition_point(|r| r.ann < *ann); - let mut i = start; - while i < rows.len() && rows[i].ann == *ann { - let group_edge = rows[i].edge.clone(); - let mut latest_visible: Option = None; - while i < rows.len() && rows[i].ann == *ann && rows[i].edge == group_edge { - if rows[i].t <= as_of_t { - latest_visible = Some(rows[i].op); - } - i += 1; - } - if latest_visible == Some(true) { - out.push(group_edge); - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::annotation_arena::{ - build_arenas_from_flakes, build_forward_branch, build_reverse_branch, - DEFAULT_TARGET_ROWS_PER_LEAF, - }; - use fluree_db_core::storage::MemoryContentStore; - use fluree_db_core::{AnnotationStats, ContentKind, FlakeValue}; - use fluree_vocab::db as db_predicates; - - fn ann_sid(name: &str) -> Sid { - Sid::new(20, name) - } - fn ref_sid(name: &str) -> Sid { - Sid::new(11, name) - } - fn p(suffix: &str) -> Sid { - Sid::new(fluree_vocab::namespaces::FLUREE_DB, suffix) - } - fn id_dt() -> Sid { - fluree_db_core::id_datatype_sid() - } - - fn make_bundle( - ann: &str, - s: &str, - pname: &str, - o: &str, - t: i64, - op: bool, - ) -> Vec { - let a = ann_sid(ann); - vec![ - fluree_db_core::Flake::new( - a.clone(), - p(db_predicates::REIFIES_SUBJECT), - FlakeValue::Ref(ref_sid(s)), - id_dt(), - t, - op, - None, - ), - fluree_db_core::Flake::new( - a.clone(), - p(db_predicates::REIFIES_PREDICATE), - FlakeValue::Ref(ref_sid(pname)), - id_dt(), - t, - op, - None, - ), - fluree_db_core::Flake::new( - a, - p(db_predicates::REIFIES_OBJECT), - FlakeValue::Ref(ref_sid(o)), - id_dt(), - t, - op, - None, - ), - ] - } - - /// Build an arena from a batch of bundle flakes, write all blobs - /// to the given store, return the populated `AnnotationIndexRoot`. - async fn build_and_store( - flakes: &[fluree_db_core::Flake], - target_rows_per_leaf: usize, - store: &MemoryContentStore, - ) -> AnnotationIndexRoot { - let out = build_arenas_from_flakes(flakes, target_rows_per_leaf); - - let mut fwd_pairs = Vec::new(); - for (summary, blob) in out.forward_leaves { - let cid = store - .put(ContentKind::AnnotationForwardLeaf, &blob) - .await - .unwrap(); - fwd_pairs.push((summary, cid)); - } - let fwd_branch_bytes = build_forward_branch(&fwd_pairs); - let fwd_branch_cid = store - .put(ContentKind::AnnotationForwardBranch, &fwd_branch_bytes) - .await - .unwrap(); - - let mut rev_pairs = Vec::new(); - for (summary, blob) in out.reverse_leaves { - let cid = store - .put(ContentKind::AnnotationReverseLeaf, &blob) - .await - .unwrap(); - rev_pairs.push((summary, cid)); - } - let rev_branch_bytes = build_reverse_branch(&rev_pairs); - let rev_branch_cid = store - .put(ContentKind::AnnotationReverseBranch, &rev_branch_bytes) - .await - .unwrap(); - - AnnotationIndexRoot { - version: 1, - max_t: out.max_t, - forward_branch_cid: fwd_branch_cid, - reverse_branch_cid: rev_branch_cid, - stats: out.stats, - } - } - - #[tokio::test] - async fn streaming_walk_yields_exactly_the_live_pairs() { - // Tiny leaves so the walk crosses leaf boundaries; one attachment - // retracted, one edge with two attachments, one lone edge. - let mut flakes = Vec::new(); - flakes.extend(make_bundle("ann_a", "alice", "worksFor", "acme", 1, true)); - flakes.extend(make_bundle("ann_b", "alice", "worksFor", "acme", 2, true)); - flakes.extend(make_bundle("ann_a", "alice", "worksFor", "acme", 3, false)); - flakes.extend(make_bundle("ann_c", "bob", "worksFor", "acme", 2, true)); - flakes.extend(make_bundle("ann_d", "carol", "knows", "bob", 4, true)); - - let store = MemoryContentStore::new(); - let root = build_and_store(&flakes, 2, &store).await; - let reader = AnnotationArenaReader::new(&root, &store); - let entries = reader.forward_leaf_entries().await.unwrap(); - assert!(entries.len() > 1, "test must span several leaves"); - - let walk = |as_of_t: i64| { - let reader = &reader; - let entries = &entries; - async move { - let mut out: Vec = Vec::new(); - for entry in entries { - for (edge, ann) in reader - .live_pairs_in_forward_leaf(&entry.leaf_cid, as_of_t) - .await - .unwrap() - { - out.push(format!("{}/{}", edge.s.name, ann.name)); - } - } - out - } - }; - assert_eq!( - walk(10).await, - ["alice/ann_b", "bob/ann_c", "carol/ann_d"], - "current state: the retracted attachment is gone, arena order kept" - ); - assert_eq!( - walk(1).await, - ["alice/ann_a"], - "as_of_t=1 sees only the first assertion" - ); - - // The reverse walk yields the same live set in reifier order. - let mut reverse: Vec = Vec::new(); - for entry in reader.reverse_leaf_entries().await.unwrap() { - for (edge, ann) in reader - .live_pairs_in_reverse_leaf(&entry.leaf_cid, 10) - .await - .unwrap() - { - reverse.push(format!("{}/{}", ann.name, edge.s.name)); - } - } - assert_eq!(reverse, ["ann_b/alice", "ann_c/bob", "ann_d/carol"]); - } - - #[tokio::test] - async fn current_annotations_returns_only_live_attachments() { - // Two annotations on the same edge: - // - ann_a: attached at t=1, retracted at t=3 → not live - // - ann_b: attached at t=2 → live - let mut flakes = Vec::new(); - flakes.extend(make_bundle("ann_a", "alice", "worksFor", "acme", 1, true)); - flakes.extend(make_bundle("ann_b", "alice", "worksFor", "acme", 2, true)); - flakes.extend(make_bundle("ann_a", "alice", "worksFor", "acme", 3, false)); - - let store = MemoryContentStore::new(); - let root = build_and_store(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF, &store).await; - - let edge = EdgeKey { - g: None, - s: ref_sid("alice"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - - let reader = AnnotationArenaReader::new(&root, &store); - let live = reader.current_annotations_for(&edge, 100).await.unwrap(); - assert_eq!(live, vec![ann_sid("ann_b")]); - - // History view: target_history sees every event for ann_a. - let hist_a = reader.target_history_for(&ann_sid("ann_a")).await.unwrap(); - assert_eq!(hist_a.len(), 2); - assert_eq!(hist_a[0].1, 1); - assert!(hist_a[0].2); - assert_eq!(hist_a[1].1, 3); - assert!(!hist_a[1].2); - } - - #[tokio::test] - async fn current_annotations_respects_as_of_t() { - // Same data: ann_a attach at t=1, retract at t=3. - // - as_of_t=2 → ann_a is still live - // - as_of_t=3 → ann_a has been retracted - let mut flakes = Vec::new(); - flakes.extend(make_bundle("ann_a", "alice", "worksFor", "acme", 1, true)); - flakes.extend(make_bundle("ann_a", "alice", "worksFor", "acme", 3, false)); - - let store = MemoryContentStore::new(); - let root = build_and_store(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF, &store).await; - - let edge = EdgeKey { - g: None, - s: ref_sid("alice"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - - let reader = AnnotationArenaReader::new(&root, &store); - let at_t2 = reader.current_annotations_for(&edge, 2).await.unwrap(); - assert_eq!(at_t2, vec![ann_sid("ann_a")], "live at t=2"); - let at_t3 = reader.current_annotations_for(&edge, 3).await.unwrap(); - assert!(at_t3.is_empty(), "retracted at t=3"); - // Earlier than the first event: nothing visible. - let at_t0 = reader.current_annotations_for(&edge, 0).await.unwrap(); - assert!(at_t0.is_empty()); - } - - #[tokio::test] - async fn current_targets_returns_live_edges_per_annotation() { - // ann_x reifies edge_1 (live) and edge_2 (retracted). - // current_targets_for("ann_x") returns only edge_1. - let mut flakes = Vec::new(); - flakes.extend(make_bundle("ann_x", "alice", "worksFor", "acme", 1, true)); - flakes.extend(make_bundle("ann_x", "bob", "worksFor", "acme", 2, true)); - flakes.extend(make_bundle("ann_x", "bob", "worksFor", "acme", 3, false)); - - let store = MemoryContentStore::new(); - let root = build_and_store(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF, &store).await; - let reader = AnnotationArenaReader::new(&root, &store); - - let mut targets = reader - .current_targets_for(&ann_sid("ann_x"), 100) - .await - .unwrap(); - // Order is `(ann, edge)` arena sort; we just check membership. - assert_eq!(targets.len(), 1); - let only = targets.pop().unwrap(); - assert_eq!(only.s, ref_sid("alice")); - } - - #[tokio::test] - async fn empty_arena_returns_empty_results() { - let store = MemoryContentStore::new(); - let root = build_and_store(&[], DEFAULT_TARGET_ROWS_PER_LEAF, &store).await; - assert_eq!(root.stats, AnnotationStats::default()); - - let reader = AnnotationArenaReader::new(&root, &store); - let edge = EdgeKey { - g: None, - s: ref_sid("alice"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - assert!(reader - .current_annotations_for(&edge, 100) - .await - .unwrap() - .is_empty()); - assert!(reader - .current_targets_for(&ann_sid("ann_a"), 100) - .await - .unwrap() - .is_empty()); - } - - #[tokio::test] - async fn cache_avoids_repeat_loads() { - // After a first lookup loads the branch + leaf, subsequent - // lookups for the same data must not error if we corrupt the - // store underneath. (We can't easily delete from - // MemoryContentStore, so we instead assert that a second call - // returns the same result — the cache is exercised by - // construction since we only put once.) - let mut flakes = Vec::new(); - flakes.extend(make_bundle("ann_a", "alice", "worksFor", "acme", 1, true)); - - let store = MemoryContentStore::new(); - let root = build_and_store(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF, &store).await; - let reader = AnnotationArenaReader::new(&root, &store); - - let edge = EdgeKey { - g: None, - s: ref_sid("alice"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - let first = reader.current_annotations_for(&edge, 100).await.unwrap(); - let second = reader.current_annotations_for(&edge, 100).await.unwrap(); - assert_eq!(first, second); - // Cache should hold one branch + at least one leaf. - assert!(reader.forward_branch.lock().is_some()); - assert!(!reader.forward_leaves.lock().is_empty()); - } - - #[tokio::test] - async fn merge_arena_assert_with_novelty_retract_resolves_to_not_live() { - // Arena holds (edge, ann_a) attached at t=1. - // Novelty holds (edge, ann_a) retracted at t=5. - // Merged view: ann_a is not live. - let flakes = make_bundle("ann_a", "alice", "worksFor", "acme", 1, true); - let store = MemoryContentStore::new(); - let root = build_and_store(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF, &store).await; - let reader = AnnotationArenaReader::new(&root, &store); - - let edge = EdgeKey { - g: None, - s: ref_sid("alice"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - - // Indexed-only path still sees ann_a as live. - let indexed = reader.current_annotations_for(&edge, 100).await.unwrap(); - assert_eq!(indexed, vec![ann_sid("ann_a")]); - - // Merged with novelty retract: not live. - let novelty = vec![(ann_sid("ann_a"), 5, false)]; - let merged = reader - .current_annotations_merged(&edge, &novelty, 100) - .await - .unwrap(); - assert!(merged.is_empty(), "novelty retract overrides arena assert"); - - // As-of t=4 — novelty retract not yet visible — ann_a still live. - let merged_t4 = reader - .current_annotations_merged(&edge, &novelty, 4) - .await - .unwrap(); - assert_eq!(merged_t4, vec![ann_sid("ann_a")]); - } - - #[tokio::test] - async fn merge_novelty_only_attachment_visible() { - // Empty arena, novelty has the only event. Merged view sees it. - let store = MemoryContentStore::new(); - let root = build_and_store(&[], DEFAULT_TARGET_ROWS_PER_LEAF, &store).await; - let reader = AnnotationArenaReader::new(&root, &store); - - let edge = EdgeKey { - g: None, - s: ref_sid("alice"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - let novelty = vec![(ann_sid("ann_new"), 10, true)]; - let merged = reader - .current_annotations_merged(&edge, &novelty, 100) - .await - .unwrap(); - assert_eq!(merged, vec![ann_sid("ann_new")]); - } - - #[tokio::test] - async fn merge_reverse_arena_assert_then_novelty_retarget() { - // Arena: ann_x reifies edge_a (t=1, asserted). - // Novelty: ann_x retracts edge_a at t=5, asserts edge_b at t=6. - // Merged: ann_x currently reifies edge_b only. - let flakes = make_bundle("ann_x", "alice", "worksFor", "acme", 1, true); - let store = MemoryContentStore::new(); - let root = build_and_store(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF, &store).await; - let reader = AnnotationArenaReader::new(&root, &store); - - let edge_a = EdgeKey { - g: None, - s: ref_sid("alice"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - let edge_b = EdgeKey { - g: None, - s: ref_sid("bob"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - let novelty = vec![(edge_a.clone(), 5, false), (edge_b.clone(), 6, true)]; - let merged = reader - .current_targets_merged(&ann_sid("ann_x"), &novelty, 100) - .await - .unwrap(); - assert_eq!(merged, vec![edge_b]); - } - - #[tokio::test] - async fn merge_collect_events_returns_only_below_as_of_t() { - // collect_forward_events / collect_reverse_events filter by - // as_of_t before returning. Future events must not leak. - let mut flakes = Vec::new(); - flakes.extend(make_bundle("ann_a", "alice", "worksFor", "acme", 5, true)); - flakes.extend(make_bundle("ann_b", "alice", "worksFor", "acme", 10, true)); - let store = MemoryContentStore::new(); - let root = build_and_store(&flakes, DEFAULT_TARGET_ROWS_PER_LEAF, &store).await; - let reader = AnnotationArenaReader::new(&root, &store); - - let edge = EdgeKey { - g: None, - s: ref_sid("alice"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - let at_t7 = reader.collect_forward_events(&edge, 7).await.unwrap(); - assert_eq!(at_t7.len(), 1); - assert_eq!(at_t7[0].0, ann_sid("ann_a")); - let at_t100 = reader.collect_forward_events(&edge, 100).await.unwrap(); - assert_eq!(at_t100.len(), 2); - } - - #[tokio::test] - async fn lookups_route_through_branch_to_correct_leaf() { - // Multiple edges across more than one leaf — exercises the - // branch routing. Use target=2 and three edges so we get - // multiple forward leaves. - let mut flakes = Vec::new(); - flakes.extend(make_bundle("ann_1", "s1", "worksFor", "acme", 1, true)); - flakes.extend(make_bundle("ann_2", "s2", "worksFor", "acme", 2, true)); - flakes.extend(make_bundle("ann_3", "s3", "worksFor", "acme", 3, true)); - - let store = MemoryContentStore::new(); - let root = build_and_store(&flakes, 1, &store).await; - // 3 distinct edges → at least 2 leaves with target=1. - assert!(root.stats.forward_rows == 3); - - let reader = AnnotationArenaReader::new(&root, &store); - for (s, ann) in [("s1", "ann_1"), ("s2", "ann_2"), ("s3", "ann_3")] { - let edge = EdgeKey { - g: None, - s: ref_sid(s), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - let live = reader.current_annotations_for(&edge, 100).await.unwrap(); - assert_eq!(live, vec![ann_sid(ann)], "edge {s} → {ann}"); - } - } - - fn edge_of(s: &str, p: &str, o: &str) -> EdgeKey { - EdgeKey { - g: None, - s: ref_sid(s), - p: ref_sid(p), - o: FlakeValue::Ref(ref_sid(o)), - dt: id_dt(), - lang: None, - list_i: None, - } - } - - #[tokio::test] - async fn current_annotations_batch_matches_point_lookups_across_leaves() { - // Several edges spread across multiple leaves (tiny leaf size - // forces branch traversal), one edge retracted, plus a probe for - // an edge with no annotation. The batch result must equal the - // per-edge point lookups, index-aligned, with the missing and - // retracted edges empty. - let mut flakes = Vec::new(); - flakes.extend(make_bundle("ann1", "alice", "knows", "bob", 1, true)); - flakes.extend(make_bundle("ann2", "carol", "knows", "dave", 1, true)); - flakes.extend(make_bundle("ann3", "erin", "knows", "frank", 1, true)); - flakes.extend(make_bundle("ann4", "gina", "knows", "hank", 1, true)); - // Retract ann3's edge. - flakes.extend(make_bundle("ann3", "erin", "knows", "frank", 4, false)); - - let store = MemoryContentStore::new(); - // Two rows per leaf → multiple forward leaves, exercising the - // monotonic branch cursor. - let root = build_and_store(&flakes, 2, &store).await; - let reader = AnnotationArenaReader::new(&root, &store); - - // Intentionally unsorted probe order, with a miss interleaved. - let probes = vec![ - edge_of("gina", "knows", "hank"), // ann4 - edge_of("nobody", "knows", "zztop"), // miss - edge_of("alice", "knows", "bob"), // ann1 - edge_of("erin", "knows", "frank"), // retracted → empty - edge_of("carol", "knows", "dave"), // ann2 - ]; - - let batch = reader - .current_annotations_batch(&probes, 100) - .await - .unwrap(); - assert_eq!(batch.len(), probes.len()); - for (i, e) in probes.iter().enumerate() { - let point = reader.current_annotations_for(e, 100).await.unwrap(); - assert_eq!(batch[i], point, "batch vs point mismatch at index {i}"); - } - assert_eq!(batch[0], vec![ann_sid("ann4")]); - assert!(batch[1].is_empty(), "missing edge"); - assert_eq!(batch[2], vec![ann_sid("ann1")]); - assert!(batch[3].is_empty(), "retracted edge"); - assert_eq!(batch[4], vec![ann_sid("ann2")]); - - // Empty input is a clean no-op. - let empty = reader.current_annotations_batch(&[], 100).await.unwrap(); - assert!(empty.is_empty()); - } -} diff --git a/fluree-db-binary-index/src/format/expanded_cas.rs b/fluree-db-binary-index/src/format/expanded_cas.rs index e64756f516..eab384e331 100644 --- a/fluree-db-binary-index/src/format/expanded_cas.rs +++ b/fluree-db-binary-index/src/format/expanded_cas.rs @@ -6,8 +6,8 @@ //! the root's branch manifests: //! //! - **Named-graph branches** (`FBR3`) routing to leaf + sidecar CIDs. -//! - **Annotation forward / reverse branches** (`EAFB1` / `EARB1`) -//! routing to annotation leaf CIDs. +//! - **Legacy annotation-arena branches**, which old roots still name and +//! whose leaves stay reachable until a build releases them. //! //! This module owns the single async expansion path so callers stay in //! lockstep when new branch-shaped artifacts are added to the root. @@ -47,16 +47,15 @@ use std::collections::HashSet; use fluree_db_core::content_id::ContentId; use fluree_db_core::storage::ContentStore; -use fluree_db_core::{AnnotationIndexRoot, Error, Result}; +use fluree_db_core::{Error, Result}; -use crate::annotation_arena::format::{AnnotationForwardBranch, AnnotationReverseBranch}; use crate::format::branch::read_branch_from_bytes; -use crate::format::index_root::IndexRoot; +use crate::format::index_root::{IndexRoot, LegacyAnnotationArena}; /// Strict expansion: returns the complete reachable CAS set or an error. /// /// Starts from `root.all_cas_ids()` and additionally fetches every -/// named-graph branch + annotation arena branch from `store`, decoding +/// named-graph branch + legacy arena branch from `store`, decoding /// each manifest to discover the leaf (and named-graph sidecar) CIDs /// they route to. The first read or decode failure short-circuits and /// returns `Err` — partial sets are never returned. @@ -123,9 +122,10 @@ impl ChainCasIds { } } - if let Some(ref annotation_index) = root.annotation_index { - self.expand_annotation_arena(store, annotation_index) - .await?; + if let Some(ref arena) = root.legacy_annotation_arena { + for branch_cid in arena.branches() { + self.expand_legacy_arena_branch(store, branch_cid).await?; + } } Ok(()) @@ -170,42 +170,21 @@ impl ChainCasIds { Ok(()) } - /// Annotation arena: forward + reverse branches → leaf CIDs. - async fn expand_annotation_arena( + /// Legacy annotation-arena branch → leaf CIDs. + async fn expand_legacy_arena_branch( &mut self, store: &dyn ContentStore, - annotation_index: &AnnotationIndexRoot, + branch_cid: &ContentId, ) -> Result<()> { - let forward_cid = &annotation_index.forward_branch_cid; - if let Some(bytes) = self - .read_unexpanded_manifest(store, forward_cid, "annotation forward branch") - .await? - { - let branch = AnnotationForwardBranch::decode(&bytes).map_err(|e| { - Error::invalid_index(format!( - "failed to decode annotation forward branch {forward_cid} during CID expansion: {e}" - )) - })?; - self.ids - .extend(branch.leaves.iter().map(|entry| entry.leaf_cid.clone())); - self.expanded_manifests.insert(forward_cid.clone()); - } - - let reverse_cid = &annotation_index.reverse_branch_cid; - if let Some(bytes) = self - .read_unexpanded_manifest(store, reverse_cid, "annotation reverse branch") + let Some(bytes) = self + .read_unexpanded_manifest(store, branch_cid, "legacy annotation arena branch") .await? - { - let branch = AnnotationReverseBranch::decode(&bytes).map_err(|e| { - Error::invalid_index(format!( - "failed to decode annotation reverse branch {reverse_cid} during CID expansion: {e}" - )) - })?; - self.ids - .extend(branch.leaves.iter().map(|entry| entry.leaf_cid.clone())); - self.expanded_manifests.insert(reverse_cid.clone()); - } - + else { + return Ok(()); + }; + self.ids + .extend(decode_legacy_arena_branch(branch_cid, &bytes)?); + self.expanded_manifests.insert(branch_cid.clone()); Ok(()) } @@ -274,61 +253,87 @@ pub async fn collect_root_cas_ids_expanded_tolerant( } } - if let Some(ref ann) = root.annotation_index { - match store.get(&ann.forward_branch_cid).await { - Ok(bytes) => match AnnotationForwardBranch::decode(&bytes) { - Ok(branch) => { - for entry in &branch.leaves { - ids.insert(entry.leaf_cid.clone()); - } - } - Err(e) => tracing::warn!( - branch_cid = %ann.forward_branch_cid, - error = %e, - "failed to decode annotation forward branch during CID expansion, skipping" - ), - }, - Err(e) => tracing::warn!( - branch_cid = %ann.forward_branch_cid, - error = %e, - "failed to read annotation forward branch during CID expansion, skipping" - ), - } - - match store.get(&ann.reverse_branch_cid).await { - Ok(bytes) => match AnnotationReverseBranch::decode(&bytes) { - Ok(branch) => { - for entry in &branch.leaves { - ids.insert(entry.leaf_cid.clone()); - } - } + if let Some(ref arena) = root.legacy_annotation_arena { + for branch_cid in arena.branches() { + let leaves = match store.get(branch_cid).await { + Ok(bytes) => decode_legacy_arena_branch(branch_cid, &bytes), + Err(e) => Err(e), + }; + match leaves { + Ok(leaves) => ids.extend(leaves), Err(e) => tracing::warn!( - branch_cid = %ann.reverse_branch_cid, + branch_cid = %branch_cid, error = %e, - "failed to decode annotation reverse branch during CID expansion, skipping" + "failed to expand legacy annotation arena branch, skipping" ), - }, - Err(e) => tracing::warn!( - branch_cid = %ann.reverse_branch_cid, - error = %e, - "failed to read annotation reverse branch during CID expansion, skipping" - ), + } } } ids } +/// Every blob of a legacy annotation arena: both branches and their leaves. +/// A build over a root that names one releases these, since the root it +/// writes does not. +pub async fn legacy_annotation_arena_cids( + store: &dyn ContentStore, + arena: &LegacyAnnotationArena, +) -> Result> { + let mut ids = Vec::new(); + for branch_cid in arena.branches() { + let bytes = store.get(branch_cid).await.map_err(|e| { + Error::invalid_index(format!( + "failed to read legacy annotation arena branch {branch_cid}: {e}" + )) + })?; + ids.extend(decode_legacy_arena_branch(branch_cid, &bytes)?); + ids.push(branch_cid.clone()); + } + Ok(ids) +} + +/// The leaf CIDs a legacy arena branch routes to. Both branch kinds frame a +/// CBOR body behind the same 12-byte header (length at bytes 8..12) and name +/// each leaf `leaf_cid`. +fn decode_legacy_arena_branch(branch_cid: &ContentId, bytes: &[u8]) -> Result> { + #[derive(serde::Deserialize)] + struct Branch { + leaves: Vec, + } + #[derive(serde::Deserialize)] + struct Leaf { + leaf_cid: ContentId, + } + let invalid = |e: &dyn std::fmt::Display| { + Error::invalid_index(format!( + "failed to decode legacy annotation arena branch {branch_cid}: {e}" + )) + }; + let body = bytes + .get(8..12) + .map(|len| u32::from_le_bytes(len.try_into().unwrap()) as usize) + .and_then(|len| bytes.get(12..12 + len)) + .ok_or_else(|| invalid(&"truncated"))?; + let branch: Branch = ciborium::de::from_reader(body).map_err(|e| invalid(&e))?; + Ok(branch + .leaves + .into_iter() + .map(|leaf| leaf.leaf_cid) + .collect()) +} + #[cfg(test)] mod tests { use super::*; - use crate::annotation_arena::format::{ - AnnotationForwardBranch, AnnotationForwardBranchEntry, AnnotationForwardLeaf, - AnnotationReverseBranch, AnnotationReverseBranchEntry, AnnotationReverseLeaf, - }; + use crate::format::branch::{build_branch_bytes, LeafEntry}; + use crate::format::index_root::NamedGraphRouting; + use crate::format::run_record::{RunSortOrder, LIST_INDEX_NONE}; + use crate::format::run_record_v2::RunRecordV2; use crate::format::wire_helpers::{DictPackRefs, DictRefs, DictTreeRefs}; use fluree_db_core::storage::MemoryContentStore; - use fluree_db_core::{AnnotationStats, ContentKind, EdgeKey, FlakeValue, Sid}; + use fluree_db_core::subject_id::SubjectId; + use fluree_db_core::ContentKind; use std::collections::{BTreeMap, HashMap}; use std::sync::Mutex; @@ -381,18 +386,6 @@ mod tests { ContentId::new(kind, seed) } - fn sample_edge() -> EdgeKey { - EdgeKey { - g: None, - s: Sid::new(1, "s"), - p: Sid::new(1, "p"), - o: FlakeValue::Ref(Sid::new(1, "o")), - dt: Sid::new(0, "http://www.w3.org/2001/XMLSchema#anyURI"), - lang: None, - list_i: None, - } - } - fn minimal_root() -> IndexRoot { let dummy_cid = ContentId::new(ContentKind::IndexLeaf, b"dummy"); let dummy_tree = DictTreeRefs { @@ -432,157 +425,127 @@ mod tests { garbage: None, sketch_ref: None, has_annotations: false, - annotation_index: None, + legacy_annotation_arena: None, term_dict: None, - had_annotation_arena: false, has_list_meta: None, o_type_table: IndexRoot::build_o_type_table(&[], &[]), ns_split_mode: fluree_db_core::ns_encoding::NsSplitMode::default(), } } - /// Build a root carrying an annotation arena that points to two real - /// branch blobs (forward + reverse) each routing to a single leaf. - /// Returns (root, fwd_leaf_cid, rev_leaf_cid). - async fn build_root_with_arena(store: &dyn ContentStore) -> (IndexRoot, ContentId, ContentId) { - // Write empty leaves to CAS so the branch entries point somewhere - // real. Their content doesn't matter — the helper only walks - // branches, not leaves. - let fwd_leaf_bytes = AnnotationForwardLeaf::default().encode(); - let fwd_leaf_cid = store - .put(ContentKind::AnnotationForwardLeaf, &fwd_leaf_bytes) + /// A legacy arena branch as its encoder wrote it: a 12-byte header, then + /// CBOR whose entries carry fields beyond `leaf_cid`. + fn legacy_branch_bytes(magic: &[u8; 4], leaf_cid: &ContentId) -> Vec { + #[derive(serde::Serialize)] + struct Entry<'a> { + first_ann: &'a str, + row_count: u64, + leaf_cid: &'a ContentId, + } + #[derive(serde::Serialize)] + struct Branch<'a> { + leaves: Vec>, + } + let mut body = Vec::new(); + ciborium::ser::into_writer( + &Branch { + leaves: vec![Entry { + first_ann: "a", + row_count: 1, + leaf_cid, + }], + }, + &mut body, + ) + .unwrap(); + let mut bytes = magic.to_vec(); + bytes.extend_from_slice(&[1, 0, 0, 0]); + bytes.extend_from_slice(&(body.len() as u32).to_le_bytes()); + bytes.extend_from_slice(&body); + bytes + } + + /// A root that still names a legacy arena, with both branches and their + /// leaves in `store`. Returns the root and every arena blob. + async fn root_with_legacy_arena(store: &dyn ContentStore) -> (IndexRoot, Vec) { + let fwd_leaf = store + .put(ContentKind::AnnotationForwardLeaf, b"fwd-leaf") .await .unwrap(); - - let rev_leaf_bytes = AnnotationReverseLeaf::default().encode(); - let rev_leaf_cid = store - .put(ContentKind::AnnotationReverseLeaf, &rev_leaf_bytes) + let rev_leaf = store + .put(ContentKind::AnnotationReverseLeaf, b"rev-leaf") .await .unwrap(); - - let fwd_branch = AnnotationForwardBranch { - leaves: vec![AnnotationForwardBranchEntry { - first_edge: sample_edge(), - first_ann: Sid::new(2, "a"), - last_edge: sample_edge(), - last_ann: Sid::new(2, "a"), - row_count: 0, - leaf_cid: fwd_leaf_cid.clone(), - }], - }; - let fwd_branch_cid = store - .put(ContentKind::AnnotationForwardBranch, &fwd_branch.encode()) + let fwd_branch = store + .put( + ContentKind::AnnotationForwardBranch, + &legacy_branch_bytes(b"EAFB", &fwd_leaf), + ) .await .unwrap(); - - let rev_branch = AnnotationReverseBranch { - leaves: vec![AnnotationReverseBranchEntry { - first_ann: Sid::new(2, "a"), - first_edge: sample_edge(), - last_ann: Sid::new(2, "a"), - last_edge: sample_edge(), - row_count: 0, - leaf_cid: rev_leaf_cid.clone(), - }], - }; - let rev_branch_cid = store - .put(ContentKind::AnnotationReverseBranch, &rev_branch.encode()) + let rev_branch = store + .put( + ContentKind::AnnotationReverseBranch, + &legacy_branch_bytes(b"EARB", &rev_leaf), + ) .await .unwrap(); - let mut root = minimal_root(); root.has_annotations = true; - root.annotation_index = Some(AnnotationIndexRoot { - version: 1, - max_t: 0, - forward_branch_cid: fwd_branch_cid, - reverse_branch_cid: rev_branch_cid, - stats: AnnotationStats::default(), + root.legacy_annotation_arena = Some(LegacyAnnotationArena { + forward_branch_cid: fwd_branch.clone(), + reverse_branch_cid: rev_branch.clone(), }); - (root, fwd_leaf_cid, rev_leaf_cid) + (root, vec![fwd_leaf, rev_leaf, fwd_branch, rev_branch]) } #[tokio::test] - async fn expands_annotation_branches_to_leaves() { + async fn expands_legacy_arena_branches_to_leaves() { let store = MemoryContentStore::new(); - let (root, fwd_leaf_cid, rev_leaf_cid) = build_root_with_arena(&store).await; - - let ids = collect_root_cas_ids_expanded(&store, &root).await.unwrap(); - - // Annotation branch CIDs appear via all_cas_ids(). - let ann = root.annotation_index.as_ref().unwrap(); - assert!( - ids.contains(&ann.forward_branch_cid), - "annotation forward branch CID missing" - ); - assert!( - ids.contains(&ann.reverse_branch_cid), - "annotation reverse branch CID missing" - ); - // Leaves added by branch expansion. - assert!( - ids.contains(&fwd_leaf_cid), - "forward leaf CID missing — annotation branch was not expanded" - ); - assert!( - ids.contains(&rev_leaf_cid), - "reverse leaf CID missing — annotation branch was not expanded" - ); + let (root, arena_blobs) = root_with_legacy_arena(&store).await; + + let strict = collect_root_cas_ids_expanded(&store, &root).await.unwrap(); + let tolerant = collect_root_cas_ids_expanded_tolerant(&store, &root).await; + let mut released = + legacy_annotation_arena_cids(&store, root.legacy_annotation_arena.as_ref().unwrap()) + .await + .unwrap(); + released.sort(); + let mut expected = arena_blobs.clone(); + expected.sort(); + assert_eq!(released, expected, "a build releases every arena blob"); + for blob in &arena_blobs { + assert!(strict.contains(blob), "strict expansion missed {blob}"); + assert!(tolerant.contains(blob), "tolerant expansion missed {blob}"); + } } - /// Roots that carry a manifest over unchanged must not make the chain - /// re-read it. This is what keeps a sweep's planning cost proportional to - /// the distinct manifests rather than to the length of the chain. + /// A build over a root with a legacy arena writes a root without it, so + /// the garbage diff between the two releases the whole arena. #[tokio::test] - async fn a_chain_reads_a_carried_over_manifest_once() { - let store = GetCountingStore::new(); - let (root, fwd_leaf_cid, rev_leaf_cid) = build_root_with_arena(&store).await; - - // What an incremental build that touched nothing in the arena - // publishes: a new root pointing at the previous arena's branches. - let mut later_root = root.clone(); - later_root.index_t = root.index_t + 1; - - let mut chain_ids = ChainCasIds::new(); - chain_ids.add_root(&store, &root).await.unwrap(); - chain_ids.add_root(&store, &later_root).await.unwrap(); - let ids = chain_ids.into_ids(); - - let ann = root.annotation_index.as_ref().unwrap(); - assert_eq!( - store.get_count(&ann.forward_branch_cid), - 1, - "annotation forward branch re-read for a root that carried it over" - ); - assert_eq!( - store.get_count(&ann.reverse_branch_cid), - 1, - "annotation reverse branch re-read for a root that carried it over" - ); + async fn the_garbage_diff_releases_a_legacy_arena() { + let store = MemoryContentStore::new(); + let (prev_root, arena_blobs) = root_with_legacy_arena(&store).await; + let mut new_root = prev_root.clone(); + new_root.legacy_annotation_arena = None; - // The leaves behind the skipped read are still live: they entered the - // set when the first root expanded that manifest. - assert!( - ids.contains(&fwd_leaf_cid), - "forward leaf missing after the second root skipped its manifest" - ); - assert!( - ids.contains(&rev_leaf_cid), - "reverse leaf missing after the second root skipped its manifest" - ); + let prev_ids = collect_root_cas_ids_expanded(&store, &prev_root) + .await + .unwrap(); + let new_ids = collect_root_cas_ids_expanded(&store, &new_root) + .await + .unwrap(); + let replaced: HashSet<_> = prev_ids.difference(&new_ids).cloned().collect(); + assert_eq!(replaced, arena_blobs.into_iter().collect::>()); } #[tokio::test] - async fn strict_errors_on_missing_annotation_branch() { + async fn strict_errors_on_missing_legacy_arena_branch() { let store = MemoryContentStore::new(); let mut root = minimal_root(); - root.has_annotations = true; - root.annotation_index = Some(AnnotationIndexRoot { - version: 1, - max_t: 0, + root.legacy_annotation_arena = Some(LegacyAnnotationArena { forward_branch_cid: cid(ContentKind::AnnotationForwardBranch, b"missing-fwd"), reverse_branch_cid: cid(ContentKind::AnnotationReverseBranch, b"missing-rev"), - stats: AnnotationStats::default(), }); // Strict: must surface the read failure rather than return a @@ -591,126 +554,80 @@ mod tests { .await .expect_err("strict mode should error on missing branch"); assert!( - err.to_string().contains("annotation forward branch"), + err.to_string().contains("legacy annotation arena branch"), "error should identify the missing branch: {err}" ); } #[tokio::test] - async fn tolerant_expansion_swallows_missing_annotation_branch() { + async fn tolerant_expansion_swallows_missing_legacy_arena_branch() { let store = MemoryContentStore::new(); let mut root = minimal_root(); - root.has_annotations = true; - root.annotation_index = Some(AnnotationIndexRoot { - version: 1, - max_t: 0, + let arena = LegacyAnnotationArena { forward_branch_cid: cid(ContentKind::AnnotationForwardBranch, b"missing-fwd"), reverse_branch_cid: cid(ContentKind::AnnotationReverseBranch, b"missing-rev"), - stats: AnnotationStats::default(), - }); + }; + root.legacy_annotation_arena = Some(arena.clone()); let ids = collect_root_cas_ids_expanded_tolerant(&store, &root).await; // Still contains the direct branch CIDs from all_cas_ids(). - let ann = root.annotation_index.as_ref().unwrap(); - assert!(ids.contains(&ann.forward_branch_cid)); - assert!(ids.contains(&ann.reverse_branch_cid)); + assert!(arena.branches().all(|branch| ids.contains(branch))); } + /// Roots that carry a manifest over unchanged must not make the chain + /// re-read it. This is what keeps a sweep's planning cost proportional to + /// the distinct manifests rather than to the length of the chain. #[tokio::test] - async fn diff_produces_replaced_annotation_leaves() { - let store = MemoryContentStore::new(); - let (prev_root, prev_fwd_leaf, prev_rev_leaf) = build_root_with_arena(&store).await; - let (new_root, new_fwd_leaf, new_rev_leaf) = build_root_with_arena(&store).await; - - // Leaves are content-addressed empty blobs, so the new and prev - // build pull the *same* leaf CIDs from CAS — assert that and - // then build a synthetic new root whose annotation branches are - // genuinely fresh, to exercise the diff. - assert_eq!(prev_fwd_leaf, new_fwd_leaf); - assert_eq!(prev_rev_leaf, new_rev_leaf); - - // Build a "new" root by writing distinct leaf bytes so leaf CIDs differ. - let fwd_leaf2 = AnnotationForwardLeaf { - rows: vec![crate::annotation_arena::format::AnnotationForwardRow { - edge: sample_edge(), - ann: Sid::new(2, "a"), - t: 1, - op: true, - }], - }; - let fwd_leaf2_cid = store - .put(ContentKind::AnnotationForwardLeaf, &fwd_leaf2.encode()) - .await - .unwrap(); - let rev_leaf2 = AnnotationReverseLeaf { - rows: vec![crate::annotation_arena::format::AnnotationReverseRow { - ann: Sid::new(2, "a"), - edge: sample_edge(), - t: 1, - op: true, - }], - }; - let rev_leaf2_cid = store - .put(ContentKind::AnnotationReverseLeaf, &rev_leaf2.encode()) - .await - .unwrap(); - - let fwd_branch2 = AnnotationForwardBranch { - leaves: vec![AnnotationForwardBranchEntry { - first_edge: sample_edge(), - first_ann: Sid::new(2, "a"), - last_edge: sample_edge(), - last_ann: Sid::new(2, "a"), - row_count: 1, - leaf_cid: fwd_leaf2_cid.clone(), - }], + async fn a_chain_reads_a_carried_over_manifest_once() { + let store = GetCountingStore::new(); + let leaf_cid = store.put(ContentKind::IndexLeaf, b"leaf").await.unwrap(); + let key = RunRecordV2 { + s_id: SubjectId(1), + o_key: 0, + p_id: 1, + t: 1, + o_i: LIST_INDEX_NONE, + o_type: 0, + g_id: 2, }; - let fwd_branch2_cid = store - .put(ContentKind::AnnotationForwardBranch, &fwd_branch2.encode()) - .await - .unwrap(); - let rev_branch2 = AnnotationReverseBranch { - leaves: vec![AnnotationReverseBranchEntry { - first_ann: Sid::new(2, "a"), - first_edge: sample_edge(), - last_ann: Sid::new(2, "a"), - last_edge: sample_edge(), + let branch = build_branch_bytes( + RunSortOrder::Spot, + 2, + &[LeafEntry { + first_key: key, + last_key: key, row_count: 1, - leaf_cid: rev_leaf2_cid.clone(), + leaf_cid: leaf_cid.clone(), + sidecar_cid: None, }], - }; - let rev_branch2_cid = store - .put(ContentKind::AnnotationReverseBranch, &rev_branch2.encode()) - .await - .unwrap(); - - let mut new_root = new_root; - new_root.annotation_index = Some(AnnotationIndexRoot { - version: 1, - max_t: 1, - forward_branch_cid: fwd_branch2_cid, - reverse_branch_cid: rev_branch2_cid, - stats: AnnotationStats::default(), - }); + ); + let branch_cid = store.put(ContentKind::IndexBranch, &branch).await.unwrap(); + let mut root = minimal_root(); + root.named_graphs = vec![NamedGraphRouting { + g_id: 2, + orders: vec![(RunSortOrder::Spot, branch_cid.clone())], + }]; - let prev_ids = collect_root_cas_ids_expanded(&store, &prev_root) - .await - .unwrap(); - let new_ids = collect_root_cas_ids_expanded(&store, &new_root) - .await - .unwrap(); + // What an incremental build that touched nothing in the graph + // publishes: a new root pointing at the previous branch. + let mut later_root = root.clone(); + later_root.index_t = root.index_t + 1; - let replaced: HashSet<_> = prev_ids.difference(&new_ids).cloned().collect(); + let mut chain_ids = ChainCasIds::new(); + chain_ids.add_root(&store, &root).await.unwrap(); + chain_ids.add_root(&store, &later_root).await.unwrap(); + let ids = chain_ids.into_ids(); - // The previous arena's leaves should appear in the diff. Without - // branch expansion they would silently leak. - assert!( - replaced.contains(&prev_fwd_leaf), - "prev forward leaf missing from garbage diff" + assert_eq!( + store.get_count(&branch_cid), + 1, + "branch re-read for a root that carried it over" ); + // The leaf behind the skipped read is still live: it entered the set + // when the first root expanded that manifest. assert!( - replaced.contains(&prev_rev_leaf), - "prev reverse leaf missing from garbage diff" + ids.contains(&leaf_cid), + "leaf missing after the second root skipped its manifest" ); } } diff --git a/fluree-db-binary-index/src/format/index_root.rs b/fluree-db-binary-index/src/format/index_root.rs index 62d6d21295..a74c7d3dd4 100644 --- a/fluree-db-binary-index/src/format/index_root.rs +++ b/fluree-db-binary-index/src/format/index_root.rs @@ -123,6 +123,21 @@ pub const ROOT_V6_MAGIC: &[u8; 4] = b"FIR6"; /// reindex before queries resume. pub const ROOT_V6_VERSION: u8 = 2; +/// What garbage collection needs from a retired annotation-arena root +/// section: its two branch manifests. Decoding ignores the section's other +/// fields. +#[derive(Debug, Clone, PartialEq, Eq, serde::Deserialize)] +pub struct LegacyAnnotationArena { + pub forward_branch_cid: ContentId, + pub reverse_branch_cid: ContentId, +} + +impl LegacyAnnotationArena { + pub fn branches(&self) -> impl Iterator { + [&self.forward_branch_cid, &self.reverse_branch_cid].into_iter() + } +} + /// Binary index root (`FIR6`). /// /// Contains all sections needed to load an index: dict refs, arena refs, @@ -213,62 +228,15 @@ pub struct IndexRoot { /// lookup entirely. pub has_annotations: bool, - /// Inline pointer to the on-disk forward/reverse annotation - /// arenas (M2b). See `fluree_db_core::annotation_index` for the - /// truth table that pairs this field with `has_annotations`. The - /// hard "zero attachments" guarantee requires both - /// `has_annotations == false` AND `annotation_index.is_none()`. - /// - /// Pre-builder: always `None`; the cascade fast-path consults - /// `has_annotations` instead. Once the M2b builder lands, - /// downstream readers migrate to this field for arena-backed - /// lookups. - /// - /// Encoder invariant: `annotation_index.is_some()` implies - /// `FLAG_HAS_ANNOTATIONS` on the wire, so the cascade fast-path - /// can never desynchronize from a populated arena. - pub annotation_index: Option, + /// Branches of a retired annotation arena, read from a root that still + /// carries one so the next build can release its blobs. Never encoded. + pub legacy_annotation_arena: Option, /// Triple-term dictionary (`OType::TRIPLE_TERM` handles ↔ encoded base /// edges). `None` until a build has interned reification links. Lives in - /// its own trailing section, flagged by `FLAG_EXT_HAS_TERM_DICT`, so its - /// lifecycle is independent of the annotation arena above. + /// its own trailing section, flagged by `FLAG_EXT_HAS_TERM_DICT`. pub term_dict: Option, - /// Sticky bit governing whether the api's - /// `ApiAttachmentEventsProvider` is allowed to bootstrap an - /// `Authoritative` annotation arena from a one-time base-index - /// scan. Despite the historical name, the load-bearing meaning - /// is closer to "base-index bootstrap is **not** allowed": - /// - /// - `false` only on fresh bulk-import roots (no indexer pass - /// has yet touched the annotation history). The live base - /// IS the complete history, so a one-time PSOT scan can - /// produce a complete `Authoritative` event set. - /// - `true` on any indexer-produced root with - /// `has_annotations=true`, regardless of whether an arena - /// was sealed by that pass. Covers: - /// * arena-seal path (`set_annotation_index(Some, ..)` flips - /// the bit and never clears it on subsequent - /// `set_annotation_index(None, ..)` defensive drops), - /// * indexer pass that processed annotation events without - /// sealing (coerced in `IncrementalRootBuilder::build()` - /// and `encode_and_write_root_v6` whenever - /// `has_annotations=true`), - /// * legacy pre-this-change roots that have - /// `annotation_index=Some(_)` but `pad=0` on the wire - /// (coerced in the decoder). - /// - /// Once set, never cleared by any root assembly path. Carried - /// in the FIR6 extended-flags byte (low byte of the - /// historically-zero `pad(2)` header field). Wire-compat: - /// - old roots with `annotation_index=None` decode to `false` - /// (matches the never-touched state), - /// - old roots with `annotation_index=Some(_)` decode to - /// `true` via the `|| annotation_index.is_some()` coercion - /// (matches "indexer already sealed an arena here"). - pub had_annotation_arena: bool, - /// Whether any indexed row carries an RDF-list position (`o_i != /// LIST_INDEX_NONE`, i.e. a `@list` value). /// @@ -598,30 +566,20 @@ impl IndexRoot { /// `fluree-db-transact::stage` — non-annotation ledgers skip /// the per-retract POST lookup entirely. const FLAG_HAS_ANNOTATIONS: u8 = 1 << 6; - /// Optional on-disk arena pointer (`AnnotationIndexRoot`) is present - /// in the inline tail. Independent of `FLAG_HAS_ANNOTATIONS`: the - /// sticky bit may be set without the arena (pre-builder roots); - /// once the builder runs the bit is set whenever the section is - /// present. - const FLAG_HAS_ANNOTATION_INDEX: u8 = 1 << 7; - - /// Extended-flags bit (lives in the low byte of the - /// historically-zero `pad(2)` header field, i.e. `data[6]`). - /// Gates the api's `ApiAttachmentEventsProvider` base-index - /// scan-fallback: `1` means the indexer owns the annotation - /// history and bootstrap is **not** allowed; `0` means - /// fresh-bulk-import bootstrap is permitted. Sticky once set. - /// See `IndexRoot.had_annotation_arena` for the full state - /// machine. - const FLAG_EXT_HAD_ANNOTATION_ARENA: u8 = 1 << 0; + /// Legacy: the root carries a retired annotation-arena section (u32 + /// length + CBOR) before the term dictionary. Read, never written; the + /// bit is reserved. + const FLAG_LEGACY_ANNOTATION_ARENA: u8 = 1 << 7; + + // Extended-flags byte (`data[6]`, the low byte of the historically-zero + // `pad(2)` header field). Bit 0 is retired and reserved. /// Extended-flags bit: `has_list_meta` is tracked (`Some(_)`). /// Zero on legacy roots, which therefore decode to `None`. const FLAG_EXT_LIST_META_TRACKED: u8 = 1 << 1; /// Extended-flags bit: at least one indexed row carries a list index. /// Only meaningful when `FLAG_EXT_LIST_META_TRACKED` is set. const FLAG_EXT_HAS_LIST_META: u8 = 1 << 2; - /// Root carries a triple-term dictionary section after the annotation - /// section. + /// Root carries a triple-term dictionary section, the root's last. const FLAG_EXT_HAS_TERM_DICT: u8 = 1 << 3; /// Encode to the binary FIR6 wire format. @@ -629,22 +587,6 @@ impl IndexRoot { /// Determinism: namespaces sorted by ns_code, named graphs by g_id, /// orders by order_id, numbig/vectors/spatial/fulltext by p_id. pub fn encode(&self) -> Vec { - // Invariant guard for the sticky bit. The field doc says - // `had_annotation_arena` is "never cleared by any root - // assembly path" — a populated `annotation_index` in - // particular implies the bit was set on the seal that - // produced it. The encoder also coerces the wire bit on - // (below), so a forgotten in-memory flip never escapes to - // disk — but the in-memory invariant is still useful to - // pin in dev/CI so a regression in the seal/drop - // bookkeeping surfaces here instead of as a phantom - // bootstrap-eligible state on the next reindex. - debug_assert!( - self.had_annotation_arena || self.annotation_index.is_none(), - "had_annotation_arena=false with annotation_index=Some(_) violates the sticky-bit \ - contract — once an arena is set, the bit must be set too. See IndexRoot field doc." - ); - let mut buf = Vec::with_capacity(8192); // ---- Header (24 bytes) ---- @@ -674,40 +616,15 @@ impl IndexRoot { Self::FLAG_LEX_SORTED_STRING_IDS } else { 0 - }) | (if self.has_annotations || self.annotation_index.is_some() { - // The cascade fast-path in `fluree-db-transact::stage` gates - // on `LedgerSnapshot.has_annotations`. A populated - // `annotation_index` without the sticky bit set would let - // post-reindex retracts skip cascade — so the encoder - // forces the two signals to agree on the wire regardless - // of caller bookkeeping. + }) | (if self.has_annotations { Self::FLAG_HAS_ANNOTATIONS } else { 0 - }) | (if self.annotation_index.is_some() { - Self::FLAG_HAS_ANNOTATION_INDEX - } else { - 0 }); buf.push(flags); // Extended-flags byte (low byte of historically-zero pad). // High byte stays zero (reserved for future extension). - // - // Encoder invariant: a populated `annotation_index` implies - // `had_annotation_arena=true` on the wire — symmetric with - // the `FLAG_HAS_ANNOTATIONS` coercion above. Without this, - // a caller that constructs a root with `annotation_index = - // Some(_)` but forgets to flip the sticky bit would produce - // a wire root that, on later defensive drop, misrepresents - // itself as "never sealed" and re-enables the provider's - // bootstrap scan-fallback. - let had_annotation_arena_on_wire = - self.had_annotation_arena || self.annotation_index.is_some(); - let mut flags_ext = if had_annotation_arena_on_wire { - Self::FLAG_EXT_HAD_ANNOTATION_ARENA - } else { - 0 - }; + let mut flags_ext = 0u8; match self.has_list_meta { Some(true) => { flags_ext |= Self::FLAG_EXT_LIST_META_TRACKED | Self::FLAG_EXT_HAS_LIST_META; @@ -870,18 +787,6 @@ impl IndexRoot { write_cid(&mut buf, sketch); } - // ---- Optional: annotation_index (M2b) ---- - // CBOR-encoded inline so the index format can grow new fields - // (e.g. histograms in M3) without bumping FIR6 itself. Length - // prefix lets old readers skip the section. - if let Some(ref ann) = self.annotation_index { - let mut body = Vec::new(); - ciborium::ser::into_writer(ann, &mut body) - .expect("ciborium serialization to Vec is infallible"); - buf.extend_from_slice(&(body.len() as u32).to_le_bytes()); - buf.extend_from_slice(&body); - } - // ---- Optional: triple-term dictionary ---- if let Some(ref td) = self.term_dict { write_term_dict_refs(&mut buf, td); @@ -919,8 +824,6 @@ impl IndexRoot { string_watermark: self.string_watermark, graph_iris: self.graph_iris.clone(), has_annotations: self.has_annotations, - annotation_index: self.annotation_index.clone(), - had_annotation_arena: self.had_annotation_arena, has_list_meta: self.has_list_meta, }) } @@ -945,14 +848,13 @@ impl IndexRoot { // Extended-flags byte at data[6]; data[7] reserved. // Old encoders (FIR6 with the original `pad(2)`) wrote // `0u16` here, so old roots decode to all extended flags - // = false — matching the "never sealed an arena" state. + // = false. let flags_ext = data[6]; let mut pos = 8; // skip past flags_ext + reserved let index_t = read_i64_at(data, &mut pos)?; let base_t = read_i64_at(data, &mut pos)?; let lex_sorted_string_ids = (flags & Self::FLAG_LEX_SORTED_STRING_IDS) != 0; let has_annotations = (flags & Self::FLAG_HAS_ANNOTATIONS) != 0; - let had_annotation_arena = (flags_ext & Self::FLAG_EXT_HAD_ANNOTATION_ARENA) != 0; let has_list_meta = decode_list_meta_flags( flags_ext, Self::FLAG_EXT_LIST_META_TRACKED, @@ -1149,15 +1051,14 @@ impl IndexRoot { None }; - let annotation_index = if flags & Self::FLAG_HAS_ANNOTATION_INDEX != 0 { + let legacy_annotation_arena = if flags & Self::FLAG_LEGACY_ANNOTATION_ARENA != 0 { let len = read_u32_at(data, &mut pos)? as usize; - ensure_bytes(data, pos, len, "annotation_index section")?; - let ann = ciborium::de::from_reader::( - &data[pos..pos + len], - ) - .map_err(|e| io_err(&format!("annotation_index decode: {e}")))?; + ensure_bytes(data, pos, len, "legacy annotation arena section")?; + let arena = + ciborium::de::from_reader::(&data[pos..pos + len]) + .map_err(|e| io_err(&format!("legacy annotation arena decode: {e}")))?; pos += len; - Some(ann) + Some(arena) } else { None }; @@ -1170,13 +1071,13 @@ impl IndexRoot { // All optional sections consumed. `pos` should now equal the // input length — anything else means a future format added - // bytes after the annotation section, or the writer emitted + // bytes after the last section, or the writer emitted // garbage. Surfacing this here keeps decode strict and makes // appending a new section a deliberate change rather than a // silent compatibility break. if pos != data.len() { return Err(io_err(&format!( - "root v6: trailing bytes after annotation_index ({} unread)", + "root v6: trailing bytes ({} unread)", data.len() - pos ))); } @@ -1209,21 +1110,7 @@ impl IndexRoot { garbage, sketch_ref, has_annotations, - // Decode-side coercion (mirrors the encode-side - // coercion in `encode`): a root with a populated - // `annotation_index` implies `had_annotation_arena = - // true`, regardless of what the extended-flags byte - // says. This makes the contract durable across the - // upgrade boundary: a pre-this-change FIR6 root has - // `pad = 0` (so the raw bit decodes as `false`) but - // may already carry `annotation_index = Some(_)`. Such - // a root has provably sealed an arena, so the sticky - // semantics must apply — otherwise a later defensive - // drop would land at the bootstrap-eligible state and - // the provider would re-seal from a live-only scan, - // losing the dropped arena's history. - had_annotation_arena: had_annotation_arena || annotation_index.is_some(), - annotation_index, + legacy_annotation_arena, term_dict, has_list_meta, }) @@ -1236,13 +1123,13 @@ impl IndexRoot { /// - Per-graph arena CIDs (numbig, vectors, spatial, fulltext). /// - Default-graph inline leaf CIDs and their sidecar CIDs. /// - Named-graph **branch** CIDs (the leaves they route to are NOT included). - /// - Annotation forward + reverse **branch** CIDs (the leaves behind those - /// branches are NOT included). + /// - A legacy annotation arena's two **branch** CIDs (the leaves behind + /// them are NOT included). /// - Sketch CID. /// /// Does NOT include the root's own CID or the garbage manifest CID. /// - /// Callers that need leaf CIDs sitting behind named-graph or annotation + /// Callers that need leaf CIDs sitting behind named-graph or legacy arena /// branches must fetch + decode those branch manifests from storage. See /// [`crate::collect_root_cas_ids_expanded`] (strict) and /// [`crate::collect_root_cas_ids_expanded_tolerant`] (best-effort) for @@ -1311,12 +1198,8 @@ impl IndexRoot { ids.push(sketch.clone()); } - // Annotation arena: forward + reverse branch CIDs. - // Leaves behind these branches are NOT included here — see the - // doc comment above and the expanded helper. - if let Some(ref ann) = self.annotation_index { - ids.push(ann.forward_branch_cid.clone()); - ids.push(ann.reverse_branch_cid.clone()); + if let Some(ref arena) = self.legacy_annotation_arena { + ids.extend(arena.branches().cloned()); } // Triple-term dictionary: forward packs + both reverse trees. @@ -1535,9 +1418,8 @@ mod tests { garbage: None, sketch_ref: None, has_annotations: false, - annotation_index: None, + legacy_annotation_arena: None, term_dict: None, - had_annotation_arena: false, has_list_meta: None, ns_split_mode: fluree_db_core::ns_encoding::NsSplitMode::default(), } @@ -1563,7 +1445,7 @@ mod tests { assert_eq!(decoded.default_graph_orders.len(), 0); assert_eq!(decoded.named_graphs.len(), 0); assert!(decoded.stats.is_none()); - assert!(decoded.annotation_index.is_none()); + assert!(decoded.legacy_annotation_arena.is_none()); } /// The term dictionary section round-trips with both reverse trees, the @@ -1606,80 +1488,73 @@ mod tests { } } + /// A root written while the annotation arena existed still loads: both + /// decoders step over its section to the term dictionary after it, GC + /// still reaches the arena's branches, and re-encoding drops it. #[test] - fn fir6_round_trip_with_annotation_index() { - let mut root = minimal_root_v6(); - let dummy_cid = fluree_db_core::ContentId::from_hex_digest( - fluree_db_core::CODEC_FLUREE_ANNOTATION_FORWARD_BRANCH, - &fluree_db_core::sha256_hex(b"fwd"), - ) - .unwrap(); - let dummy_rev = fluree_db_core::ContentId::from_hex_digest( - fluree_db_core::CODEC_FLUREE_ANNOTATION_REVERSE_BRANCH, - &fluree_db_core::sha256_hex(b"rev"), + fn fir6_decodes_a_legacy_annotation_arena_section() { + #[derive(serde::Serialize)] + struct Section { + version: u8, + max_t: i64, + forward_branch_cid: ContentId, + reverse_branch_cid: ContentId, + stats: BTreeMap<&'static str, u64>, + } + let cid = |tag: &[u8]| ContentId::new(fluree_db_core::ContentKind::Commit, tag); + let arena = LegacyAnnotationArena { + forward_branch_cid: cid(b"fwd"), + reverse_branch_cid: cid(b"rev"), + }; + let mut section = Vec::new(); + ciborium::ser::into_writer( + &Section { + version: 1, + max_t: 9, + forward_branch_cid: arena.forward_branch_cid.clone(), + reverse_branch_cid: arena.reverse_branch_cid.clone(), + stats: BTreeMap::from([("forward_rows", 3)]), + }, + &mut section, ) .unwrap(); - root.annotation_index = Some(fluree_db_core::AnnotationIndexRoot { - version: 1, - max_t: 99, - forward_branch_cid: dummy_cid.clone(), - reverse_branch_cid: dummy_rev.clone(), - stats: fluree_db_core::AnnotationStats { - forward_rows: 100, - reverse_rows: 100, - distinct_edges: 25, - distinct_annotations: 75, - ..Default::default() + + let mut root = minimal_root_v6(); + root.has_annotations = true; + let section_at = root.encode().len(); + root.term_dict = Some(crate::format::wire_helpers::TermDictRefs { + forward_packs: vec![], + reverse: crate::format::wire_helpers::DictTreeRefs { + branch: cid(b"term-branch"), + leaves: vec![], }, + watermarks: vec![], + term_count: 0, + object_reverse: None, }); - // Sticky-bit contract: a populated `annotation_index` implies - // the seal pass set `had_annotation_arena=true`. The encoder - // debug-asserts the in-memory pair so a regression in the - // seal bookkeeping surfaces here rather than as a phantom - // bootstrap-eligible state on the next reindex. - root.had_annotation_arena = true; - - // `minimal_root_v6` leaves `has_annotations = false`. The - // encoder must coerce the sticky bit on whenever an arena is - // present, so the cascade fast-path can never desynchronize - // from a populated arena. - assert!(!root.has_annotations); - let bytes = root.encode(); - assert_ne!(bytes[5] & IndexRoot::FLAG_HAS_ANNOTATION_INDEX, 0); - assert_ne!( - bytes[5] & IndexRoot::FLAG_HAS_ANNOTATIONS, - 0, - "arena present must imply sticky bit on the wire" - ); + let current = root.encode(); + let mut legacy = current[..section_at].to_vec(); + legacy[5] |= IndexRoot::FLAG_LEGACY_ANNOTATION_ARENA; + legacy[6] |= 1 << 0; // the retired sticky bit legacy encoders set + legacy.extend_from_slice(&(section.len() as u32).to_le_bytes()); + legacy.extend_from_slice(§ion); + legacy.extend_from_slice(¤t[section_at..]); + + let decoded = IndexRoot::decode(&legacy).unwrap(); + assert_eq!(decoded.legacy_annotation_arena.as_ref(), Some(&arena)); + assert_eq!(decoded.term_dict, root.term_dict); + assert!(decoded.has_annotations); + let snap = fluree_db_core::LedgerSnapshot::from_root_bytes(&legacy) + .expect("core metadata decode skips the section"); + assert!(snap.has_annotations); - let decoded = IndexRoot::decode(&bytes).unwrap(); - assert!( - decoded.has_annotations, - "decoded sticky bit follows the encoded flag" - ); - let ann = decoded - .annotation_index - .expect("annotation_index roundtrip"); - assert_eq!(ann.version, 1); - assert_eq!(ann.max_t, 99); - assert_eq!(ann.forward_branch_cid, dummy_cid); - assert_eq!(ann.reverse_branch_cid, dummy_rev); - assert_eq!(ann.stats.forward_rows, 100); - assert_eq!(ann.stats.distinct_edges, 25); - assert_eq!(ann.stats.distinct_annotations, 75); - - // Metadata-only path (`LedgerSnapshot::from_root_bytes`) must - // surface the same pointer — callers that only load metadata - // still see the section. - let snap = fluree_db_core::LedgerSnapshot::from_root_bytes(&bytes).unwrap(); - assert!(snap.has_annotations, "snapshot path also sees sticky bit"); - let snap_ann = snap.annotation_index.expect("metadata path roundtrip"); - assert_eq!(snap_ann, ann); + let ids = decoded.all_cas_ids(); + assert!(arena.branches().all(|branch| ids.contains(branch))); + + let reencoded = decoded.encode(); + assert_eq!(reencoded, current, "re-encoding drops the legacy section"); } - /// A caller holding the decoded root builds the snapshot from it instead - /// of decoding the bytes again; that snapshot must be the one the - /// metadata-only decoder builds from the same bytes. #[test] fn snapshot_metadata_from_the_decoded_root_matches_the_metadata_decoder() { use fluree_db_core::index_stats::{ClassStatEntry, GraphStatsEntry}; @@ -1692,7 +1567,6 @@ mod tests { root.subject_watermarks = vec![5, 9]; root.string_watermark = 11; root.has_annotations = true; - root.had_annotation_arena = true; root.has_list_meta = Some(true); root.stats = Some(IndexStats { flakes: 3, @@ -1735,11 +1609,7 @@ mod tests { &s.subject_watermarks, s.string_watermark, graphs, - ( - s.has_annotations, - &s.annotation_index, - s.had_annotation_arena - ), + s.has_annotations, s.has_list_meta, ) ) @@ -1812,165 +1682,10 @@ mod tests { ); } - #[test] - fn fir6_round_trip_had_annotation_arena_sticky() { - // The sticky bit lives in the extended-flags byte at - // `data[6]` (low byte of the historically-zero `pad(2)` - // header field). Old encoders wrote `0u16` there, so old - // roots decode to `had_annotation_arena = false` — - // matching the "never sealed" state. - let mut root = minimal_root_v6(); - assert!( - !root.had_annotation_arena, - "minimal helper starts with the bit clear" - ); - let bytes_clear = root.encode(); - assert_eq!( - bytes_clear[6] & IndexRoot::FLAG_EXT_HAD_ANNOTATION_ARENA, - 0, - "extended-flags byte (data[6]) carries the bit; clear when had_annotation_arena=false" - ); - let decoded_clear = IndexRoot::decode(&bytes_clear).unwrap(); - assert!(!decoded_clear.had_annotation_arena); - - // Flip the bit, re-encode, verify byte position + decode. - root.had_annotation_arena = true; - let bytes_set = root.encode(); - assert_ne!( - bytes_set[6] & IndexRoot::FLAG_EXT_HAD_ANNOTATION_ARENA, - 0, - "extended-flags byte must reflect had_annotation_arena=true" - ); - let decoded_set = IndexRoot::decode(&bytes_set).unwrap(); - assert!(decoded_set.had_annotation_arena); - - // Metadata-only path (`LedgerSnapshot::from_root_bytes`) - // must surface the same value. - let snap = fluree_db_core::LedgerSnapshot::from_root_bytes(&bytes_set).unwrap(); - assert!( - snap.had_annotation_arena, - "metadata-only decode also surfaces the sticky bit" - ); - - // Backward compat 1: an artificially-zeroed extended-flags - // byte on a no-arena root (simulating a pre-this-change FIR6 - // root that never sealed an arena) decodes to - // `had_annotation_arena = false`, NOT a panic or - // unsupported-version error. - let mut bytes_legacy = bytes_set.clone(); - bytes_legacy[6] = 0; - bytes_legacy[7] = 0; - let decoded_legacy = IndexRoot::decode(&bytes_legacy).unwrap(); - assert!( - !decoded_legacy.had_annotation_arena, - "old roots with pad=0 and no arena decode to bit clear \ - (the 'never sealed' state)" - ); - - // Backward compat 2: a pre-this-change FIR6 root that - // *already had an arena sealed* (annotation_index = - // Some(_)) has `pad = 0` on the wire because the encoder - // didn't know about the extended-flags byte. The decoder - // must coerce `had_annotation_arena = true` from the - // presence of `annotation_index`, otherwise a later - // defensive drop would land in the bootstrap-eligible - // state and the provider's base-index scan-fallback could - // re-seal from a live-only scan, losing the dropped - // arena's history. - let dummy_fwd = fluree_db_core::ContentId::from_hex_digest( - fluree_db_core::CODEC_FLUREE_ANNOTATION_FORWARD_BRANCH, - &fluree_db_core::sha256_hex(b"fwd"), - ) - .unwrap(); - let dummy_rev = fluree_db_core::ContentId::from_hex_digest( - fluree_db_core::CODEC_FLUREE_ANNOTATION_REVERSE_BRANCH, - &fluree_db_core::sha256_hex(b"rev"), - ) - .unwrap(); - let mut root_with_arena = minimal_root_v6(); - root_with_arena.annotation_index = Some(fluree_db_core::AnnotationIndexRoot { - version: 1, - max_t: 7, - forward_branch_cid: dummy_fwd, - reverse_branch_cid: dummy_rev, - stats: fluree_db_core::AnnotationStats::default(), - }); - // Set the sticky bit true (matches the in-memory invariant - // enforced by `encode()`'s `debug_assert!`) so the encode - // call succeeds; we then forcibly zero the extended-flags - // bytes to simulate a pre-this-change encoder that wrote - // `pad = 0` on the wire even when an arena was sealed. - // What's actually under test here is the *decoder's* - // legacy-byte coercion, not the encoder. - root_with_arena.had_annotation_arena = true; - let mut bytes_legacy_with_arena = root_with_arena.encode(); - bytes_legacy_with_arena[6] = 0; - bytes_legacy_with_arena[7] = 0; - let decoded = IndexRoot::decode(&bytes_legacy_with_arena).unwrap(); - assert!( - decoded.had_annotation_arena, - "legacy root with annotation_index=Some + pad=0 must coerce \ - had_annotation_arena=true on decode (otherwise a later \ - defensive drop would silently re-enable provider bootstrap)" - ); - let snap_legacy = - fluree_db_core::LedgerSnapshot::from_root_bytes(&bytes_legacy_with_arena).unwrap(); - assert!( - snap_legacy.had_annotation_arena, - "metadata-only decode applies the same coercion" - ); - } - - // debug_assert!-only invariant: not compiled in release/`bench` profile - // builds (where bench-gate runs the unit tests), so gate on debug_assertions - // — otherwise `#[should_panic]` fails because the assert is stripped. - #[cfg(debug_assertions)] - #[test] - #[should_panic(expected = "had_annotation_arena=false with annotation_index=Some")] - fn fir6_encoder_debug_asserts_sticky_bit_when_arena_present() { - // The in-memory contract is now the source of truth: an - // `IndexRoot` with `annotation_index = Some(_)` must also - // carry `had_annotation_arena = true`. `encode()` debug- - // asserts the invariant so a regression in seal/drop - // bookkeeping fails fast in dev/CI rather than relying on - // the wire-format coercion to paper over it at release - // time. - // - // The encoder's `|| self.annotation_index.is_some()` - // coercion (around line 633) is retained as a release-build - // safety net — if `debug_assertions = false` and a caller - // somehow violates the invariant anyway, the wire bit is - // still forced on, preserving correct - // bootstrap-vs-defensive-drop semantics on the next - // reindex. - let dummy_fwd = fluree_db_core::ContentId::from_hex_digest( - fluree_db_core::CODEC_FLUREE_ANNOTATION_FORWARD_BRANCH, - &fluree_db_core::sha256_hex(b"fwd"), - ) - .unwrap(); - let dummy_rev = fluree_db_core::ContentId::from_hex_digest( - fluree_db_core::CODEC_FLUREE_ANNOTATION_REVERSE_BRANCH, - &fluree_db_core::sha256_hex(b"rev"), - ) - .unwrap(); - let mut root = minimal_root_v6(); - root.annotation_index = Some(fluree_db_core::AnnotationIndexRoot { - version: 1, - max_t: 7, - forward_branch_cid: dummy_fwd, - reverse_branch_cid: dummy_rev, - stats: fluree_db_core::AnnotationStats::default(), - }); - // Deliberately leave the sticky bit clear: the debug_assert - // in `encode()` must fire. - root.had_annotation_arena = false; - let _ = root.encode(); - } - #[test] fn fir6_rejects_trailing_bytes() { // A future format extension that appended bytes after the - // annotation section must not silently load on this build. + // last section must not silently load on this build. // Both decode paths reject the stray byte explicitly. let bytes = minimal_root_v6().encode(); let mut tampered = bytes.clone(); @@ -2210,12 +1925,9 @@ mod tests { orders: vec![(RunSortOrder::Spot, branch_cid.clone())], }]; root.sketch_ref = Some(sketch_cid.clone()); - root.annotation_index = Some(fluree_db_core::AnnotationIndexRoot { - version: 1, - max_t: 0, + root.legacy_annotation_arena = Some(LegacyAnnotationArena { forward_branch_cid: ann_fwd_cid.clone(), reverse_branch_cid: ann_rev_cid.clone(), - stats: fluree_db_core::AnnotationStats::default(), }); let ids = root.all_cas_ids(); @@ -2229,7 +1941,7 @@ mod tests { assert!(ids.contains(&branch_cid), "missing branch_cid"); // Sketch present assert!(ids.contains(&sketch_cid), "missing sketch_cid"); - // Annotation arena branch CIDs present (leaves behind them are not). + // Legacy arena branch CIDs present (leaves behind them are not). assert!( ids.contains(&ann_fwd_cid), "missing annotation forward branch CID" diff --git a/fluree-db-binary-index/src/lib.rs b/fluree-db-binary-index/src/lib.rs index bc46f4560e..bc4ffd15ec 100644 --- a/fluree-db-binary-index/src/lib.rs +++ b/fluree-db-binary-index/src/lib.rs @@ -10,7 +10,6 @@ pub mod error; pub mod types; pub mod wasm_compat; -pub mod annotation_arena; pub mod arena; pub mod dict; pub mod dict_novelty_safe; @@ -33,9 +32,10 @@ pub use read::replay::{batch_has_rows_above_t, replay_leaflet, replay_leaflet_at // ── Format types ──────────────────────────────────────────────────────────── pub use format::branch::{BranchManifest, LeafEntry}; pub use format::expanded_cas::{ - collect_root_cas_ids_expanded, collect_root_cas_ids_expanded_tolerant, ChainCasIds, + collect_root_cas_ids_expanded, collect_root_cas_ids_expanded_tolerant, + legacy_annotation_arena_cids, ChainCasIds, }; -pub use format::index_root::IndexRoot; +pub use format::index_root::{IndexRoot, LegacyAnnotationArena}; pub use format::run_record::{cmp_for_order, cmp_psot, cmp_spot, RunRecord, RunSortOrder}; pub use format::wire_helpers::{ BinaryGarbageRef, BinaryPrevIndexRef, DictPackRefs, DictRefs, DictTreeRefs, FulltextArenaRef, diff --git a/fluree-db-binary-index/tests/it_arena_novelty_merge.rs b/fluree-db-binary-index/tests/it_arena_novelty_merge.rs deleted file mode 100644 index 5497634c4a..0000000000 --- a/fluree-db-binary-index/tests/it_arena_novelty_merge.rs +++ /dev/null @@ -1,232 +0,0 @@ -//! End-to-end test: arena builder → CAS → reader, merged with a real -//! `AttachmentNovelty` overlay. -//! -//! Exercises the full M2b read path against a populated arena and an -//! overlay holding post-arena events. The `(Sid, t, op)` shape returned -//! by `AttachmentNovelty::collect_forward_events` flows directly into -//! `AnnotationArenaReader::current_annotations_merged` with no -//! intermediate plumbing — this test pins that contract. - -use fluree_db_binary_index::annotation_arena::{ - build_arenas_from_flakes, build_forward_branch, build_reverse_branch, AnnotationArenaReader, - DEFAULT_TARGET_ROWS_PER_LEAF, -}; -use fluree_db_core::storage::{ContentStore, MemoryContentStore}; -use fluree_db_core::{edge::EdgeKey, AnnotationIndexRoot, ContentKind, Flake, FlakeValue, Sid}; -use fluree_db_novelty::AttachmentNovelty; -use fluree_vocab::db as db_predicates; - -fn ann_sid(name: &str) -> Sid { - Sid::new(20, name) -} -fn ref_sid(name: &str) -> Sid { - Sid::new(11, name) -} -fn id_dt() -> Sid { - fluree_db_core::id_datatype_sid() -} -fn p(suffix: &str) -> Sid { - Sid::new(fluree_vocab::namespaces::FLUREE_DB, suffix) -} - -fn make_bundle(ann: &str, s: &str, predicate: &str, o: &str, t: i64, op: bool) -> Vec { - let a = ann_sid(ann); - vec![ - Flake::new( - a.clone(), - p(db_predicates::REIFIES_SUBJECT), - FlakeValue::Ref(ref_sid(s)), - id_dt(), - t, - op, - None, - ), - Flake::new( - a.clone(), - p(db_predicates::REIFIES_PREDICATE), - FlakeValue::Ref(ref_sid(predicate)), - id_dt(), - t, - op, - None, - ), - Flake::new( - a, - p(db_predicates::REIFIES_OBJECT), - FlakeValue::Ref(ref_sid(o)), - id_dt(), - t, - op, - None, - ), - ] -} - -async fn build_and_store(flakes: &[Flake], store: &MemoryContentStore) -> AnnotationIndexRoot { - let out = build_arenas_from_flakes(flakes, DEFAULT_TARGET_ROWS_PER_LEAF); - - let mut fwd_pairs = Vec::new(); - for (summary, blob) in out.forward_leaves { - let cid = store - .put(ContentKind::AnnotationForwardLeaf, &blob) - .await - .unwrap(); - fwd_pairs.push((summary, cid)); - } - let fwd_branch_bytes = build_forward_branch(&fwd_pairs); - let fwd_branch_cid = store - .put(ContentKind::AnnotationForwardBranch, &fwd_branch_bytes) - .await - .unwrap(); - - let mut rev_pairs = Vec::new(); - for (summary, blob) in out.reverse_leaves { - let cid = store - .put(ContentKind::AnnotationReverseLeaf, &blob) - .await - .unwrap(); - rev_pairs.push((summary, cid)); - } - let rev_branch_bytes = build_reverse_branch(&rev_pairs); - let rev_branch_cid = store - .put(ContentKind::AnnotationReverseBranch, &rev_branch_bytes) - .await - .unwrap(); - - AnnotationIndexRoot { - version: 1, - max_t: out.max_t, - forward_branch_cid: fwd_branch_cid, - reverse_branch_cid: rev_branch_cid, - stats: out.stats, - } -} - -#[tokio::test] -async fn arena_assert_then_novelty_retract_resolves_correctly() { - // Arena holds the historical attachment; novelty holds a later - // retract. Merged read returns no live annotation; indexed-only - // read still sees the assertion. - let edge_flakes = make_bundle("ann_a", "alice", "worksFor", "acme", 5, true); - let store = MemoryContentStore::new(); - let root = build_and_store(&edge_flakes, &store).await; - let reader = AnnotationArenaReader::new(&root, &store); - - let edge = EdgeKey { - g: None, - s: ref_sid("alice"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - - // Indexed-only: ann_a still appears live. - let indexed = reader.current_annotations_for(&edge, 100).await.unwrap(); - assert_eq!(indexed, vec![ann_sid("ann_a")]); - - // Build an overlay with the matching retract bundle. - let mut overlay = AttachmentNovelty::new(); - let retract_bundle = edge.to_reifies_facts(&ann_sid("ann_a"), 10, false); - overlay.observe_flakes(&retract_bundle).unwrap(); - - let novelty_events = overlay.collect_forward_events(&edge); - assert_eq!(novelty_events.len(), 1, "overlay holds one retract event"); - - // Merged read: the arena assert + novelty retract resolves to no - // live annotation. - let merged = reader - .current_annotations_merged(&edge, &novelty_events, 100) - .await - .unwrap(); - assert!(merged.is_empty(), "novelty retract overrides arena assert"); - - // As-of t=8 — novelty retract not yet visible — ann_a still live. - let merged_t8 = reader - .current_annotations_merged(&edge, &novelty_events, 8) - .await - .unwrap(); - assert_eq!(merged_t8, vec![ann_sid("ann_a")]); -} - -#[tokio::test] -async fn novelty_only_attachments_visible_against_empty_arena() { - // No arena rows; the only attachment lives in novelty. Merged read - // sees it. - let store = MemoryContentStore::new(); - let root = build_and_store(&[], &store).await; - let reader = AnnotationArenaReader::new(&root, &store); - - let edge = EdgeKey { - g: None, - s: ref_sid("alice"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - - let mut overlay = AttachmentNovelty::new(); - let bundle = edge.to_reifies_facts(&ann_sid("ann_new"), 10, true); - overlay.observe_flakes(&bundle).unwrap(); - - let events = overlay.collect_forward_events(&edge); - let merged = reader - .current_annotations_merged(&edge, &events, 100) - .await - .unwrap(); - assert_eq!(merged, vec![ann_sid("ann_new")]); -} - -#[tokio::test] -async fn reverse_arena_plus_novelty_retarget() { - // ann_x reifies edge_a in arena. Novelty retracts that and asserts - // that ann_x now reifies edge_b. Merged reverse lookup returns - // edge_b only. - let edge_a_flakes = make_bundle("ann_x", "alice", "worksFor", "acme", 1, true); - let store = MemoryContentStore::new(); - let root = build_and_store(&edge_a_flakes, &store).await; - let reader = AnnotationArenaReader::new(&root, &store); - - let edge_a = EdgeKey { - g: None, - s: ref_sid("alice"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - let edge_b = EdgeKey { - g: None, - s: ref_sid("bob"), - p: ref_sid("worksFor"), - o: FlakeValue::Ref(ref_sid("acme")), - dt: id_dt(), - lang: None, - list_i: None, - }; - - let mut overlay = AttachmentNovelty::new(); - overlay - .observe_flakes(&edge_a.to_reifies_facts(&ann_sid("ann_x"), 5, false)) - .unwrap(); - overlay - .observe_flakes(&edge_b.to_reifies_facts(&ann_sid("ann_x"), 6, true)) - .unwrap(); - - let novelty_events = overlay.collect_reverse_events(&ann_sid("ann_x")); - assert_eq!( - novelty_events.len(), - 2, - "overlay surfaces both retract and re-assert events" - ); - - let merged = reader - .current_targets_merged(&ann_sid("ann_x"), &novelty_events, 100) - .await - .unwrap(); - assert_eq!(merged, vec![edge_b]); -} diff --git a/fluree-db-cli/src/commands/create.rs b/fluree-db-cli/src/commands/create.rs index bde3bdcdbc..be7f87481c 100644 --- a/fluree-db-cli/src/commands/create.rs +++ b/fluree-db-cli/src/commands/create.rs @@ -896,63 +896,6 @@ async fn run_bulk_import( println!(); } - // Annotation arena auto-seal pass. - // - // The bulk-import path writes `IndexRoot.annotation_index = None` - // even when the imported dataset contains `f:reifies*` flakes — - // `build_indexes_from_commits` doesn't drive the annotation - // arena builder, so a freshly-imported ledger always lands in - // arena-less / scan-fallback state until a reindex. - // - // To close the workflow gap, we follow a successful annotation- - // bearing import with `Fluree::reindex(...)`, which DOES seal the - // arena via the api's `AttachmentEventsProvider`. The cost is one - // extra index pass on annotation-bearing imports — the same work - // a user would otherwise run manually via `fluree index` as a - // second step. Non-annotation imports skip this entirely (the - // sticky bit gate on `Fluree::reindex` and the early-return in - // the indexer's arena builder both make it cheap to be wrong). - if result.has_annotations { - if !quiet { - eprintln!( - "{} Sealing annotation arena (one-shot reindex pass) — ledger contains `f:reifies*` flakes…", - "info:".cyan().bold() - ); - } - match fluree - .reindex(ledger, fluree_db_api::ReindexOptions::default()) - .await - { - Ok(_) => { - // Report what actually landed: the reindex can complete - // without sealing anything (no attachment-event coverage), - // and an unsealed ledger answers quoted-triple queries - // through the slow generic join chain. - let sealed = fluree - .ledger(ledger) - .await - .map(|state| state.snapshot.annotation_index.is_some()) - .unwrap_or(false); - if !quiet { - if sealed { - eprintln!("{} Annotation arena sealed.", "info:".cyan().bold()); - } else { - eprintln!( - "{} Annotation arena was not sealed (no attachment events resolved); quoted-triple queries fall back to the generic join chain until `fluree reindex` seals it.", - "warning:".yellow().bold() - ); - } - } - } - Err(e) => { - eprintln!( - "{} Annotation arena seal failed (import succeeded; arena will land on the next `fluree index`): {e}", - "warning:".yellow().bold() - ); - } - } - } - // Success: remove the crash breadcrumb so the presence of files in // `/crash/` continues to be a strong signal of *failed/crashed* // runs that need investigation. diff --git a/fluree-db-cli/src/commands/index.rs b/fluree-db-cli/src/commands/index.rs index a11a33e48a..bf8ba891a1 100644 --- a/fluree-db-cli/src/commands/index.rs +++ b/fluree-db-cli/src/commands/index.rs @@ -44,41 +44,9 @@ pub async fn index_ledger(fluree: &Fluree, ledger_id: &str) -> CliResult` directly. The heavy on-disk arena -//! formats (forward/reverse branch + leaf blobs) stay in -//! `fluree-db-binary-index::annotation_arena::format`. -//! -//! ## Empty-vs-absent semantics -//! -//! The indexed-arena guarantee depends on **both** signals on -//! [`crate::db::LedgerSnapshot`]: -//! -//! | `has_annotations` | `annotation_index` | Meaning | -//! |-------------------|--------------------|----------------------------------------| -//! | `false` | `None` | Hard guarantee: zero attachments. Cascade and reads short-circuit. | -//! | `true` | `Some(_)` | Builder ran. Forward/reverse arenas are authoritative for `t ≤ max_t`; novelty supplies the tail. | -//! | `true` | `None` | Pre-builder transitional state: snapshot may carry `f:reifies*` flakes but no arena yet — readers fall back to scan, cascade still runs. | -//! | `false` | `Some(_)` | Invariant violation. The FIR6 encoder coerces `FLAG_HAS_ANNOTATIONS` whenever an arena is present, so this state never reaches the wire. | -//! -//! Builders that produce arenas must set both signals; the encoder -//! defends against forgetting the sticky bit but cannot fix the -//! inverse (arena present, bool false in memory). -//! -//! See `docs/design/edge-annotations.md` for the design contract. - -use crate::ContentId; -use serde::{Deserialize, Serialize}; - -/// Aggregate counters populated at arena build time. Surfaced for -/// cost-based planning (M3) and storage inspection. -/// -/// **Every field is `#[serde(default)]`** so older arena roots — -/// written before any given field landed — deserialize cleanly with -/// zeros. This applies to the *original* four counters as well as -/// the per-slot NDV counters added in the M3.1 follow-up: a missing -/// field on the wire is treated as "no information" by the planner, -/// which falls back to the regular `IndexStats.properties` HLL. -/// Tagging all fields keeps the struct safely reorderable — if a -/// future contributor inserts a new field anywhere in source order, -/// old arenas still load. -#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] -pub struct AnnotationStats { - /// Total forward-arena rows (one per asserted/retracted attachment event). - #[serde(default)] - pub forward_rows: u64, - /// Total reverse-arena rows (mirror of `forward_rows` after compaction). - #[serde(default)] - pub reverse_rows: u64, - /// Distinct edges with at least one current (live) attachment. - #[serde(default)] - pub distinct_edges: u64, - /// Distinct annotation subjects. - #[serde(default)] - pub distinct_annotations: u64, - /// Number of live `(edge, ann)` attachment pairs. - /// - /// Equal to `distinct_annotations` under the v1 single-target- - /// per-ann invariant (enforced at stage time in - /// `fluree-db-transact::stage` — re-attaching an SID to a - /// different edge is a transaction error). When the invariant - /// holds, this field is redundant; we store it explicitly so the - /// planner stays accurate when reading older / replayed-from- - /// corrupt-history ledgers where the same ann SID may have - /// multiple live targets. The `f:reifies*` row count for the - /// required slots is `live_attachment_pairs`, not - /// `distinct_annotations`. - /// - /// `#[serde(default)]` so older arena roots written before this - /// field landed deserialize cleanly with `0`. The merge layer - /// treats `0` as "use distinct_annotations" — the safe equality - /// for healthy v1 ledgers. - #[serde(default)] - pub live_attachment_pairs: u64, - - // ----------------------------------------------------------------- - // Per-slot NDV counters across the live (currently-asserted) rows. - // - // For the three **required** slots (subject, predicate, object), - // each live `(edge, ann)` pair contributes exactly one row, so - // the row count equals `live_attachment_pairs` (which equals - // `distinct_annotations` under the v1 single-target invariant). - // We track only the NDV here. - // - // For the **optional** slots (graph, lang, listIndex), both the - // row count and the NDV vary: the row count is the number of - // live `(edge, ann)` pairs whose reified edge carries that slot, - // the NDV is the number of distinct values observed in that - // slot across those pairs. (`datatype` is a special case — see - // its field comment below.) - // ----------------------------------------------------------------- - /// Distinct subject SIDs across live reified edges. - /// Used as `ndv_values` for `?ann f:reifiesSubject ?s` probes. - #[serde(default)] - pub distinct_reified_subjects: u64, - /// Distinct predicate SIDs across live reified edges. - #[serde(default)] - pub distinct_reified_predicates: u64, - /// Distinct object values across live reified edges. - #[serde(default)] - pub distinct_reified_objects: u64, - - /// Live `f:reifiesGraph` rows = number of live `(edge, ann)` - /// pairs whose reified edge is in a named graph. Per pair, not - /// per distinct edge — parallel annotations on the same named- - /// graph edge each contribute one row. Generally `< - /// live_attachment_pairs` (the slot is omitted for default-graph - /// edges) but can exceed `distinct_annotations` when a single - /// ann SID is attached to multiple named-graph edges (the - /// multi-target anomaly the v1 stage-time invariant rejects). - #[serde(default)] - pub reifies_graph_rows: u64, - /// Distinct named-graph SIDs across live reified edges. - #[serde(default)] - pub distinct_reified_graphs: u64, - /// Distinct **annotation** SIDs that appear in `f:reifiesGraph` - /// rows. Used as `ndv_subjects` for ` f:reifiesGraph - /// ?g` probes — the right denominator for BoundSubject - /// selectivity (each row's subject is the ann SID, and not every - /// ann SID has a graph row when the slot is sparse). Under the - /// v1 single-target invariant this equals `reifies_graph_rows`; - /// under the multi-target anomaly it can be smaller. - #[serde(default)] - pub distinct_graph_anns: u64, - - /// **Always 0 from the arena builder.** The arena reconstructs - /// `EdgeKey.dt` from the flake-level dt of `f:reifiesObject`, so - /// it cannot tell whether the on-wire bundle actually emitted a - /// separate `f:reifiesDatatype` flake (full bundle path) or - /// omitted it (JSON-LD-compatible cascade). Reporting a synth - /// here would let `merge_annotation_stats` overwrite the real - /// `IndexStats.properties` HLL with a phantom row count. The - /// HLL is the source of truth for this slot. Field kept on the - /// struct for forward-compat in case a future builder tracks - /// the actual flake presence. - #[serde(default)] - pub reifies_datatype_rows: u64, - /// Always 0 from the arena builder; see `reifies_datatype_rows`. - #[serde(default)] - pub distinct_reified_datatypes: u64, - - /// Live `f:reifiesLang` rows. Per `(edge, ann)` pair; same - /// invariant story as `reifies_graph_rows`. - #[serde(default)] - pub reifies_lang_rows: u64, - /// Distinct language-tag values across live reified edges. - #[serde(default)] - pub distinct_reified_langs: u64, - /// Distinct annotation SIDs that appear in `f:reifiesLang` - /// rows. See `distinct_graph_anns` for rationale. - #[serde(default)] - pub distinct_lang_anns: u64, - - /// Live `f:reifiesListIndex` rows. v1 always 0 — list-element - /// annotations are deferred (see `docs/concepts/edge-annotations.md` "Current limits"). - #[serde(default)] - pub reifies_list_index_rows: u64, - /// Distinct list-index values across live reified edges. - #[serde(default)] - pub distinct_reified_list_indices: u64, - /// Distinct annotation SIDs that appear in `f:reifiesListIndex` - /// rows. See `distinct_graph_anns` for rationale. - #[serde(default)] - pub distinct_list_index_anns: u64, -} - -/// Inline section in the binary index root. -/// -/// Carries CIDs for the forward/reverse arena root branches plus -/// build-time stats. Always emitted (with empty CIDs and zero stats) -/// when the indexed snapshot might contain attachments — never as -/// `None` if uncertain. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct AnnotationIndexRoot { - /// Format version of the on-disk arena artifacts. Independent of - /// the parent index root's own version so the arena format can - /// roll forward without a full FIR6 bump. - pub version: u8, - /// Highest commit `t` reflected in either arena. Reads with - /// `as_of_t` above this fall back to novelty for any newer - /// attachments. - pub max_t: i64, - /// Forward-arena branch CID (`EAFB1`). - pub forward_branch_cid: ContentId, - /// Reverse-arena branch CID (`EARB1`). - pub reverse_branch_cid: ContentId, - /// Build-time stats. Always present (zero-valued for empty arenas). - pub stats: AnnotationStats, -} diff --git a/fluree-db-core/src/db.rs b/fluree-db-core/src/db.rs index 30070bac5c..708dafa9c7 100644 --- a/fluree-db-core/src/db.rs +++ b/fluree-db-core/src/db.rs @@ -62,55 +62,6 @@ pub struct LedgerSnapshotMetadata { /// `fluree-db-transact::stage` so non-annotation ledgers skip /// the per-retract POST scan entirely. pub has_annotations: bool, - /// Optional inline pointer to the on-disk annotation arenas. - /// - /// Combined with `has_annotations` and `had_annotation_arena`: - /// - `has_annotations=false, annotation_index=None` — hard - /// guarantee: zero attachments. - /// - `has_annotations=true, annotation_index=Some(_)` — builder - /// ran; arenas authoritative through `max_t`, novelty covers - /// the tail. - /// - `has_annotations=true, annotation_index=None, - /// had_annotation_arena=false` — fresh bulk-import state, no - /// indexer pass has yet processed the annotation-bearing - /// flakes. Readers fall back to scan; the provider may - /// bootstrap an `Authoritative` arena from a one-time - /// base-index scan. - /// - `has_annotations=true, annotation_index=None, - /// had_annotation_arena=true` — indexer-owned annotation - /// history with no current arena (defensive drop, or an - /// indexer pass that processed annotation events without - /// sealing). Readers fall back to scan; the provider MUST - /// NOT bootstrap from a live-only scan — the indexer - /// already owns history the live base doesn't fully - /// reflect, so a live-only reseal would silently drop - /// retract/reassert rows. - /// - `has_annotations=false, annotation_index=Some(_)` is an - /// invariant violation; the FIR6 encoder coerces the sticky - /// bit when an arena is present. - pub annotation_index: Option, - /// Sticky bit governing the `ApiAttachmentEventsProvider`'s - /// base-index scan-fallback. Despite the name, the - /// load-bearing meaning is "base-index bootstrap is **not** - /// allowed" — it's true on *any* indexer-produced root that - /// has annotation history (`has_annotations=true`), regardless - /// of whether an arena was actually sealed by that pass. - /// - /// Only fresh bulk-import roots leave the bit false, because - /// the import pipeline writes annotation flakes directly to - /// the base index without an indexer-owned event stream. - /// Those roots are the unique state where the provider may - /// reconstruct an `Authoritative` arena from a one-time - /// base-index scan. - /// - /// Once set, never cleared — including across defensive - /// drops, no-provider indexer passes, and any other state - /// transition. The FIR6 encoder coerces the bit on whenever - /// `annotation_index.is_some()`; the decoder coerces on whenever - /// either the extended-flags bit is set *or* an - /// `annotation_index` is present (handles pre-this-change - /// roots that already sealed an arena). - pub had_annotation_arena: bool, /// Whether any indexed row carries an RDF-list position. `Some(false)` /// lets the write path skip list-meta hydration; `None` means the root @@ -193,19 +144,6 @@ pub struct LedgerSnapshot { /// automatically. pub range_provider: Option>, - /// Optional CAS handle for arena-backed reads. - /// - /// Set by ledger-load paths that have a content store available - /// (the same one backing `range_provider`). Formatter / cascade - /// callers consult [`Self::annotation_index`] alongside this field - /// to decide whether to use the on-disk arena or fall back to the - /// scan-based hydration path. - /// - /// `Arc` (rather than a borrowed reference) - /// because the snapshot is `Clone` and outlives any individual - /// query / cascade scope. - pub content_store: Option>, - /// Ledger-wide graph IRI → GraphId registry. /// /// Populated from index root (via `seed_from_root_iris`) or ledger creation @@ -220,29 +158,6 @@ pub struct LedgerSnapshot { /// fast-path so non-annotation ledgers pay zero per-retract /// cost. pub has_annotations: bool, - /// On-disk annotation-arena pointer (forward/reverse branch CIDs + - /// stats). See `crate::annotation_index` for the truth table that - /// pairs this field with `has_annotations` and - /// `had_annotation_arena`. - pub annotation_index: Option, - /// Sticky bit governing whether the - /// `ApiAttachmentEventsProvider` is allowed to bootstrap an - /// `Authoritative` annotation arena from a one-time base-index - /// scan. Despite the name, the load-bearing meaning is - /// "base-index bootstrap is **not** allowed": - /// - `false` only on fresh bulk-import roots (no indexer pass - /// has touched the annotation history yet) — bootstrap is - /// safe because the live base IS the complete history. - /// - `true` on any indexer-produced root with - /// `has_annotations=true`, including defensive drops and - /// indexer passes that didn't seal an arena — bootstrap is - /// unsafe because the indexer owns history the live base - /// doesn't fully reflect. - /// - /// Sticky: once set, never cleared. Carried in - /// `IndexRoot.had_annotation_arena` via the FIR6 extended-flags - /// byte. - pub had_annotation_arena: bool, /// Whether any indexed row carries an RDF-list position. `Some(false)` /// lets the write path skip list-meta hydration; `None` means the root @@ -270,10 +185,7 @@ impl Clone for LedgerSnapshot { range_provider: self.range_provider.clone(), graph_registry: self.graph_registry.clone(), has_annotations: self.has_annotations, - annotation_index: self.annotation_index.clone(), - had_annotation_arena: self.had_annotation_arena, has_list_meta: self.has_list_meta, - content_store: self.content_store.clone(), } } } @@ -328,12 +240,9 @@ impl LedgerSnapshot { string_watermark: 0, range_provider: None, has_annotations: false, - annotation_index: None, - had_annotation_arena: false, // An empty snapshot has no indexed rows, so "no list rows" is // exact — everything lives in novelty, which tracks its own bit. has_list_meta: Some(false), - content_store: None, } } @@ -370,10 +279,7 @@ impl LedgerSnapshot { range_provider: None, graph_registry, has_annotations: meta.has_annotations, - annotation_index: meta.annotation_index, - had_annotation_arena: meta.had_annotation_arena, has_list_meta: meta.has_list_meta, - content_store: None, }) } @@ -398,24 +304,6 @@ impl LedgerSnapshot { self } - /// Attach a content store handle. Required (alongside - /// [`Self::annotation_index`]) for arena-backed annotation reads; - /// callers that only need range queries can leave this `None`. - pub fn with_content_store(mut self, store: Arc) -> Self { - self.content_store = Some(store); - self - } - - /// True iff this snapshot has both the index root section - /// pointing at on-disk arenas AND a CAS handle to read them. The - /// hot path for arena-backed lookups gates on this — when `false`, - /// callers fall back to the M2a scan-based hydration / cascade - /// paths. - #[inline] - pub fn has_arena_reader(&self) -> bool { - self.annotation_index.is_some() && self.content_store.is_some() - } - /// Encode an IRI to a SID using this db's namespace codes. /// /// If no registered namespace prefix matches the IRI, falls back to the @@ -707,25 +595,18 @@ fn decode_fir6_metadata(bytes: &[u8]) -> std::io::Result /// by the cascade fast-path in `fluree-db-transact::stage` to /// skip the per-retract POST scan on non-annotation ledgers. const FLAG_HAS_ANNOTATIONS: u8 = 1 << 6; - /// Optional `AnnotationIndexRoot` section is present in the inline - /// tail. Decoder uses ciborium; metadata-only callers skip it. - const FLAG_HAS_ANNOTATION_INDEX: u8 = 1 << 7; + /// Legacy: the root carries a retired annotation-arena section, skipped. + const FLAG_LEGACY_ANNOTATION_ARENA: u8 = 1 << 7; let has_annotations = flags & FLAG_HAS_ANNOTATIONS != 0; - let has_annotation_index_section = flags & FLAG_HAS_ANNOTATION_INDEX != 0; + let has_legacy_arena_section = flags & FLAG_LEGACY_ANNOTATION_ARENA != 0; // Extended-flags byte at bytes[6] (must match - // binary-index IndexRoot's `FLAG_EXT_*` set). Old roots - // wrote `0u16` to this position so the raw bit decodes - // false; the post-decode coercion below handles legacy - // roots whose `annotation_index` was sealed before this - // change shipped. - const FLAG_EXT_HAD_ANNOTATION_ARENA: u8 = 1 << 0; + // binary-index IndexRoot's `FLAG_EXT_*` set). Bit 0 is retired. const FLAG_EXT_LIST_META_TRACKED: u8 = 1 << 1; const FLAG_EXT_HAS_LIST_META: u8 = 1 << 2; const FLAG_EXT_HAS_TERM_DICT: u8 = 1 << 3; let flags_ext = bytes[6]; let has_term_dict_section = flags_ext & FLAG_EXT_HAS_TERM_DICT != 0; - let had_annotation_arena = flags_ext & FLAG_EXT_HAD_ANNOTATION_ARENA != 0; let has_list_meta = if flags_ext & FLAG_EXT_LIST_META_TRACKED == 0 { None } else { @@ -1067,28 +948,11 @@ fn decode_fir6_metadata(bytes: &[u8]) -> std::io::Result skip_cid(bytes, &mut pos)?; } - // Optional `AnnotationIndexRoot` section. Decoded eagerly via - // ciborium so the metadata-only fast path surfaces the same - // pointer that `IndexRoot::decode` would produce. The section - // is small (a couple of CIDs and counters), so eager decode - // costs ~tens of bytes — comparable to `stats`/`schema` already - // parsed above. - let annotation_index = if has_annotation_index_section { + if has_legacy_arena_section { let len = read_u32(bytes, &mut pos)? as usize; - ensure(bytes, pos, len, "annotation_index section")?; - let ann = - ciborium::de::from_reader::(&bytes[pos..pos + len]) - .map_err(|e| { - std::io::Error::new( - std::io::ErrorKind::InvalidData, - format!("FIR6: annotation_index decode: {e}"), - ) - })?; + ensure(bytes, pos, len, "legacy annotation arena section")?; pos += len; - Some(ann) - } else { - None - }; + } // Optional triple-term dictionary section: the metadata view has no use // for it, so it is skipped structurally. @@ -1096,29 +960,16 @@ fn decode_fir6_metadata(bytes: &[u8]) -> std::io::Result skip_term_dict_refs(bytes, &mut pos)?; } - // Trailing-byte sentinel: any unread bytes after the annotation - // section indicate a future format extension or a writer bug. + // Trailing-byte sentinel: any unread bytes after the last section + // indicate a future format extension or a writer bug. // Surface that explicitly rather than silently drop the bytes. if pos != bytes.len() { return Err(std::io::Error::new( std::io::ErrorKind::InvalidData, - format!( - "FIR6: trailing bytes after annotation_index ({} unread)", - bytes.len() - pos - ), + format!("FIR6: trailing bytes ({} unread)", bytes.len() - pos), )); } - // Decode-side coercion (mirrors `IndexRoot::decode`'s - // backward-compat fix): a root with a populated - // `annotation_index` implies `had_annotation_arena = true`, - // even if the extended-flags byte is zero (pre-this-change - // FIR6 roots that already had a sealed arena). Without this, - // a later defensive drop on such a root would land in the - // bootstrap-eligible state and the provider's base-index - // scan-fallback would silently lose retract/reassert history. - let had_annotation_arena = had_annotation_arena || annotation_index.is_some(); - Ok(LedgerSnapshotMetadata { ledger_id, t: index_t, @@ -1131,8 +982,6 @@ fn decode_fir6_metadata(bytes: &[u8]) -> std::io::Result string_watermark, graph_iris, has_annotations, - annotation_index, - had_annotation_arena, has_list_meta, }) } @@ -1177,43 +1026,6 @@ mod tests { assert!(err.to_string().contains("too short")); } - #[test] - fn has_arena_reader_requires_both_index_and_store() { - // Genesis snapshot: neither annotation_index nor content_store - // — has_arena_reader returns false. - let snap = LedgerSnapshot::genesis("test:main"); - assert!(!snap.has_arena_reader(), "genesis has no arena"); - - // Adding only the content store still doesn't enable arena - // reads — the snapshot must also point at on-disk arenas. - let store = Arc::new(crate::storage::MemoryContentStore::new()) - as Arc; - let snap = snap.with_content_store(store); - assert!( - !snap.has_arena_reader(), - "store without annotation_index does not enable arena reads" - ); - - // Adding annotation_index with no store still doesn't suffice. - let mut snap = snap; - snap.content_store = None; - snap.annotation_index = Some(crate::AnnotationIndexRoot { - version: 1, - max_t: 0, - forward_branch_cid: ContentId::new(ContentKind::AnnotationForwardBranch, b"empty-fwd"), - reverse_branch_cid: ContentId::new(ContentKind::AnnotationReverseBranch, b"empty-rev"), - stats: crate::AnnotationStats::default(), - }); - assert!( - !snap.has_arena_reader(), - "annotation_index without store does not enable arena reads" - ); - - // Both present — arena reader is available. - snap.content_store = Some(Arc::new(crate::storage::MemoryContentStore::new())); - assert!(snap.has_arena_reader()); - } - #[test] fn test_encode_decode_sid() { let mut ns = HashMap::new(); @@ -1231,8 +1043,6 @@ mod tests { string_watermark: 0, graph_iris: vec![], has_annotations: false, - annotation_index: None, - had_annotation_arena: false, has_list_meta: None, }) .unwrap(); diff --git a/fluree-db-core/src/edge.rs b/fluree-db-core/src/edge.rs index 17dbd6c6b5..3dddbed172 100644 --- a/fluree-db-core/src/edge.rs +++ b/fluree-db-core/src/edge.rs @@ -3,33 +3,12 @@ //! //! Annotations in Fluree reify a specific edge: the `(graph, subject, //! predicate, object, datatype, language, list-index)` tuple of a base -//! flake. `EdgeKey` captures exactly that tuple in a form suitable for: -//! -//! - keying the in-memory [`AttachmentNovelty`](../../fluree-db-novelty) -//! forward multimap (`EdgeKey -> Vec`), -//! - keying the on-disk forward arena once the M2 indexer lands, -//! - encoding to a durable system-fact bundle via the seven `f:reifies*` -//! predicates (the M1 source of truth), and -//! - decoding back from those facts at warmup or read time. -//! -//! Total ordering and serde derive cleanly from the field types. -//! -//! See `docs/design/edge-annotations.md` for the durable attachment -//! encoding contract this type implements. +//! flake. `EdgeKey` captures exactly that tuple; export keys its +//! edge → reifier lookup by it. -use crate::flake::{Flake, FlakeMeta}; -use crate::namespaces::{ - is_reifies_datatype, is_reifies_graph, is_reifies_lang, is_reifies_list_index, - is_reifies_object, is_reifies_predicate, is_reifies_subject, reifies_datatype_sid, - reifies_graph_sid, reifies_lang_sid, reifies_object_sid, reifies_predicate_sid, - reifies_subject_sid, -}; +use crate::flake::Flake; use crate::sid::Sid; use crate::value::FlakeValue; -#[cfg(test)] -use fluree_vocab::db as fluree_db_predicates; -#[cfg(test)] -use fluree_vocab::namespaces::FLUREE_DB; use fluree_vocab::namespaces::{JSON_LD, XSD}; use fluree_vocab::xsd_names; use serde::{Deserialize, Serialize}; @@ -43,7 +22,7 @@ pub fn id_datatype_sid() -> Sid { Sid::new(JSON_LD, "id") } -/// Datatype SID for `xsd:string` literals (used for `f:reifiesLang`). +/// Datatype SID for `xsd:string` literals. #[inline] pub fn xsd_string_datatype_sid() -> Sid { Sid::new(XSD, xsd_names::STRING) @@ -52,31 +31,7 @@ pub fn xsd_string_datatype_sid() -> Sid { /// A stable identifier for a base triple eligible to carry annotations. /// /// Fields mirror [`Flake`] one-for-one (minus `t`/`op`/`ann`-side bits) so -/// the conversion from a base flake is mechanical and the round-trip -/// encoding via `f:reifies*` system facts is lossless. -/// -/// ## Field order is a persisted on-disk key order — do not reorder -/// -/// `Ord` is derived, so declaration order *is* comparison priority, and the -/// edge-annotation arenas persist that order: forward-leaf rows are stored -/// sorted by `(EdgeKey, ann_sid, t, op)`, and branch entries carry -/// `first_edge`/`last_edge` bounds that readers seek against in `EdgeKey` -/// order (the sort contract and `AnnotationForwardBranchEntry` in -/// `fluree-db-binary-index/src/annotation_arena/format.rs`; the branch -/// cursor and the `partition_point` leaf seeks in `reader.rs`). Reordering -/// these fields changes the derived comparator out from under every -/// persisted arena — seeks against leaves sorted by the old order silently -/// return wrong rows — so a reorder requires an arena format migration, at -/// minimum an `ARENA_VERSION` bump plus a rebuild of existing arenas. -/// -/// In particular, `lang` sorting before `list_i` here is the *opposite* -/// nesting from [`FlakeMeta`]'s `Ord`, which compares the list index first -/// and breaks ties on the language tag. That mismatch is deliberate, not an -/// oversight to fix: nothing compares an `EdgeKey` with `FlakeMeta::cmp`, -/// so the two relations never have to agree. `list_i` is always `None` in -/// v1, which keeps the divergence invisible today — it becomes observable -/// exactly when list-occurrence annotations land, and "harmonizing" the -/// nesting then would invalidate every arena written before it. +/// the conversion from a base flake is mechanical. #[derive(Clone, Debug, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] pub struct EdgeKey { /// Named graph the edge lives in. `None` = default graph. @@ -91,8 +46,8 @@ pub struct EdgeKey { pub dt: Sid, /// Language tag for langString objects, when applicable. pub lang: Option, - /// List index for list-element flakes. v1 always `None` — - /// list-occurrence annotations are deferred (see decisions section). + /// List index for list-element flakes; always `None`, as triple terms + /// carry no list position. pub list_i: Option, } @@ -137,423 +92,6 @@ impl EdgeKey { }; self.lang.as_deref() == flake_lang && self.list_i == flake_list_i } - - /// Encode this edge as the durable `f:reifies*` system-fact bundle - /// for annotation subject `ann` at transaction time `t`, with the - /// given assertion `op` (`true` = assert, `false` = retract). - /// - /// **Full bundle shape** including the optional `f:reifiesDatatype` - /// flake. Used by in-Rust producers that have direct knowledge of - /// the datatype and want the redundant-but-explicit encoding for - /// diagnostic clarity (e.g., direct flake construction in tests). - /// - /// Most write paths in v1 — including the JSON-LD pre-expansion - /// lowering and the cascade pass — must use - /// [`Self::to_reifies_facts_jsonld_compatible`] instead, which - /// omits `f:reifiesDatatype` so the inverse retract bundle is - /// byte-symmetric with what was originally asserted. Asserting - /// the full form and retracting the JSON-LD-compatible form (or - /// vice versa) leaves a phantom retract / orphan in the durable - /// log. - /// - /// Callers must not split or reorder the bundle — partial bundles - /// are rejected by the replay validator (see - /// `docs/design/edge-annotations.md`). - pub fn to_reifies_facts(&self, ann: &Sid, t: i64, op: bool) -> Vec { - let mut facts = Vec::with_capacity(7); - let id_dt = id_datatype_sid(); - let str_dt = xsd_string_datatype_sid(); - - // Helper: emit each f:reifies* flake **into the same graph** - // as the reified edge. The JSON-LD lowering places annotation - // siblings under `@graph: `, so the asserted flakes have - // `g = Some(g_sid)` for named graphs. A retract bundle that - // hardcoded `g = None` would not match the original assertion - // (different `g` = different fact identity in Fluree's flake - // model), leaving named-graph annotations orphaned. - let make = |s: Sid, p: Sid, o: FlakeValue, dt: Sid| -> Flake { - match &self.g { - Some(g) => Flake::new_in_graph(g.clone(), s, p, o, dt, t, op, None), - None => Flake::new(s, p, o, dt, t, op, None), - } - }; - - // f:reifiesGraph — present iff edge is in a named graph. - // Predicate SIDs are cloned from the process-wide cache in - // `namespaces.rs` (Arc refcount bump, no allocation) instead - // of `Sid::new(FLUREE_DB, "reifies…")` per-flake. - if let Some(g) = &self.g { - facts.push(make( - ann.clone(), - reifies_graph_sid().clone(), - FlakeValue::Ref(g.clone()), - id_dt.clone(), - )); - } - - // f:reifiesSubject — required. - facts.push(make( - ann.clone(), - reifies_subject_sid().clone(), - FlakeValue::Ref(self.s.clone()), - id_dt.clone(), - )); - - // f:reifiesPredicate — required. - facts.push(make( - ann.clone(), - reifies_predicate_sid().clone(), - FlakeValue::Ref(self.p.clone()), - id_dt.clone(), - )); - - // f:reifiesObject — required. Preserves the original object's - // datatype on the flake so typed-equality lookups round-trip, - // and its language tag / list index in `FlakeMeta` so this - // flake is the structural inverse of the base edge's object. - // The JSON-LD writer expands a language-tagged value object to - // a flake carrying `m.lang`, so a cascade retract built here - // must match it (same `(s, p, o, dt, m)`) — otherwise the - // retract cancels nothing and leaves a durable orphan - // f:reifiesObject flake. `from_reifies_facts` ignores object - // `m` on decode, so this is transparent to the round-trip. - let mut object_flake = make( - ann.clone(), - reifies_object_sid().clone(), - self.o.clone(), - self.dt.clone(), - ); - object_flake.m = edge_key_to_flake_meta(self.lang.as_deref(), self.list_i); - facts.push(object_flake); - - // f:reifiesDatatype — required. Names the dt SID itself so - // queries can filter on the original object's datatype without - // inspecting the object value. - facts.push(make( - ann.clone(), - reifies_datatype_sid().clone(), - FlakeValue::Ref(self.dt.clone()), - id_dt, - )); - - // f:reifiesLang — optional, only when the original object - // carried a language tag. - if let Some(lang) = &self.lang { - facts.push(make( - ann.clone(), - reifies_lang_sid().clone(), - FlakeValue::String(lang.clone()), - str_dt, - )); - } - - // f:reifiesListIndex — deferred (v1 always omitted). - - facts - } - - /// Encode this edge as the **JSON-LD-compatible** `f:reifies*` - /// bundle: the same shape that - /// `fluree-db-transact::parse::edge_annotations` emits when - /// lowering an `@annotation` block before JSON-LD expansion. - /// - /// Differs from [`Self::to_reifies_facts`] by **omitting the - /// optional `f:reifiesDatatype` flake**. The decoder treats it as - /// optional and reconstructs the canonical datatype from the - /// flake-level `dt` of `f:reifiesObject`, so the bundle is - /// information-equivalent — but a *retraction* must match the - /// shape the original assertion produced or it leaves a phantom - /// retract in the flake log (a retract for a flake that was - /// never asserted). - /// - /// Use this variant in any path whose paired assertion shape is - /// the JSON-LD lowering: today that's the cascade pass in - /// `fluree-db-transact::stage::cascade_attachment_retracts` and - /// any future cleanup paths driven by `AttachmentNovelty`. - pub fn to_reifies_facts_jsonld_compatible(&self, ann: &Sid, t: i64, op: bool) -> Vec { - // Emit the full bundle, then drop the `f:reifiesDatatype` - // flake. Cheaper than reimplementing the seven-flake - // construction inline, and avoids drift between the two - // builders. - let mut facts = self.to_reifies_facts(ann, t, op); - facts.retain(|f| !is_reifies_datatype(&f.p)); - facts - } - - /// Inverse of [`Self::to_reifies_facts`]: reconstruct an `EdgeKey` - /// from its bundle of `f:reifies*` flakes. - /// - /// Validates **bundle completeness**: - /// - exactly one each of `Subject`, `Predicate`, `Object`, `Datatype` - /// - at most one `Graph` (absent = default graph) - /// - at most one `Lang` - /// - never any `ListIndex` (v1) - /// - all flakes share the same annotation subject (caller's - /// responsibility; this fn doesn't cross-validate `s` across the - /// slice) - /// - /// Validates **graph consistency** across the bundle: - /// - every `f:reifies*` flake in the slice must share the same - /// flake-level `g` (rejects `MixedFlakeGraphs` otherwise — a - /// tampered or partial bundle where rows leaked across graphs). - /// - when `f:reifiesGraph` is present, its value SID must match - /// the flake-level `g` of the bundle. When absent, the bundle's - /// flake-level `g` must be `None` (default graph). Mismatch is - /// `GraphMismatch`. - /// - /// Non-`f:reifies*` flakes (annotation metadata about the - /// annotation subject) are ignored — they describe the annotation, - /// not the edge, and may legitimately live in a different graph. - /// - /// Returns `Err` with a structured [`EdgeKeyDecodeError`] when the - /// bundle is malformed; the replay validator surfaces this through - /// telemetry. - pub fn from_reifies_facts(facts: &[Flake]) -> Result { - let mut g: Option = None; - let mut s_pos: Option = None; - let mut p_pos: Option = None; - let mut o_pos: Option<(FlakeValue, Sid)> = None; - let mut dt_pos: Option = None; - let mut lang: Option = None; - // Flake-level `g` shared by every `f:reifies*` flake in this - // bundle. `None` until the first `f:reifies*` flake is seen; - // `Some(None)` means "default graph"; `Some(Some(sid))` means - // "named graph `sid`". - let mut bundle_g: Option> = None; - - for f in facts { - // Classify the predicate. Non-`f:reifies*` flakes describe - // the annotation subject as ordinary RDF and don't - // participate in the bundle's graph anchor — skip them - // before tracking `bundle_g` so a metadata fact in a - // different graph doesn't false-trigger `MixedFlakeGraphs`. - let is_bundle_flake = crate::namespaces::is_reserved_reifies_predicate(&f.p); - if !is_bundle_flake { - continue; - } - - // Bundle graph consistency: every f:reifies* flake must - // share the same flake-level `g`. The novelty observer - // and arena builder already enforce this at the caller - // layer by grouping, but the invariant belongs here so - // any future caller that doesn't pre-group can't slip a - // mixed-graph bundle through. - match &bundle_g { - None => bundle_g = Some(f.g.clone()), - Some(seen) if seen != &f.g => { - return Err(EdgeKeyDecodeError::MixedFlakeGraphs); - } - _ => {} - } - - if is_reifies_graph(&f.p) { - if g.is_some() { - return Err(EdgeKeyDecodeError::Duplicate("f:reifiesGraph")); - } - let FlakeValue::Ref(sid) = &f.o else { - return Err(EdgeKeyDecodeError::WrongType("f:reifiesGraph")); - }; - g = Some(sid.clone()); - } else if is_reifies_subject(&f.p) { - if s_pos.is_some() { - return Err(EdgeKeyDecodeError::Duplicate("f:reifiesSubject")); - } - let FlakeValue::Ref(sid) = &f.o else { - return Err(EdgeKeyDecodeError::WrongType("f:reifiesSubject")); - }; - s_pos = Some(sid.clone()); - } else if is_reifies_predicate(&f.p) { - if p_pos.is_some() { - return Err(EdgeKeyDecodeError::Duplicate("f:reifiesPredicate")); - } - let FlakeValue::Ref(sid) = &f.o else { - return Err(EdgeKeyDecodeError::WrongType("f:reifiesPredicate")); - }; - p_pos = Some(sid.clone()); - } else if is_reifies_object(&f.p) { - if o_pos.is_some() { - return Err(EdgeKeyDecodeError::Duplicate("f:reifiesObject")); - } - o_pos = Some((f.o.clone(), f.dt.clone())); - } else if is_reifies_datatype(&f.p) { - if dt_pos.is_some() { - return Err(EdgeKeyDecodeError::Duplicate("f:reifiesDatatype")); - } - let FlakeValue::Ref(sid) = &f.o else { - return Err(EdgeKeyDecodeError::WrongType("f:reifiesDatatype")); - }; - dt_pos = Some(sid.clone()); - } else if is_reifies_lang(&f.p) { - if lang.is_some() { - return Err(EdgeKeyDecodeError::Duplicate("f:reifiesLang")); - } - let FlakeValue::String(s) = &f.o else { - return Err(EdgeKeyDecodeError::WrongType("f:reifiesLang")); - }; - lang = Some(s.clone()); - } else if is_reifies_list_index(&f.p) { - // v1 deferral: even seeing one is malformed. - return Err(EdgeKeyDecodeError::DeferredFeature("f:reifiesListIndex")); - } - // Non-`f:reifies*` flakes (annotation metadata) are ignored - // — they describe the annotation subject, not the edge. - } - - let s = s_pos.ok_or(EdgeKeyDecodeError::Missing("f:reifiesSubject"))?; - let p = p_pos.ok_or(EdgeKeyDecodeError::Missing("f:reifiesPredicate"))?; - let (o, o_dt_from_flake) = o_pos.ok_or(EdgeKeyDecodeError::Missing("f:reifiesObject"))?; - - // `f:reifiesDatatype` is optional: when present it must agree - // with the flake-level `dt` of the `f:reifiesObject` row (both - // encode the same value). When absent, the flake-level dt is - // canonical — the bundle is still complete because object - // datatype round-trips via the flake's own `dt` field. The - // pre-expansion JSON-LD lowering path emits the optional - // form; the in-Rust `to_reifies_facts` builder emits both for - // diagnostic clarity. - let dt = match dt_pos { - Some(separate) if separate != o_dt_from_flake => { - return Err(EdgeKeyDecodeError::DatatypeMismatch); - } - Some(separate) => separate, - None => o_dt_from_flake, - }; - - // Reconcile the decoded `f:reifiesGraph` value with the - // bundle's flake-level graph. They name the same thing from - // two angles: - // - `g` (from the optional `f:reifiesGraph` flake) is the - // graph the bundle *reifies* (the graph of the base edge). - // - `bundle_g` is the graph the bundle *lives in* (where - // the f:reifies* flakes themselves were asserted). - // The annotation subject lives in the same named graph as - // the edge it reifies (see `build_annotation_sibling` and - // `EdgeKey::to_reifies_facts`), so disagreement here means a - // tampered or buggy bundle. `bundle_g` is `Some` here because - // we required `f:reifiesSubject` above and that returns - // `Missing` otherwise. - let flake_level_g = bundle_g - .as_ref() - .expect("bundle_g set when at least f:reifiesSubject present") - .as_ref(); - if g.as_ref() != flake_level_g { - return Err(EdgeKeyDecodeError::GraphMismatch); - } - - Ok(Self { - g, - s, - p, - o, - dt, - lang, - // v1: always None. - list_i: None, - }) - } -} - -/// Errors decoding an [`EdgeKey`] from a bundle of `f:reifies*` flakes. -/// -/// All variants are recoverable — the replay validator skips the -/// malformed annotation, increments a telemetry counter, and continues. -/// The annotation's *non*-`f:reifies` metadata facts remain visible as -/// ordinary RDF (just without the attachment binding). -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum EdgeKeyDecodeError { - /// A required `f:reifies*` predicate was absent from the bundle. - Missing(&'static str), - /// A required `f:reifies*` predicate appeared more than once. - Duplicate(&'static str), - /// The flake's object had the wrong [`FlakeValue`] type for the - /// predicate (e.g. `f:reifiesSubject` carried a literal). - WrongType(&'static str), - /// `f:reifiesObject`'s flake-level datatype did not match the - /// `f:reifiesDatatype` value. Indicates a tampered or buggy bundle. - DatatypeMismatch, - /// Two or more `f:reifies*` flakes in the bundle had different - /// flake-level `g` values. The whole bundle must live in one - /// graph; a split slice indicates tampering, a partial-write, or - /// a caller that failed to pre-group rows by graph. - MixedFlakeGraphs, - /// The optional `f:reifiesGraph` value disagrees with the bundle's - /// flake-level `g`. The annotation subject lives in the same - /// named graph as the edge it reifies, so these two views of the - /// graph must match — a mismatch indicates a tampered or buggy - /// bundle (e.g. `f:reifiesGraph` missing on a named-graph edge, - /// or set to the wrong graph). - GraphMismatch, - /// A predicate that v1 explicitly defers was present. - DeferredFeature(&'static str), -} - -impl std::fmt::Display for EdgeKeyDecodeError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::Missing(p) => write!(f, "missing {p}"), - Self::Duplicate(p) => write!(f, "duplicate {p}"), - Self::WrongType(p) => write!(f, "wrong object type for {p}"), - Self::DatatypeMismatch => { - write!(f, "f:reifiesObject dt mismatch with f:reifiesDatatype") - } - Self::MixedFlakeGraphs => { - write!(f, "f:reifies* flakes in one bundle span multiple graphs") - } - Self::GraphMismatch => write!( - f, - "f:reifiesGraph value disagrees with the bundle's flake-level graph" - ), - Self::DeferredFeature(p) => write!(f, "{p} is deferred to a future milestone"), - } - } -} - -impl std::error::Error for EdgeKeyDecodeError {} - -/// Convert an `EdgeKey` to a fresh [`FlakeMeta`] capturing the lang/list -/// fields, returning `None` when both fields are absent (matching the -/// existing `Flake.m: Option` convention). -fn edge_key_to_flake_meta(lang: Option<&str>, list_i: Option) -> Option { - if lang.is_none() && list_i.is_none() { - return None; - } - Some(FlakeMeta { - lang: lang.map(String::from), - i: list_i, - }) -} - -impl EdgeKey { - /// Reconstruct a base [`Flake`] equivalent to the one this key was - /// derived from. Useful for cascade-retract paths that need to emit - /// an inverse flake matching the original by structure. - /// - /// `t` and `op` are caller-supplied — the cascade decides whether - /// it is asserting or retracting. - pub fn to_base_flake(&self, t: i64, op: bool) -> Flake { - let m = edge_key_to_flake_meta(self.lang.as_deref(), self.list_i); - match &self.g { - Some(g) => Flake::new_in_graph( - g.clone(), - self.s.clone(), - self.p.clone(), - self.o.clone(), - self.dt.clone(), - t, - op, - m, - ), - None => Flake::new( - self.s.clone(), - self.p.clone(), - self.o.clone(), - self.dt.clone(), - t, - op, - m, - ), - } - } } #[cfg(test)] @@ -601,383 +139,4 @@ mod tests { other.g = Some(Sid::new(13, "graph_b")); assert!(!key.matches(&other)); } - - #[test] - fn reifies_round_trip_default_graph_no_lang() { - let f = sample_flake(); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann1"); - let bundle = key.to_reifies_facts(&ann, 42, true); - // 4 required + 0 optional = 4 facts (no graph, no lang, no list_i). - assert_eq!(bundle.len(), 4); - let decoded = EdgeKey::from_reifies_facts(&bundle).expect("decode succeeds"); - assert_eq!(decoded, key); - } - - #[test] - fn reifies_round_trip_named_graph_with_lang() { - let mut f = sample_flake(); - f.g = Some(Sid::new(13, "graph_a")); - f.o = FlakeValue::String("Engineer".into()); - f.dt = Sid::new(2, "string"); // xsd:string - f.m = Some(FlakeMeta { - lang: Some("fr".into()), - i: None, - }); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann_named"); - let bundle = key.to_reifies_facts(&ann, 42, true); - assert_eq!(bundle.len(), 6, "graph + S + P + O + Dt + lang"); - let decoded = EdgeKey::from_reifies_facts(&bundle).expect("decode succeeds"); - assert_eq!(decoded, key); - } - - #[test] - fn reifies_object_flake_carries_lang_meta_for_cascade_symmetry() { - // BUGS-2: the f:reifiesObject flake must carry the language tag - // in its FlakeMeta so a cascade retract built from the EdgeKey - // is the structural inverse of the JSON-LD-asserted flake (which - // carries m.lang). Otherwise the retract cancels nothing and the - // f:reifiesObject flake survives as a durable orphan. - let mut f = sample_flake(); - f.o = FlakeValue::String("chat".into()); - f.dt = Sid::new(2, "string"); // stand-in for rdf:langString - f.m = Some(FlakeMeta { - lang: Some("fr".into()), - i: None, - }); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann_lang"); - - let assertion = key.to_reifies_facts(&ann, 5, true); - let obj = assertion - .iter() - .find(|fl| is_reifies_object(&fl.p)) - .expect("f:reifiesObject flake present"); - assert_eq!( - obj.m.as_ref().and_then(|m| m.lang.as_deref()), - Some("fr"), - "f:reifiesObject flake must carry m.lang" - ); - - // The retract bundle must match the assertion flake-for-flake - // (same o/dt/m) so the cascade actually cancels it. - let retract = key.to_reifies_facts(&ann, 7, false); - let obj_r = retract - .iter() - .find(|fl| is_reifies_object(&fl.p)) - .expect("f:reifiesObject flake present in retract"); - assert_eq!(obj.m, obj_r.m, "object flake meta must match on retract"); - assert_eq!(obj.o, obj_r.o); - assert_eq!(obj.dt, obj_r.dt); - } - - #[test] - fn decode_rejects_missing_required() { - let f = sample_flake(); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann1"); - let mut bundle = key.to_reifies_facts(&ann, 42, true); - bundle.retain(|f| !is_reifies_subject(&f.p)); - let err = EdgeKey::from_reifies_facts(&bundle).unwrap_err(); - assert_eq!(err, EdgeKeyDecodeError::Missing("f:reifiesSubject")); - } - - #[test] - fn decode_rejects_duplicate_required() { - let f = sample_flake(); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann1"); - let mut bundle = key.to_reifies_facts(&ann, 42, true); - let dup = bundle - .iter() - .find(|f| is_reifies_predicate(&f.p)) - .expect("predicate flake exists") - .clone(); - bundle.push(dup); - let err = EdgeKey::from_reifies_facts(&bundle).unwrap_err(); - assert_eq!(err, EdgeKeyDecodeError::Duplicate("f:reifiesPredicate")); - } - - #[test] - fn decode_rejects_list_index_in_v1() { - let f = sample_flake(); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann1"); - let mut bundle = key.to_reifies_facts(&ann, 42, true); - bundle.push(Flake::new( - ann.clone(), - Sid::new(FLUREE_DB, fluree_db_predicates::REIFIES_LIST_INDEX), - FlakeValue::Long(0), - Sid::new(XSD, xsd_names::INTEGER), - 42, - true, - None, - )); - let err = EdgeKey::from_reifies_facts(&bundle).unwrap_err(); - assert_eq!( - err, - EdgeKeyDecodeError::DeferredFeature("f:reifiesListIndex") - ); - } - - #[test] - fn decode_ignores_unrelated_flakes() { - // Annotation metadata flakes (e.g. `ann ex:role "Engineer"`) - // share the bundle but must be passed through transparently. - let f = sample_flake(); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann1"); - let mut bundle = key.to_reifies_facts(&ann, 42, true); - bundle.push(Flake::new( - ann, - Sid::new(13, "role"), - FlakeValue::String("Engineer".into()), - Sid::new(XSD, xsd_names::STRING), - 42, - true, - None, - )); - let decoded = EdgeKey::from_reifies_facts(&bundle).expect("metadata is ignored"); - assert_eq!(decoded, key); - } - - #[test] - fn jsonld_compatible_bundle_omits_datatype_but_round_trips() { - // The JSON-LD-compatible encoding drops the redundant - // `f:reifiesDatatype` flake. The decoder treats it as - // optional and reconstructs the canonical datatype from the - // flake-level `dt` of `f:reifiesObject`, so round-trip is - // lossless. - let f = sample_flake(); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann1"); - - let full = key.to_reifies_facts(&ann, 42, true); - let compact = key.to_reifies_facts_jsonld_compatible(&ann, 42, true); - - assert_eq!(full.len(), 4, "full default-graph bundle has 4 facts"); - assert_eq!( - compact.len(), - 3, - "JSON-LD-compatible bundle drops f:reifiesDatatype" - ); - assert!( - !compact.iter().any(|f| is_reifies_datatype(&f.p)), - "JSON-LD-compatible bundle must not contain f:reifiesDatatype: {compact:?}" - ); - - let decoded = EdgeKey::from_reifies_facts(&compact) - .expect("decoder treats f:reifiesDatatype as optional"); - assert_eq!(decoded, key); - } - - #[test] - fn named_graph_bundle_emits_flakes_in_the_named_graph() { - // Regression: every flake in a named-graph bundle (assertion - // or retract) must have `g = Some(graph_sid)` matching the - // reified edge's graph. A `g = None` retract would not - // match a `g = Some(...)` assertion in Fluree's flake - // identity model, leaving named-graph annotations orphaned. - let mut f = sample_flake(); - f.g = Some(Sid::new(13, "graph_a")); - f.o = FlakeValue::String("Engineer".into()); - f.dt = Sid::new(2, "string"); - f.m = Some(FlakeMeta { - lang: Some("fr".into()), - i: None, - }); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann1"); - - // Full bundle. - let full = key.to_reifies_facts(&ann, 42, true); - for flake in &full { - assert_eq!( - flake.g.as_ref(), - Some(&Sid::new(13, "graph_a")), - "named-graph bundle flake must carry the edge's graph: {flake:?}" - ); - } - - // JSON-LD-compatible (cascade) bundle. - let compact = key.to_reifies_facts_jsonld_compatible(&ann, 42, true); - for flake in &compact { - assert_eq!( - flake.g.as_ref(), - Some(&Sid::new(13, "graph_a")), - "JSON-LD-compat named-graph bundle must carry the edge's graph: {flake:?}" - ); - } - } - - #[test] - fn default_graph_bundle_keeps_g_none() { - // Symmetric counterpart: default-graph edges produce - // `g = None` flakes (the absence-encodes-default convention - // matches the assertion side). - let f = sample_flake(); - assert!(f.g.is_none(), "sample is default-graph"); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann1"); - let bundle = key.to_reifies_facts_jsonld_compatible(&ann, 42, true); - for flake in &bundle { - assert!( - flake.g.is_none(), - "default-graph bundle must keep g=None: {flake:?}" - ); - } - } - - #[test] - fn jsonld_compatible_bundle_retract_matches_assertion_shape() { - // The cascade contract: a retract bundle must be the - // structural inverse of what the write path asserted, with - // only `t` and `op` differing. This test pins that for the - // JSON-LD encoding: an assertion at t=5 + a retract at t=7 - // produces flake pairs that share `(s, p, o, dt, m)` and - // differ only in `(t, op)`. - let f = sample_flake(); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann1"); - - let assertion = key.to_reifies_facts_jsonld_compatible(&ann, 5, true); - let retract = key.to_reifies_facts_jsonld_compatible(&ann, 7, false); - - assert_eq!( - assertion.len(), - retract.len(), - "assertion and retract bundles must have the same size" - ); - for (a, r) in assertion.iter().zip(retract.iter()) { - assert_eq!(a.s, r.s, "subject must match"); - assert_eq!(a.p, r.p, "predicate must match"); - assert_eq!(a.o, r.o, "object value must match"); - assert_eq!(a.dt, r.dt, "datatype must match"); - assert_eq!(a.m, r.m, "metadata must match"); - assert!(a.op && !r.op, "ops must invert"); - assert_ne!(a.t, r.t, "t must differ"); - } - } - - #[test] - fn to_base_flake_round_trips_with_from_flake() { - let f = sample_flake(); - let key = EdgeKey::from_flake(&f); - let rebuilt = key.to_base_flake(f.t, f.op); - // EdgeKey-identity components should match. - assert_eq!(rebuilt.s, f.s); - assert_eq!(rebuilt.p, f.p); - assert_eq!(rebuilt.o, f.o); - assert_eq!(rebuilt.dt, f.dt); - assert_eq!(rebuilt.g, f.g); - } - - #[test] - fn decode_rejects_mixed_flake_graphs() { - // Build a default-graph bundle, then retarget one flake's - // `g` to a named graph. The decoder must reject the slice - // outright — every `f:reifies*` flake in one bundle shares - // one graph. Callers that produce mixed-graph slices have - // a logic bug; the decoder firewalls them rather than - // silently decoding to a wrong-graph EdgeKey. - let f = sample_flake(); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann_mix"); - let mut bundle = key.to_reifies_facts(&ann, 42, true); - // Pick one bundle flake and move it into a different graph. - let target = bundle - .iter_mut() - .find(|f| is_reifies_object(&f.p)) - .expect("f:reifiesObject flake exists"); - target.g = Some(Sid::new(13, "rogue_graph")); - let err = EdgeKey::from_reifies_facts(&bundle).unwrap_err(); - assert_eq!(err, EdgeKeyDecodeError::MixedFlakeGraphs); - } - - #[test] - fn decode_ignores_mixed_graphs_on_non_reifies_metadata() { - // Non-`f:reifies*` flakes describe the annotation subject, - // not the edge, and may legitimately live in another graph - // (uncommon but not malformed). The decoder skips them - // before tracking bundle graph, so they must NOT trigger - // `MixedFlakeGraphs` even when their `g` differs. - let f = sample_flake(); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann_meta"); - let mut bundle = key.to_reifies_facts(&ann, 42, true); - bundle.push(Flake::new_in_graph( - Sid::new(13, "side_graph"), - ann, - Sid::new(13, "role"), - FlakeValue::String("Engineer".into()), - Sid::new(XSD, xsd_names::STRING), - 42, - true, - None, - )); - let decoded = - EdgeKey::from_reifies_facts(&bundle).expect("metadata in another graph is OK"); - assert_eq!(decoded, key); - } - - #[test] - fn decode_rejects_graph_mismatch_when_named_graph_flake_missing() { - // A named-graph bundle (`f.g = Some(graph_a)` on every - // f:reifies* flake) without an `f:reifiesGraph` flake. The - // bundle decodes to `EdgeKey { g: None, ... }` (since no - // `f:reifiesGraph` was found) but its flakes live in - // `Some(graph_a)`. Mismatch → `GraphMismatch`. - let mut f = sample_flake(); - f.g = Some(Sid::new(13, "graph_a")); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann_g1"); - let mut bundle = key.to_reifies_facts(&ann, 42, true); - bundle.retain(|f| !is_reifies_graph(&f.p)); - let err = EdgeKey::from_reifies_facts(&bundle).unwrap_err(); - assert_eq!(err, EdgeKeyDecodeError::GraphMismatch); - } - - #[test] - fn decode_rejects_graph_mismatch_when_reifies_graph_points_elsewhere() { - // Symmetric counterpart: the `f:reifiesGraph` value points - // at a different graph SID than the one the bundle's flakes - // live in. A tampered or buggy writer. - let mut f = sample_flake(); - f.g = Some(Sid::new(13, "graph_a")); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann_g2"); - let mut bundle = key.to_reifies_facts(&ann, 42, true); - let graph_flake = bundle - .iter_mut() - .find(|f| is_reifies_graph(&f.p)) - .expect("f:reifiesGraph flake exists in a named-graph bundle"); - graph_flake.o = FlakeValue::Ref(Sid::new(13, "graph_b")); - let err = EdgeKey::from_reifies_facts(&bundle).unwrap_err(); - assert_eq!(err, EdgeKeyDecodeError::GraphMismatch); - } - - #[test] - fn decode_rejects_graph_mismatch_when_default_graph_flake_carries_reifies_graph() { - // Inverse of the missing-flake case: bundle flakes are all - // `g = None` (default graph) but a stray `f:reifiesGraph` - // claims it reifies a named-graph edge. The two views of - // the graph disagree. - let f = sample_flake(); - assert!(f.g.is_none(), "sample is default-graph"); - let key = EdgeKey::from_flake(&f); - let ann = Sid::new(13, "ann_g3"); - let mut bundle = key.to_reifies_facts(&ann, 42, true); - bundle.push(Flake::new( - ann, - Sid::new(FLUREE_DB, fluree_db_predicates::REIFIES_GRAPH), - FlakeValue::Ref(Sid::new(13, "graph_a")), - id_datatype_sid(), - 42, - true, - None, - )); - let err = EdgeKey::from_reifies_facts(&bundle).unwrap_err(); - assert_eq!(err, EdgeKeyDecodeError::GraphMismatch); - } } diff --git a/fluree-db-core/src/lib.rs b/fluree-db-core/src/lib.rs index bf4d7a8287..ca4fa7dc7d 100644 --- a/fluree-db-core/src/lib.rs +++ b/fluree-db-core/src/lib.rs @@ -26,7 +26,6 @@ pub mod address; pub mod address_path; -pub mod annotation_index; pub mod cancellation; pub mod clock; pub mod coerce; @@ -97,7 +96,6 @@ pub mod wasm_cache; // Re-export main types pub use address::{extract_identifier, extract_path, parse_fluree_address, ParsedFlureeAddress}; -pub use annotation_index::{AnnotationIndexRoot, AnnotationStats}; pub use cancellation::{QueryCancellation, QueryCancellationReason}; pub use coerce::{coerce_json_value, coerce_value, CoercionError, CoercionResult}; pub use commit::{ @@ -124,7 +122,7 @@ pub use content_kind::{ pub use datatype_constraint::DatatypeConstraint; pub use db::{load_ledger_snapshot, LedgerSnapshot, LedgerSnapshotMetadata}; pub use dict_novelty::DictNovelty; -pub use edge::{id_datatype_sid, xsd_string_datatype_sid, EdgeKey, EdgeKeyDecodeError}; +pub use edge::{id_datatype_sid, xsd_string_datatype_sid, EdgeKey}; pub use error::{Error, Result}; pub use flake::{normalize_lang_tag, Flake, FlakeMeta}; pub use graph_db_ref::GraphDbRef; diff --git a/fluree-db-core/src/stats_view.rs b/fluree-db-core/src/stats_view.rs index 178468da5c..f41d34efbb 100644 --- a/fluree-db-core/src/stats_view.rs +++ b/fluree-db-core/src/stats_view.rs @@ -3,7 +3,6 @@ //! `StatsView` provides O(1) lookups of property and class statistics, //! built from `IndexStats` at query time. -use crate::annotation_index::AnnotationStats; use crate::db::LedgerSnapshot; use crate::ids::{GraphId, RuntimePredicateId}; use crate::index_stats::{ClassStatEntry, IndexStats}; @@ -439,176 +438,6 @@ impl StatsView { pub fn has_graph_stats(&self) -> bool { !self.graph_properties.is_empty() } - - /// Overlay arena-derived statistics for the seven `f:reifies*` - /// system predicates onto this view, using the per-slot NDV - /// counters tracked by the arena builder. - /// - /// When a snapshot's `annotation_index` is present, the arena's - /// live counters are a more accurate source for `f:reifies*` - /// predicate cardinality than the generic `IndexStats.properties` - /// HLL, because: - /// - /// - `IndexStats.count` mixes asserts and retracts; the arena - /// counters are live-only. - /// - On freshly-indexed ledgers the property stats may be stale - /// or absent, but the arena counters are always current. - /// - /// **Required slots** (`f:reifiesSubject`, `f:reifiesPredicate`, - /// `f:reifiesObject`): every live `(edge, ann)` pair contributes - /// exactly one row, so `count = live_attachment_pairs`. Under - /// the v1 single-target-per-ann invariant - /// `live_attachment_pairs == distinct_annotations`, but a legacy - /// or replayed-from-corrupt-history ledger can have one ann SID - /// attached to multiple edges, in which case the pair count is - /// the right denominator. `ndv_subjects = distinct_annotations` - /// (the row's subject is the ann SID; even with multi-target - /// the distinct subject set is still the ann SIDs). - /// `ndv_values` uses the per-slot NDV - /// (`distinct_reified_subjects`, `_predicates`, `_objects`) - /// when available. Older arena roots were written before - /// per-slot NDVs were tracked and report `0` for those fields; - /// in that case we fall back to `ndv_values = 1` (the safe - /// upper bound — the planner sees every `BoundObject` probe as - /// a scan, which is conservative but not wrong). - /// - /// **Optional slots** (`f:reifiesGraph`, `f:reifiesLang`, - /// `f:reifiesListIndex`): synthesized **only when their per-slot - /// row count is non-zero**. The row count (e.g. - /// `reifies_graph_rows`) is the number of live `(edge, ann)` - /// pairs whose edge carries that slot — usually strictly less - /// than `live_attachment_pairs`, and equal to the per-slot - /// distinct ann SID count under the v1 single-target - /// invariant. The multi-target anomaly can push it above - /// `distinct_annotations` (one ann SID with many graph edges). - /// `ndv_subjects` uses the per-slot ann-SID count - /// (`distinct__anns`) when available, falling back to - /// `min(rows, distinct_annotations)` for older arena roots that - /// predate the per-slot counters. When the row count is zero - /// (older arenas with default-zeroed fields, or workloads that - /// never use that slot), we leave the entry to the regular - /// `IndexStats.properties` HLL. - /// - /// **`f:reifiesDatatype` is intentionally not synthesized.** The - /// arena reconstructs `EdgeKey.dt` from the flake-level dt of - /// `f:reifiesObject` and cannot tell whether the on-wire bundle - /// actually emitted a separate `f:reifiesDatatype` flake. The - /// arena builder reports zero for the datatype row count; - /// `merge_annotation_stats` ignores datatype entirely and lets - /// the regular HLL handle it. - /// - /// **Why per-slot NDV matters.** Without it (the original M3.1 - /// shipped state), `BoundObject` selectivity for - /// `?ann f:reifiesObject ex:acme` was `count / 1 = - /// distinct_annotations` — the same as a scan, so the planner - /// got nothing from arena stats beyond the row total. With it, - /// the estimate becomes `distinct_annotations / - /// distinct_reified_objects` which can drop selectivity by - /// orders of magnitude when objects are diverse. - /// - /// Existing entries for the seven `f:reifies*` predicates are - /// overwritten — the arena is authoritative for live attachment - /// counts. - pub fn merge_annotation_stats( - &mut self, - ann: &AnnotationStats, - namespace_codes: &HashMap, - ) { - use fluree_vocab::db as p; - use fluree_vocab::namespaces::FLUREE_DB; - - if ann.distinct_annotations == 0 { - return; - } - - // Required slots: `count` is the number of live `(edge, ann)` - // pairs (one row per pair per required slot). Older arena - // roots predate `live_attachment_pairs` and report it as 0; - // the v1 stage-time invariant says one ann SID has one live - // target, so falling back to `distinct_annotations` is safe - // for those. A multi-target anomaly on a current-format - // arena will report a `live_attachment_pairs` strictly - // greater than `distinct_annotations`, and the planner sees - // the larger row count. - let row_count = if ann.live_attachment_pairs > 0 { - ann.live_attachment_pairs - } else { - ann.distinct_annotations - }; - let req = |ndv: u64| PropertyStatData { - count: row_count, - ndv_values: ndv.max(1), - ndv_subjects: ann.distinct_annotations, - }; - - let prefix = namespace_codes.get(&FLUREE_DB); - let mut insert = |name: &str, data: PropertyStatData| { - self.properties.insert(Sid::new(FLUREE_DB, name), data); - if let Some(prefix) = prefix { - let iri: Arc = Arc::from(format!("{prefix}{name}")); - self.properties_by_iri.insert(iri, data); - } - }; - - insert(p::REIFIES_SUBJECT, req(ann.distinct_reified_subjects)); - insert(p::REIFIES_PREDICATE, req(ann.distinct_reified_predicates)); - insert(p::REIFIES_OBJECT, req(ann.distinct_reified_objects)); - - // Optional slots: synth only when the arena observed non-zero - // rows. `count = rows`, `ndv_values = distinct values`, - // `ndv_subjects = per-slot ann-SID count` when available. - // - // Older arena roots predate the per-slot ann-SID counters - // and report `0` for them. In that case we fall back to - // `min(rows, distinct_annotations).max(1)` — a heuristic - // that's exact under the v1 single-target invariant (rows - // == distinct slot anns) but is wrong in either direction - // when one ann SID has many slot rows: a sparse slot where - // one anomaly ann holds most rows ends up with - // `ndv_subjects == rows` (the cap doesn't bind) and - // `BoundSubject` undercounts. The principled fix when this - // matters is to reindex with the per-slot counters - // populated — current builders always emit them. - let cap_subjects = ann.distinct_annotations; - let mut opt = |name: &str, rows: u64, ndv: u64, slot_anns: u64| { - if rows == 0 { - return; - } - let subjects = if slot_anns > 0 { - slot_anns - } else { - rows.min(cap_subjects).max(1) - }; - insert( - name, - PropertyStatData { - count: rows, - ndv_values: ndv.max(1), - ndv_subjects: subjects.max(1), - }, - ); - }; - opt( - p::REIFIES_GRAPH, - ann.reifies_graph_rows, - ann.distinct_reified_graphs, - ann.distinct_graph_anns, - ); - // `f:reifiesDatatype` is intentionally skipped — arena - // builder reports zeros (see `AnnotationStats::reifies_datatype_rows`). - opt( - p::REIFIES_LANG, - ann.reifies_lang_rows, - ann.distinct_reified_langs, - ann.distinct_lang_anns, - ); - opt( - p::REIFIES_LIST_INDEX, - ann.reifies_list_index_rows, - ann.distinct_reified_list_indices, - ann.distinct_list_index_anns, - ); - } } #[cfg(test)] @@ -716,324 +545,6 @@ mod tests { assert_eq!(prop.ndv_subjects, 45); } - #[test] - fn merge_annotation_stats_uses_per_slot_ndv_for_required() { - use fluree_vocab::db as p; - use fluree_vocab::namespaces::FLUREE_DB; - - // 800 annotations across 200 edges (one target per ann - // under the v1 invariant, so live_attachment_pairs == 800): - // 50 distinct subjects, 4 distinct predicates, 200 distinct - // objects. The planner's BoundObject formula - // `count / ndv_values` should give: - // reifiesSubject: 800 / 50 = 16 annotations per subject - // reifiesPredicate: 800 / 4 = 200 annotations per predicate - // reifiesObject: 800 / 200 = 4 annotations per object - let mut view = StatsView::default(); - let ann = AnnotationStats { - forward_rows: 1_000, - reverse_rows: 1_000, - distinct_edges: 200, - distinct_annotations: 800, - live_attachment_pairs: 800, - distinct_reified_subjects: 50, - distinct_reified_predicates: 4, - distinct_reified_objects: 200, - ..Default::default() - }; - let mut ns = HashMap::new(); - ns.insert(FLUREE_DB, "https://ns.flur.ee/db#".to_string()); - - view.merge_annotation_stats(&ann, &ns); - - let subj = view - .get_property(&Sid::new(FLUREE_DB, p::REIFIES_SUBJECT)) - .expect("reifiesSubject synth missing"); - assert_eq!(subj.count, 800); - assert_eq!(subj.ndv_subjects, 800); - assert_eq!( - subj.ndv_values, 50, - "reifiesSubject ndv = distinct subjects" - ); - - let pred = view - .get_property(&Sid::new(FLUREE_DB, p::REIFIES_PREDICATE)) - .expect("reifiesPredicate synth missing"); - assert_eq!( - pred.ndv_values, 4, - "reifiesPredicate ndv = distinct predicates" - ); - - let obj = view - .get_property(&Sid::new(FLUREE_DB, p::REIFIES_OBJECT)) - .expect("reifiesObject synth missing"); - assert_eq!(obj.ndv_values, 200, "reifiesObject ndv = distinct objects"); - - // Optional slots: zero rows in this fixture → not synthesized, - // falls through to whatever IndexStats.properties holds. - for name in [ - p::REIFIES_GRAPH, - p::REIFIES_DATATYPE, - p::REIFIES_LANG, - p::REIFIES_LIST_INDEX, - ] { - assert!( - view.get_property(&Sid::new(FLUREE_DB, name)).is_none(), - "optional slot {name} must not be synth'd when row count is zero" - ); - } - } - - #[test] - fn merge_annotation_stats_uses_pair_count_when_multi_target_anomaly() { - // Anomalous shape: 100 distinct annotation SIDs, but one of - // them is attached to 3 different edges (legacy / replayed- - // from-corrupt-history — the v1 stage-time invariant should - // prevent this on healthy ledgers). live_attachment_pairs - // is 102, which is the correct row count for the required - // slots. The planner should see count = 102, not 100, so - // BoundObject estimates don't undercount. - use fluree_vocab::db as p; - use fluree_vocab::namespaces::FLUREE_DB; - - let mut view = StatsView::default(); - let ann = AnnotationStats { - distinct_edges: 100, - distinct_annotations: 100, - live_attachment_pairs: 102, - distinct_reified_subjects: 100, - distinct_reified_predicates: 5, - distinct_reified_objects: 100, - ..Default::default() - }; - view.merge_annotation_stats(&ann, &HashMap::new()); - - let subj = view - .get_property(&Sid::new(FLUREE_DB, p::REIFIES_SUBJECT)) - .expect("reifiesSubject synth missing"); - assert_eq!( - subj.count, 102, - "row count must follow live_attachment_pairs, not distinct_annotations" - ); - assert_eq!( - subj.ndv_subjects, 100, - "ndv_subjects = distinct ann SIDs (subject of each row)" - ); - } - - #[test] - fn merge_annotation_stats_falls_back_to_distinct_annotations_for_old_arenas() { - // Older arena roots predate `live_attachment_pairs` and - // deserialize as 0. Under the v1 invariant the pair count - // equals distinct_annotations, so the merge falls back to - // distinct_annotations as the row count. - use fluree_vocab::db as p; - use fluree_vocab::namespaces::FLUREE_DB; - - let mut view = StatsView::default(); - let ann = AnnotationStats { - distinct_annotations: 50, - // live_attachment_pairs missing (== 0) - distinct_reified_subjects: 10, - ..Default::default() - }; - view.merge_annotation_stats(&ann, &HashMap::new()); - let subj = view - .get_property(&Sid::new(FLUREE_DB, p::REIFIES_SUBJECT)) - .expect("reifiesSubject synth missing"); - assert_eq!( - subj.count, 50, - "older arena: count falls back to distinct_annotations" - ); - } - - #[test] - fn merge_annotation_stats_uses_per_slot_ann_count_for_sparse_anomaly() { - // Reviewer's case: 1000 annotations total, but only one of - // them carries a graph slot — and that one is attached to - // 21 distinct named-graph edges (the multi-target anomaly). - // So `reifies_graph_rows = 21`, `distinct_graph_anns = 1`. - // Without the per-slot ann counter, the cap fallback - // `min(rows, distinct_annotations) = min(21, 1000) = 21` - // would set ndv_subjects = 21, making BoundSubject estimate - // `21 / 21 = 1` row when the true answer for the anomalous - // ann is 21 rows. With the per-slot counter, ndv_subjects - // = 1 → BoundSubject = 21 / 1 = 21. Exact. - use fluree_vocab::db as p; - use fluree_vocab::namespaces::FLUREE_DB; - - let mut view = StatsView::default(); - let ann = AnnotationStats { - distinct_edges: 1020, - distinct_annotations: 1000, - live_attachment_pairs: 1020, - distinct_reified_subjects: 1000, - distinct_reified_predicates: 5, - distinct_reified_objects: 1020, - reifies_graph_rows: 21, - distinct_reified_graphs: 1, - distinct_graph_anns: 1, - ..Default::default() - }; - view.merge_annotation_stats(&ann, &HashMap::new()); - - let graph = view - .get_property(&Sid::new(FLUREE_DB, p::REIFIES_GRAPH)) - .expect("reifiesGraph synth missing"); - assert_eq!(graph.count, 21); - assert_eq!( - graph.ndv_subjects, 1, - "ndv_subjects must use distinct_graph_anns, not the row count" - ); - } - - #[test] - fn merge_annotation_stats_caps_optional_slot_subjects_under_multi_target() { - // Multi-target anomaly: 50 distinct annotation SIDs but - // 70 live `(edge, ann)` pairs (one ann is attached to 21 - // distinct named-graph edges = 21 rows from a single - // subject). All 70 pairs are in named graphs across 3 - // distinct graph SIDs, so reifies_graph_rows = 70. - // Under the old `ndv_subjects = rows` rule, the planner - // would compute `BoundSubject` selectivity as - // `70 / 70 = 1` row per known annotation, when the actual - // number of rows per known annotation can be 21. With the - // cap, ndv_subjects = min(70, 50) = 50, giving - // `70 / 50 ≈ 2` per known annotation — closer to the - // truth, never an undercount. - use fluree_vocab::db as p; - use fluree_vocab::namespaces::FLUREE_DB; - - let mut view = StatsView::default(); - let ann = AnnotationStats { - distinct_edges: 70, - distinct_annotations: 50, - live_attachment_pairs: 70, - distinct_reified_subjects: 50, - distinct_reified_predicates: 5, - distinct_reified_objects: 70, - reifies_graph_rows: 70, - distinct_reified_graphs: 3, - ..Default::default() - }; - view.merge_annotation_stats(&ann, &HashMap::new()); - - let graph = view - .get_property(&Sid::new(FLUREE_DB, p::REIFIES_GRAPH)) - .expect("reifiesGraph synth missing"); - assert_eq!(graph.count, 70); - assert_eq!(graph.ndv_values, 3); - assert_eq!( - graph.ndv_subjects, 50, - "ndv_subjects must be capped by distinct_annotations under multi-target" - ); - } - - #[test] - fn merge_annotation_stats_falls_back_when_per_slot_ndv_absent() { - // Older arena roots written before per-slot NDV tracking - // landed deserialize with zeroed `distinct_reified_*`. The - // merge must treat zero as "no information" and fall back to - // the safe `ndv_values = 1` upper bound rather than producing - // a degenerate `count / 0` estimate. - use fluree_vocab::db as p; - use fluree_vocab::namespaces::FLUREE_DB; - - let mut view = StatsView::default(); - let ann = AnnotationStats { - distinct_edges: 200, - distinct_annotations: 800, - // distinct_reified_* all default to 0 - ..Default::default() - }; - view.merge_annotation_stats(&ann, &HashMap::new()); - - for name in [p::REIFIES_SUBJECT, p::REIFIES_PREDICATE, p::REIFIES_OBJECT] { - let entry = view - .get_property(&Sid::new(FLUREE_DB, name)) - .unwrap_or_else(|| panic!("{name} synth missing")); - assert_eq!(entry.count, 800); - assert_eq!( - entry.ndv_values, 1, - "{name} ndv_values must fall back to 1 when arena reports zero" - ); - } - } - - #[test] - fn merge_annotation_stats_synthesizes_optional_slots_when_present() { - // 100 annotations, 20 of them in named graphs across 3 - // distinct graph SIDs, 10 with langString objects across 2 - // distinct languages. The optional-slot synth uses (rows, - // ndv) directly rather than `distinct_annotations`. - use fluree_vocab::db as p; - use fluree_vocab::namespaces::FLUREE_DB; - - let mut view = StatsView::default(); - let ann = AnnotationStats { - distinct_edges: 100, - distinct_annotations: 100, - distinct_reified_subjects: 100, - distinct_reified_predicates: 5, - distinct_reified_objects: 100, - reifies_graph_rows: 20, - distinct_reified_graphs: 3, - reifies_lang_rows: 10, - distinct_reified_langs: 2, - // datatype + listIndex omitted → stays 0 - ..Default::default() - }; - view.merge_annotation_stats(&ann, &HashMap::new()); - - let graph = view - .get_property(&Sid::new(FLUREE_DB, p::REIFIES_GRAPH)) - .expect("reifiesGraph synth missing"); - assert_eq!(graph.count, 20, "reifiesGraph count = rows in live state"); - assert_eq!(graph.ndv_subjects, 20); - assert_eq!(graph.ndv_values, 3); - - let lang = view - .get_property(&Sid::new(FLUREE_DB, p::REIFIES_LANG)) - .expect("reifiesLang synth missing"); - assert_eq!(lang.count, 10); - assert_eq!(lang.ndv_values, 2); - - // listIndex still zero rows → not synthesized. - assert!(view - .get_property(&Sid::new(FLUREE_DB, p::REIFIES_LIST_INDEX)) - .is_none()); - - // f:reifiesDatatype is never synthesized from the arena even - // when the caller hands non-zero counts — the arena builder - // emits zero by contract because it cannot reliably observe - // the on-wire flake count. Pin that the synth path skips it - // unconditionally. - let mut view2 = StatsView::default(); - let ann_with_dt = AnnotationStats { - distinct_annotations: 100, - reifies_datatype_rows: 100, // hypothetical — builder never sets this - distinct_reified_datatypes: 4, - ..Default::default() - }; - view2.merge_annotation_stats(&ann_with_dt, &HashMap::new()); - assert!( - view2 - .get_property(&Sid::new(FLUREE_DB, p::REIFIES_DATATYPE)) - .is_none(), - "reifiesDatatype must never be synthesized from arena stats" - ); - } - - #[test] - fn merge_annotation_stats_zero_annotations_is_noop() { - let mut view = StatsView::default(); - let ann = AnnotationStats::default(); - let ns = HashMap::new(); - view.merge_annotation_stats(&ann, &ns); - assert!(view.properties.is_empty()); - assert!(view.properties_by_iri.is_empty()); - } - #[test] fn test_class_lookup() { let class_sid = Sid::new(2, "Person"); diff --git a/fluree-db-indexer/src/build/annotation_arena.rs b/fluree-db-indexer/src/build/annotation_arena.rs deleted file mode 100644 index 5d56ad0f6b..0000000000 --- a/fluree-db-indexer/src/build/annotation_arena.rs +++ /dev/null @@ -1,421 +0,0 @@ -//! Indexer-side orchestration for sealing edge-annotation arenas. -//! -//! Glues the pure builder in -//! `fluree_db_binary_index::annotation_arena` to the CAS write seam -//! the indexer already drives for branches/leaves. The function takes -//! a pre-decoded set of attachment events (sourced from the running -//! ledger's `AttachmentNovelty.iter_event_pairs()` and threaded in -//! via `IndexerConfig.attachment_events`) plus the previous root's -//! arena, builds forward + reverse blobs, writes them, and returns -//! the populated [`AnnotationIndexRoot`] that the root encoder will -//! seal. -//! -//! ## What this module owns -//! -//! - Merging events from two sources (previous arena + novelty). -//! - Driving the CAS writes for arena leaves and branches. -//! - Returning a structurally-correct `AnnotationIndexRoot` plus the -//! replaced leaf CIDs the GC pass needs. -//! -//! ## What this module does NOT own -//! -//! - **Decoding events from the commit stream.** The orchestrator -//! layer collects the pre-decoded events from the running -//! `AttachmentNovelty` and threads them through `IndexerConfig`. - -use fluree_db_binary_index::annotation_arena::{ - build_arenas_from_event_pairs, build_forward_branch, build_reverse_branch, - AnnotationArenaReader, DEFAULT_TARGET_ROWS_PER_LEAF, -}; -// Note: collect_all_forward_events was used when this module merged -// the previous arena's events into a delta. The current contract is -// that callers pass the complete event history (typically from -// AttachmentNovelty.iter_event_pairs()), so the previous arena -// participates only for GC reachability via all_leaf_cids. -use fluree_db_core::storage::ContentStore; -use fluree_db_core::{AnnotationIndexRoot, ContentKind, EdgeKey, Sid}; - -use crate::error::{IndexerError, Result}; - -/// Output of [`build_and_persist_annotation_arena`]. -/// -/// `replaced_leaf_cids` enumerates every leaf CID referenced by the -/// previous arena (if any). `new_leaf_cids` enumerates every leaf CID -/// referenced by the arena just sealed. Pass BOTH to -/// `IncrementalRootBuilder::set_annotation_index` so it can record -/// only the old CIDs the new arena no longer references as garbage — -/// content-addressed storage means a re-sealed unchanged arena -/// produces identical CIDs, and GC must not delete leaves/branches the -/// new root still points at. The previous and new branch CIDs are -/// reconciled inside `set_annotation_index` (old from -/// `root.annotation_index`, new from the passed `new_index`). -/// -#[derive(Debug, Default)] -pub struct PersistedArenaResult { - pub new_index: Option, - pub replaced_leaf_cids: Vec, - pub new_leaf_cids: Vec, -} - -/// Build and persist an annotation arena from a complete event set. -/// -/// Writes forward + reverse leaf and branch blobs to CAS and returns -/// the populated [`AnnotationIndexRoot`] plus the previous arena's -/// leaf CIDs (for GC bookkeeping) in [`PersistedArenaResult`]. -/// -/// ## Contract: `events` is the complete history, not a delta -/// -/// `events` must contain every `f:reifies*` attachment event the new -/// snapshot should publish — typically the running ledger's full -/// `AttachmentNovelty.iter_event_pairs()` (which preserves events -/// across reindexes), clipped to `t <= IndexRoot.index_t`. The arena -/// is rebuilt from scratch from this set; the previous arena -/// participates **only** for GC reachability — its leaf CIDs are -/// returned in `replaced_leaf_cids` so callers can record them as -/// reclaimable. -/// -/// This rebuild-from-scratch contract is safer than a delta merge: -/// the api-side `AttachmentNovelty` accumulates events across -/// reindexes (no `clear_up_to` on attachments), so a "delta" + "base -/// arena" merge would double-count any event indexed in a prior pass. -/// -/// Returns `Ok(PersistedArenaResult { new_index: None, .. })` only -/// when **both** `events` is empty AND there is no previous arena — -/// preserving the "zero attachments" guarantee for non-annotation -/// ledgers. When the previous arena is `Some` and `events` is empty, -/// the new root advertises an empty arena (still authoritative — -/// empty events explicitly assert "no attachments live anywhere"). -pub async fn build_and_persist_annotation_arena( - content_store: &dyn ContentStore, - previous_index: Option<&AnnotationIndexRoot>, - events: Vec<(EdgeKey, Sid, i64, bool)>, -) -> Result { - if previous_index.is_none() && events.is_empty() { - return Ok(PersistedArenaResult::default()); - } - - // Collect previous-arena leaf CIDs for GC. We do NOT merge the - // previous arena's events with `events` — `events` is the - // complete history per the contract above. - let replaced_leaf_cids: Vec = match previous_index { - Some(prev) => { - let reader = AnnotationArenaReader::new(prev, content_store); - reader.all_leaf_cids().await.map_err(IndexerError::Core)? - } - None => Vec::new(), - }; - - let out = build_arenas_from_event_pairs(events, DEFAULT_TARGET_ROWS_PER_LEAF); - - // Forward leaves first. - let mut fwd_pairs = Vec::with_capacity(out.forward_leaves.len()); - for (summary, blob) in out.forward_leaves { - let cid = content_store - .put(ContentKind::AnnotationForwardLeaf, &blob) - .await - .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; - fwd_pairs.push((summary, cid)); - } - let fwd_branch_bytes = build_forward_branch(&fwd_pairs); - let fwd_branch_cid = content_store - .put(ContentKind::AnnotationForwardBranch, &fwd_branch_bytes) - .await - .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; - - let mut rev_pairs = Vec::with_capacity(out.reverse_leaves.len()); - for (summary, blob) in out.reverse_leaves { - let cid = content_store - .put(ContentKind::AnnotationReverseLeaf, &blob) - .await - .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; - rev_pairs.push((summary, cid)); - } - let rev_branch_bytes = build_reverse_branch(&rev_pairs); - let rev_branch_cid = content_store - .put(ContentKind::AnnotationReverseBranch, &rev_branch_bytes) - .await - .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; - - // Leaf CIDs the new arena references — handed to - // `set_annotation_index` so it can exclude any that the previous - // arena also referenced (identical re-sealed leaves) from the - // garbage manifest. - let new_leaf_cids: Vec = fwd_pairs - .iter() - .map(|(_, cid)| cid.clone()) - .chain(rev_pairs.iter().map(|(_, cid)| cid.clone())) - .collect(); - - Ok(PersistedArenaResult { - new_index: Some(AnnotationIndexRoot { - version: 1, - max_t: out.max_t, - forward_branch_cid: fwd_branch_cid, - reverse_branch_cid: rev_branch_cid, - stats: out.stats, - }), - replaced_leaf_cids, - new_leaf_cids, - }) -} - -#[cfg(test)] -mod tests { - use super::*; - use fluree_db_core::storage::MemoryContentStore; - use fluree_db_core::FlakeValue; - use std::sync::Arc; - - fn ann(name: &str) -> Sid { - Sid::new(20, name) - } - fn refs(name: &str) -> Sid { - Sid::new(11, name) - } - fn id_dt() -> Sid { - fluree_db_core::id_datatype_sid() - } - fn edge(s: &str, p: &str, o: &str) -> EdgeKey { - EdgeKey { - g: None, - s: refs(s), - p: refs(p), - o: FlakeValue::Ref(refs(o)), - dt: id_dt(), - lang: None, - list_i: None, - } - } - - #[tokio::test] - async fn empty_inputs_skip_arena_seal() { - let store: Arc = Arc::new(MemoryContentStore::new()); - let result = build_and_persist_annotation_arena(&store, None, Vec::new()) - .await - .unwrap(); - assert!(result.new_index.is_none()); - assert!(result.replaced_leaf_cids.is_empty()); - } - - #[tokio::test] - async fn novelty_only_seals_new_arena() { - let store: Arc = Arc::new(MemoryContentStore::new()); - let events = vec![ - (edge("alice", "worksFor", "acme"), ann("ann_1"), 5, true), - (edge("alice", "worksFor", "acme"), ann("ann_2"), 6, true), - ]; - let result = build_and_persist_annotation_arena(&store, None, events) - .await - .unwrap(); - let new_index = result.new_index.expect("arena sealed"); - assert_eq!(new_index.max_t, 6); - assert_eq!(new_index.stats.forward_rows, 2); - assert_eq!(new_index.stats.distinct_edges, 1); - assert_eq!(new_index.stats.distinct_annotations, 2); - assert!( - result.replaced_leaf_cids.is_empty(), - "no previous arena → no replaced leaves" - ); - - // Roundtrip the new arena via a reader to confirm CAS writes - // landed correctly. - let reader = AnnotationArenaReader::new(&new_index, store.as_ref()); - let live = reader - .current_annotations_for(&edge("alice", "worksFor", "acme"), 100) - .await - .unwrap(); - let mut sids: Vec = live.into_iter().collect(); - sids.sort(); - assert_eq!(sids, vec![ann("ann_1"), ann("ann_2")]); - } - - #[tokio::test] - async fn empty_events_with_previous_arena_seals_empty_arena() { - // Under the complete-history contract, `Some(vec![])` - // explicitly asserts "no attachments live anywhere." The - // new arena is empty — the previous arena's content is - // NOT merged in. This is the correct shape for a ledger - // whose attachments were all retracted between passes. - let store: Arc = Arc::new(MemoryContentStore::new()); - let first = build_and_persist_annotation_arena( - &store, - None, - vec![(edge("alice", "worksFor", "acme"), ann("ann_1"), 5, true)], - ) - .await - .unwrap(); - let prev = first.new_index.unwrap(); - - let second = build_and_persist_annotation_arena(&store, Some(&prev), Vec::new()) - .await - .unwrap(); - let new_index = second.new_index.expect("empty-events still seals an arena"); - assert_eq!( - new_index.stats.forward_rows, 0, - "empty events → empty arena under complete-history contract" - ); - assert_eq!(new_index.stats.distinct_edges, 0); - - // Previous arena's leaves are still recorded for GC. - assert_eq!(second.replaced_leaf_cids.len(), 2); - } - - #[tokio::test] - async fn truncates_events_above_job_t() { - // The indexer clips `attachment_events` to `t <= job_t` before - // calling the orchestrator helper. We verify the helper itself - // does not over-shoot when it gets pre-clipped input: max_t - // matches the highest event in the input, never higher. - let store: Arc = Arc::new(MemoryContentStore::new()); - let result = build_and_persist_annotation_arena( - &store, - None, - vec![ - (edge("alice", "worksFor", "acme"), ann("ann_1"), 5, true), - (edge("alice", "worksFor", "acme"), ann("ann_2"), 7, true), - ], - ) - .await - .unwrap(); - let new_index = result.new_index.unwrap(); - assert_eq!( - new_index.max_t, 7, - "max_t must reflect input, not exceed it" - ); - } - - #[tokio::test] - async fn rebuilds_from_complete_history_with_previous_arena_only_for_gc() { - // Contract: `events` is the COMPLETE history. The previous - // arena participates only for GC reachability — its events - // are NOT merged in. - let store: Arc = Arc::new(MemoryContentStore::new()); - - // First seal: ann_1 attached at t=5. - let first = build_and_persist_annotation_arena( - &store, - None, - vec![(edge("alice", "worksFor", "acme"), ann("ann_1"), 5, true)], - ) - .await - .unwrap(); - let prev = first.new_index.expect("first seal produces an arena"); - - let reader = AnnotationArenaReader::new(&prev, store.as_ref()); - let prev_leaves: std::collections::HashSet<_> = - reader.all_leaf_cids().await.unwrap().into_iter().collect(); - assert_eq!(prev_leaves.len(), 2, "one forward + one reverse leaf"); - - // Second seal: complete history is the original assert PLUS - // a retract at t=8. Caller passes BOTH events, not just the - // delta. - let second = build_and_persist_annotation_arena( - &store, - Some(&prev), - vec![ - (edge("alice", "worksFor", "acme"), ann("ann_1"), 5, true), - (edge("alice", "worksFor", "acme"), ann("ann_1"), 8, false), - ], - ) - .await - .unwrap(); - let new_index = second.new_index.expect("second seal"); - assert_eq!(new_index.max_t, 8); - assert_eq!( - new_index.stats.forward_rows, 2, - "rebuild reflects the complete history, no double-count" - ); - assert_eq!( - new_index.stats.distinct_edges, 0, - "ann_1 retracted → not live" - ); - - // Previous arena's leaves recorded for GC. - let replaced: std::collections::HashSet<_> = - second.replaced_leaf_cids.iter().cloned().collect(); - assert_eq!(replaced, prev_leaves); - - // Live read: ann_1 not visible at t=100. - let reader = AnnotationArenaReader::new(&new_index, store.as_ref()); - let live = reader - .current_annotations_for(&edge("alice", "worksFor", "acme"), 100) - .await - .unwrap(); - assert!(live.is_empty()); - } - - #[tokio::test] - async fn augment_path_merges_and_dedupes() { - // The Augment path in Phase 3d concats prev events with - // caller events, then sorts + dedups by full tuple. Verify - // the dedup actually drops exact-tuple duplicates. - let mut combined: Vec<(EdgeKey, Sid, i64, bool)> = vec![ - (edge("alice", "worksFor", "acme"), ann("ann_1"), 5, true), - (edge("alice", "worksFor", "acme"), ann("ann_1"), 5, true), // duplicate - (edge("alice", "worksFor", "acme"), ann("ann_1"), 8, false), - ]; - combined.sort(); - combined.dedup(); - assert_eq!( - combined.len(), - 2, - "exact-tuple duplicates must collapse to one" - ); - } - - #[tokio::test] - async fn rebuild_does_not_double_count_when_caller_supplies_complete_history() { - // If a caller mistakenly passed only a delta, the previous - // arena's events would be missed but the rebuild stays - // self-consistent. Here we verify the no-merge contract - // explicitly: identical input → identical row count. - let store: Arc = Arc::new(MemoryContentStore::new()); - - let events = vec![ - (edge("alice", "worksFor", "acme"), ann("ann_1"), 5, true), - (edge("alice", "worksFor", "acme"), ann("ann_2"), 6, true), - ]; - - let first = build_and_persist_annotation_arena(&store, None, events.clone()) - .await - .unwrap(); - let prev = first.new_index.unwrap(); - assert_eq!(prev.stats.forward_rows, 2); - - // Re-seal with the same complete history. The prev arena is - // present but its events are NOT merged — total stays at 2. - let second = build_and_persist_annotation_arena(&store, Some(&prev), events) - .await - .unwrap(); - let new_index = second.new_index.unwrap(); - assert_eq!( - new_index.stats.forward_rows, 2, - "previous arena must not be merged into the rebuild" - ); - } - - /// Sanity-only: this orchestrator helper trusts what it's given. - /// The `Augment`-without-prev-arena-but-sticky=true gate lives in - /// `build/incremental.rs` (where `has_annotations` is in scope). - /// Verify that here we'd happily seal an incomplete arena from - /// partial events, which is precisely why the upper-layer gate - /// is needed. - #[tokio::test] - async fn function_itself_does_not_gate_partial_events() { - let store: Arc = Arc::new(MemoryContentStore::new()); - let result = build_and_persist_annotation_arena( - &store, - None, - vec![(edge("alice", "worksFor", "acme"), ann("ann_1"), 5, true)], - ) - .await - .unwrap(); - let new_index = result.new_index.unwrap(); - assert_eq!( - new_index.stats.forward_rows, 1, - "function happily seals partial input — caller is responsible \ - for ensuring `events` is complete or that the result is OK \ - to publish" - ); - } -} diff --git a/fluree-db-indexer/src/build/incremental.rs b/fluree-db-indexer/src/build/incremental.rs index 792bf881a0..3b36e142d7 100644 --- a/fluree-db-indexer/src/build/incremental.rs +++ b/fluree-db-indexer/src/build/incremental.rs @@ -3971,156 +3971,12 @@ pub async fn incremental_index( } } - // ---- Phase 3d: Annotation arena seal ---- - // - // Coverage envelope from the caller: - // - // Authoritative(events) → events are the complete history; - // rebuild from scratch. Previous arena - // participates only for GC. - // Augment(events) → partial coverage; merge with the - // previous arena's events, dedupe by - // (edge, ann, t, op), rebuild. Stays - // correct under reload/eviction - // scenarios where the running - // AttachmentNovelty doesn't cover - // pre-index history. - // Unknown / None → defensive drop when base has arena; - // no-op when base has none. - // - // Events are clipped to `t <= record.commit_t` so a concurrent - // commit's events can't leak into a root whose `index_t` is still - // `commit_t`. Keeps `AnnotationIndexRoot.max_t <= IndexRoot.index_t`. - let job_t = record.commit_t; - use crate::config::AttachmentEventCoverage; - let coverage = config.attachment_events.clone(); - let prev_arena = base_root.annotation_index.as_ref(); - - match coverage { - Some(AttachmentEventCoverage::Authoritative(mut events)) => { - events.retain(|(_, _, t, _)| *t <= job_t); - let result = crate::build::annotation_arena::build_and_persist_annotation_arena( - content_store.as_ref(), - prev_arena, - events, - ) - .await?; - if let Some(ref ann) = result.new_index { - debug_assert!( - ann.max_t <= job_t, - "AnnotationIndexRoot.max_t ({}) must not exceed IndexRoot.index_t ({})", - ann.max_t, - job_t - ); - } - root_builder.set_annotation_index( - result.new_index, - result.replaced_leaf_cids, - result.new_leaf_cids, - ); - } - Some(AttachmentEventCoverage::Augment(mut events)) => { - events.retain(|(_, _, t, _)| *t <= job_t); - // Pull the previous arena's events (if any) and merge - // with the caller's. Dedupe by full tuple so overlap is - // safe — e.g. a continuously-running ledger whose - // overlay still holds pre-index events that match the - // base arena. - // - // Coverage gate: `Augment` is only safe to seal when we - // can recover historical events from somewhere. With a - // base arena, that's the merge source. Without one, the - // base might still carry indexed `f:reifies*` facts (the - // sticky bit) — sealing from the partial events alone - // would publish an incomplete arena and hide history. - // Stay in scan-fallback in that case until either: - // - a future pass supplies `Authoritative(events)`, or - // - resolver-side event collection lets the indexer - // produce its own complete history. - if prev_arena.is_none() && base_root.has_annotations { - tracing::warn!( - ledger_id = %ledger_id, - "incremental indexer received Augment coverage but base \ - root has indexed f:reifies* facts without an arena \ - (has_annotations=true, annotation_index=None). Augment \ - events alone can't recover historical attachments; \ - leaving annotation_index=None so hydration uses scan. \ - A later pass with Authoritative coverage (or slice \ - 3h's resolver-side collection) will seal an \ - authoritative arena." - ); - root_builder.set_annotation_index(None, Vec::new(), Vec::new()); - } else { - let prev_events: Vec<_> = if let Some(prev) = prev_arena { - let reader = - fluree_db_binary_index::annotation_arena::AnnotationArenaReader::new( - prev, - content_store.as_ref(), - ); - reader - .collect_all_forward_events() - .await - .map_err(crate::error::IndexerError::Core)? - } else { - Vec::new() - }; - let mut combined: Vec<(fluree_db_core::EdgeKey, fluree_db_core::Sid, i64, bool)> = - Vec::with_capacity(prev_events.len() + events.len()); - combined.extend(prev_events); - combined.extend(events); - // Sort + dedup — `(edge, ann, t, op)` tuples already - // implement Ord. After this `combined` carries every - // distinct event observed across both sources. - combined.sort(); - combined.dedup(); - - let result = crate::build::annotation_arena::build_and_persist_annotation_arena( - content_store.as_ref(), - prev_arena, - combined, - ) - .await?; - if let Some(ref ann) = result.new_index { - debug_assert!( - ann.max_t <= job_t, - "AnnotationIndexRoot.max_t ({}) must not exceed IndexRoot.index_t ({})", - ann.max_t, - job_t - ); - } - root_builder.set_annotation_index( - result.new_index, - result.replaced_leaf_cids, - result.new_leaf_cids, - ); - } - } - Some(AttachmentEventCoverage::Unknown) | None => { - if let Some(prev) = prev_arena { - // Delta unknown but base has an arena — defensively - // drop it. Collect the previous leaf CIDs so GC can - // reclaim them. - let reader = fluree_db_binary_index::annotation_arena::AnnotationArenaReader::new( - prev, - content_store.as_ref(), - ); - let prev_leaf_cids = reader - .all_leaf_cids() - .await - .map_err(crate::error::IndexerError::Core)?; - tracing::warn!( - ledger_id = %ledger_id, - "incremental indexer received Unknown / None coverage but \ - base root carries an arena; dropping arena on new root \ - to avoid publishing stale-but-authoritative attachment \ - state. Hydration falls back to scan path until the next \ - reindex pass supplies events." - ); - root_builder.set_annotation_index(None, prev_leaf_cids, Vec::new()); - } - // Else: non-annotation ledger fast path, or already in - // scan-fallback state. Nothing to do. - } + // ---- Phase 3d: release a legacy annotation arena ---- + if let Some(ref arena) = base_root.legacy_annotation_arena { + root_builder.release_legacy_annotation_arena( + fluree_db_binary_index::legacy_annotation_arena_cids(content_store.as_ref(), arena) + .await?, + ); } // ---- Phase 4: Root assembly ---- diff --git a/fluree-db-indexer/src/build/mod.rs b/fluree-db-indexer/src/build/mod.rs index 73a71feb11..178a3cc852 100644 --- a/fluree-db-indexer/src/build/mod.rs +++ b/fluree-db-indexer/src/build/mod.rs @@ -11,7 +11,6 @@ //! - [`spatial`]: Spatial index building (S2 complex geometries) //! - [`types`]: Shared types (`UploadedIndexes`, `UploadedDicts`, etc.) -pub(crate) mod annotation_arena; pub(crate) mod commit_chain; pub(crate) mod dicts; pub(crate) mod fulltext; diff --git a/fluree-db-indexer/src/build/rebuild.rs b/fluree-db-indexer/src/build/rebuild.rs index ba422371c8..e2f3e4cf6a 100644 --- a/fluree-db-indexer/src/build/rebuild.rs +++ b/fluree-db-indexer/src/build/rebuild.rs @@ -1344,7 +1344,6 @@ where db_stats: Some(db_stats), db_schema, sketch_ref, - attachment_events: config.attachment_events.clone(), prev_index: prev_index.clone(), term_dict, }; diff --git a/fluree-db-indexer/src/build/root_assembly.rs b/fluree-db-indexer/src/build/root_assembly.rs index 00cc6a4009..b32998a914 100644 --- a/fluree-db-indexer/src/build/root_assembly.rs +++ b/fluree-db-indexer/src/build/root_assembly.rs @@ -132,9 +132,9 @@ pub(crate) async fn attach_garbage_manifest( /// Load and decode the previous index root. /// -/// Best-effort: any load/decode failure degrades to `None`. The annotation -/// arm then stays in scan-fallback, and the garbage manifest is omitted -/// rather than written empty, instead of failing the rebuild. +/// Best-effort: any load/decode failure degrades to `None`, and the garbage +/// manifest is omitted rather than written empty, instead of failing the +/// rebuild. async fn load_prev_root( content_store: &dyn ContentStore, prev_root_id: &ContentId, @@ -146,7 +146,7 @@ async fn load_prev_root( /// The CIDs `new_root` supersedes: everything the prior root reached that /// this one no longer does. /// -/// "Reachable" includes leaves behind named-graph and annotation branch +/// "Reachable" includes leaves behind named-graph and legacy arena branch /// manifests via `collect_root_cas_ids_expanded`. Diffing only the direct /// CAS refs (`all_cas_ids()`) would silently leak those leaves on every /// reindex. @@ -211,34 +211,12 @@ pub(crate) struct Fir6Inputs { pub db_schema: Option, /// CAS reference for the serialized HLL sketch blob. pub sketch_ref: Option, - /// Edge-annotation event coverage envelope (M2b slice 3g). - /// - /// Routed from `IndexerConfig.attachment_events` through - /// `rebuild.rs`. Same coverage semantics as the incremental - /// path, but without a base arena to merge against: - /// - /// - `Authoritative(events)` — caller guarantees full history; - /// seal the arena from this set. - /// - `Augment(events)` / `Unknown` / `None` — we can't prove - /// completeness without a base arena (the rebuild path - /// doesn't yet collect events from the resolver). Stay in - /// scan-fallback (`annotation_index = None`) until an - /// explicitly `Authoritative` source is supplied — i.e. a - /// future rebuild whose caller passes the full event set, - /// or an incremental pass that can prove its overlay - /// coverage is authoritative. Augment alone cannot reseal - /// from this state because the indexer has no way to - /// recover the missing history. - pub attachment_events: Option, /// The index version this root supersedes — the prior head root's CID and /// `index_t` (`NsRecord`'s `index_head_id` and `index_t`) — when one /// exists. /// - /// Serves two purposes. It becomes the published root's `prev_index` link, - /// which GC and drop walk to enumerate superseded artifacts. It also lets - /// the `Augment` arena arm recover the base arena's event history from the - /// prior root — without it a full rebuild under `Augment` coverage - /// silently drops a previously-sealed arena. + /// It becomes the published root's `prev_index` link, which GC and drop + /// walk to enumerate superseded artifacts. pub prev_index: Option, /// Triple-term dictionary interned by this build, if any reification /// links were synthesized. @@ -265,8 +243,7 @@ pub(crate) async fn encode_and_write_root_v6( inputs.index_t, )?; - // Loaded once and shared: the annotation arm reads its sealed arena and - // the garbage manifest diffs against its reachable set. + // The garbage manifest diffs against the prior root's reachable set. let prev_root = match inputs.prev_index.as_ref() { Some(prev) => load_prev_root(content_store, &prev.id).await, None => None, @@ -334,14 +311,8 @@ pub(crate) async fn encode_and_write_root_v6( garbage: None, sketch_ref: inputs.sketch_ref, has_annotations, - annotation_index: None, + legacy_annotation_arena: None, term_dict: inputs.term_dict, - // Sticky bit flipped to `true` below if the rebuild path - // seals an `Authoritative` arena. Rebuilds always start - // from scratch with no prior root, so this is the only - // signal carried forward — defensive-drop semantics live - // exclusively on the incremental path. - had_annotation_arena: false, has_list_meta: Some(inputs.saw_list_meta), }; @@ -351,164 +322,6 @@ pub(crate) async fn encode_and_write_root_v6( stats.distribute_total_size_by_flakes(root.total_commit_size); } - // ---- Annotation arena seal (M2b slice 3g, full-rebuild path) ---- - // - // Same coverage envelope as the incremental Phase 3d, but - // without a previous arena to merge against (full rebuild - // starts from scratch). Decision matrix: - // - // Authoritative(events) → caller asserts complete history; - // seal authoritative arena. - // Augment(events) → caller has events but can't prove - // completeness; without a base arena - // to merge with, we have no way to - // recover historical attachments - // beyond the supplied events. Stay - // in scan-fallback (annotation_index - // = None) until an explicitly - // `Authoritative` source is provided - // (a future rebuild whose caller - // passes the full event set, or - // resolver-side event collection - // that lets the indexer produce its - // own complete history). Augment - // alone cannot reseal from this - // state — the indexer cannot - // reconstruct the missing history. - // Unknown / None → no caller events; no-op. - // - // Events clipped to t <= inputs.index_t for the same reason as - // incremental: keeps `AnnotationIndexRoot.max_t <= - // IndexRoot.index_t`. - use crate::config::AttachmentEventCoverage; - let job_t = inputs.index_t; - match inputs.attachment_events { - Some(AttachmentEventCoverage::Authoritative(mut events)) => { - events.retain(|(_, _, t, _)| *t <= job_t); - let result = crate::build::annotation_arena::build_and_persist_annotation_arena( - content_store, - None, - events, - ) - .await?; - if let Some(ref ann) = result.new_index { - debug_assert!( - ann.max_t <= job_t, - "AnnotationIndexRoot.max_t ({}) must not exceed IndexRoot.index_t ({})", - ann.max_t, - job_t - ); - } - if result.new_index.is_some() { - // Sticky flag: an arena was sealed at this t. Even if - // a later pass defensively drops it, this bit stays - // true so the provider's bootstrap scan-fallback is - // suppressed. - root.had_annotation_arena = true; - } - root.annotation_index = result.new_index; - // No previous arena → no leaves to GC; replaced_leaf_cids - // is empty by construction. - debug_assert!(result.replaced_leaf_cids.is_empty()); - } - Some(AttachmentEventCoverage::Augment(events)) => { - // A full rebuild starts from scratch, but the *previous root* - // may carry a sealed arena whose event history plus the - // Augment delta is complete — the same merge contract the - // incremental Phase 3d applies. Without recovering it here, a - // full reindex under `Augment` coverage silently drops a - // previously-sealed arena (and the sticky bit then blocks the - // bootstrap scan from ever resealing). - let prev_arena = prev_root - .as_ref() - .and_then(|prev| prev.annotation_index.clone()); - match prev_arena { - Some(prev) => { - let reader = - fluree_db_binary_index::annotation_arena::AnnotationArenaReader::new( - &prev, - content_store, - ); - match reader.collect_all_forward_events().await { - Ok(mut merged) => { - merged.extend(events); - // Full-tuple sort + dedup: an event indexed by a - // prior pass may also still sit in the running - // overlay's delta. - merged.sort(); - merged.dedup(); - merged.retain(|(_, _, t, _)| *t <= job_t); - // `previous_index: None`: the prior arena's blobs - // stay reachable from the previous root and follow - // the same old-generation GC lifecycle as every - // other replaced artifact of a full rebuild. - // Identical re-sealed leaves re-derive the same - // CIDs, so nothing is duplicated. - let result = - crate::build::annotation_arena::build_and_persist_annotation_arena( - content_store, - None, - merged, - ) - .await?; - if let Some(ref ann) = result.new_index { - debug_assert!( - ann.max_t <= job_t, - "AnnotationIndexRoot.max_t ({}) must not exceed \ - IndexRoot.index_t ({})", - ann.max_t, - job_t - ); - } - if result.new_index.is_some() { - root.had_annotation_arena = true; - } - root.annotation_index = result.new_index; - tracing::debug!( - ledger_id = %inputs.ledger_id, - "full-rebuild resealed annotation arena from previous \ - root's arena + Augment delta" - ); - } - Err(e) => { - tracing::warn!( - ledger_id = %inputs.ledger_id, - error = %e, - "failed to read previous arena events for Augment \ - merge; leaving annotation_index=None (scan-fallback)" - ); - } - } - } - None => { - tracing::warn!( - ledger_id = %inputs.ledger_id, - "full-rebuild path received Augment coverage but has no \ - base arena to merge with; cannot prove history \ - completeness. Leaving annotation_index=None — the next \ - incremental pass with running overlay coverage will \ - seal an authoritative arena." - ); - } - } - } - Some(AttachmentEventCoverage::Unknown) | None => { - // Non-annotation ledger fast path or scan-fallback state. - } - } - - // Sticky-bit coercion: see the canonical contract on - // `IndexRoot.had_annotation_arena` in - // `fluree-db-binary-index/src/format/index_root.rs`. Mirrors - // `IncrementalRootBuilder::build()` — every indexer pass on - // an annotation-bearing ledger represents history the indexer - // owns; the provider must not later reconstruct a live-only - // `Authoritative` arena from such a root. Bulk import is the - // only path that leaves the bit false. - if root.has_annotations { - root.had_annotation_arena = true; - } - // GC and drop both enumerate superseded artifacts by walking the // prev-index chain, so a root published without this link orphans every // earlier version and the blobs only those versions reference. @@ -637,9 +450,8 @@ mod tests { garbage: None, sketch_ref: None, has_annotations: false, - had_annotation_arena: false, has_list_meta: None, - annotation_index: None, + legacy_annotation_arena: None, term_dict: None, o_type_table: IndexRoot::build_o_type_table(&[], &[]), ns_split_mode: fluree_db_core::ns_encoding::NsSplitMode::default(), diff --git a/fluree-db-indexer/src/config.rs b/fluree-db-indexer/src/config.rs index 56e44a73e0..3bc3043ef4 100644 --- a/fluree-db-indexer/src/config.rs +++ b/fluree-db-indexer/src/config.rs @@ -28,79 +28,6 @@ pub trait FulltextConfigProvider: std::fmt::Debug + Send + Sync { ) -> Vec; } -/// Coverage envelope for caller-supplied attachment events. -/// -/// Tells the indexer how to combine the supplied events with the base -/// root's existing arena (when present). The wrong coverage label -/// will either drop history (if `Authoritative` is asserted on a -/// post-reload tail) or double-count rows (if `Augment` is supplied -/// with already-indexed events that don't deduplicate cleanly), so -/// providers must be precise. -#[derive(Debug, Clone)] -pub enum AttachmentEventCoverage { - /// Caller asserts `events` is the **complete** history of every - /// `f:reifies*` event the snapshot has ever observed. The indexer - /// rebuilds the arena from scratch from this set; the previous - /// arena is enumerated only for GC reachability. - /// - /// Safe only when the caller has seen every commit since - /// genesis (or at least since the predicate dictionary first - /// observed an `f:reifies*` SID). A long-running ledger with a - /// continuously populated `AttachmentNovelty` qualifies; an - /// `AttachmentNovelty` reconstructed by `LedgerState::load` from - /// post-index commits **does not** — `load_novelty` only walks - /// commits with `t > snapshot.t`. - Authoritative(Vec<(fluree_db_core::EdgeKey, fluree_db_core::Sid, i64, bool)>), - - /// Caller has events but can't guarantee full-history coverage. - /// The indexer concatenates `events` with the previous arena's - /// events, deduplicates by `(edge, ann, t, op)` tuple, and - /// rebuilds. The dedup step makes this correct regardless of - /// whether the caller's events overlap the previous arena. - /// - /// Safe default for orchestrator-driven providers that read the - /// running `AttachmentNovelty.iter_event_pairs()` — that - /// overlay's coverage shifts between "full history" (for a - /// continuously running ledger) and "post-index tail" (after a - /// reload or eviction), and the merge-and-dedupe path stays - /// correct in both shapes. - Augment(Vec<(fluree_db_core::EdgeKey, fluree_db_core::Sid, i64, bool)>), - - /// Caller couldn't produce events this pass. Indexer treats as - /// delta-unknown — when the base root carries an arena, the new - /// root drops it (recording old leaves as replaced for GC) and - /// hydration falls back to scan until the next pass. - Unknown, -} - -/// Per-ledger attachment-events resolver for arena sealing on the -/// background-indexer path. -/// -/// Implementations (typically in the api layer) resolve a ledger ID -/// to a coverage envelope at job-dispatch time. The indexer takes -/// the resulting [`AttachmentEventCoverage`] and stamps it into a -/// per-job `IndexerConfig.attachment_events`. -/// -/// **Return semantics:** -/// - `Some(Authoritative(events))`: full history; indexer rebuilds -/// from scratch. -/// - `Some(Augment(events))`: partial coverage; indexer merges with -/// previous arena's events and dedupes. -/// - `Some(Unknown)` or `None`: defensive drop (treated identically). -/// -/// Implementations should be cheap on the happy path (one privileged -/// read of the running ledger state) and return -/// `Some(Unknown)` / `None` rather than panic when the ledger isn't -/// loaded — a missing ledger shouldn't block the indexing run; the -/// defensive drop covers correctness. -#[async_trait] -pub trait AttachmentEventsProvider: std::fmt::Debug + Send + Sync { - async fn attachment_events( - &self, - ledger_id: &fluree_db_core::LedgerId, - ) -> Option; -} - /// Scope of a configured full-text property entry. /// /// Mirrors the `f:targetGraph` sentinels used in config graph writes: @@ -348,43 +275,6 @@ pub struct IndexerConfig { /// `None` means no limit (backwards-compatible default). pub incremental_max_commit_bytes: Option, - /// Edge-annotation attachment events to seal into the new arena. - /// - /// `Some(coverage)` carries an [`AttachmentEventCoverage`] - /// envelope that tells the indexer how to combine the events - /// with the base root's existing arena: - /// - /// - `Authoritative(events)`: caller has the complete history; - /// the indexer rebuilds the arena from scratch from this set. - /// - `Augment(events)`: caller has partial events; the indexer - /// merges with the previous arena's events, dedupes by - /// `(edge, ann, t, op)`, and rebuilds. - /// - `Unknown`: caller couldn't produce events; defensive drop. - /// - /// `None` is treated identically to `Some(Unknown)`. - /// - /// Direct callers (CLI tools, tests, custom orchestrators with a - /// snapshot in hand) populate this field directly. The - /// `BackgroundIndexerWorker` path uses - /// [`Self::attachment_events_provider`] to resolve coverage - /// per-ledger at job-dispatch time. - pub attachment_events: Option, - - /// Per-ledger attachment-events resolver for orchestrator paths. - /// - /// `BackgroundIndexerWorker` holds a single `IndexerConfig` for - /// its lifetime, so `attachment_events` (a per-job value) can't - /// be set on the static config. The worker calls this provider - /// at job dispatch time to fetch the running ledger's - /// `AttachmentNovelty.iter_event_pairs()` and stamps the result - /// into a per-job clone of the config before invoking the - /// indexer. `None` (the default) yields the M2a behavior: - /// arenas are not sealed via the background path. - /// - /// See `attachment_events` for the delta-unknown semantics that - /// apply when this provider returns `None` for a given ledger. - pub attachment_events_provider: Option>, - /// Maximum number of *existing* subjects whose `rdf:type` set may change in /// a single incremental batch before incremental indexing aborts and defers /// to a full rebuild. Re-typing (or deleting) an existing subject forces a @@ -497,8 +387,6 @@ impl Default for IndexerConfig { incremental_retype_max_subjects: DEFAULT_INCREMENTAL_RETYPE_MAX_SUBJECTS, fulltext_configured_properties: Vec::new(), fulltext_config_provider: None, - attachment_events: None, - attachment_events_provider: None, pending_commit_cids: None, force_serial_commit_walk: false, warm_cache_source: None, @@ -549,8 +437,6 @@ impl IndexerConfig { incremental_retype_max_subjects: DEFAULT_INCREMENTAL_RETYPE_MAX_SUBJECTS, fulltext_configured_properties: Vec::new(), fulltext_config_provider: None, - attachment_events: None, - attachment_events_provider: None, pending_commit_cids: None, force_serial_commit_walk: false, warm_cache_source: None, @@ -581,8 +467,6 @@ impl IndexerConfig { incremental_retype_max_subjects: DEFAULT_INCREMENTAL_RETYPE_MAX_SUBJECTS, fulltext_configured_properties: Vec::new(), fulltext_config_provider: None, - attachment_events: None, - attachment_events_provider: None, pending_commit_cids: None, force_serial_commit_walk: false, warm_cache_source: None, @@ -613,8 +497,6 @@ impl IndexerConfig { incremental_retype_max_subjects: DEFAULT_INCREMENTAL_RETYPE_MAX_SUBJECTS, fulltext_configured_properties: Vec::new(), fulltext_config_provider: None, - attachment_events: None, - attachment_events_provider: None, pending_commit_cids: None, force_serial_commit_walk: false, warm_cache_source: None, @@ -636,18 +518,6 @@ impl IndexerConfig { self } - /// Attach a per-ledger attachment-events resolver. The - /// `BackgroundIndexerWorker` calls the provider at job dispatch - /// time; direct callers (CLI, tests) typically populate - /// `attachment_events` instead. - pub fn with_attachment_events_provider( - mut self, - provider: Arc, - ) -> Self { - self.attachment_events_provider = Some(provider); - self - } - /// Attach a late-bound resolver for the shared read cache to warm on write. /// Only co-located deployments plug this in; separate-machine indexers leave /// it unset (they can't reach the query server's cache). diff --git a/fluree-db-indexer/src/drop.rs b/fluree-db-indexer/src/drop.rs index e3f3adf0c6..8afb1f502a 100644 --- a/fluree-db-indexer/src/drop.rs +++ b/fluree-db-indexer/src/drop.rs @@ -193,16 +193,14 @@ mod tests { prev_index: Option, garbage: Option, ) -> Vec { - minimal_fir6_root(t, prev_index, garbage, None).encode() + minimal_fir6_root(t, prev_index, garbage).encode() } - /// Build a minimal IndexRoot in struct form so callers can attach - /// optional sections (e.g. annotation_index) before encoding. + /// Build a minimal IndexRoot in struct form. fn minimal_fir6_root( t: i64, prev_index: Option, garbage: Option, - annotation_index: Option, ) -> IndexRoot { let dummy_cid = ContentId::new(ContentKind::IndexLeaf, b"dummy"); let dummy_tree = DictTreeRefs { @@ -241,10 +239,9 @@ mod tests { prev_index, garbage, sketch_ref: None, - has_annotations: annotation_index.is_some(), - had_annotation_arena: annotation_index.is_some(), + has_annotations: false, has_list_meta: None, - annotation_index, + legacy_annotation_arena: None, term_dict: None, o_type_table: IndexRoot::build_o_type_table(&[], &[]), ns_split_mode: fluree_db_core::ns_encoding::NsSplitMode::default(), @@ -421,119 +418,6 @@ mod tests { assert!(cids.contains(&context_cid)); } - #[tokio::test] - async fn test_collect_includes_annotation_arena() { - // Verify the CID-walk fallback covers annotation arena branches - // AND the leaves they route to. On non-listable / permanent - // backends (IPFS) this is the only signal that lets the host - // unpin annotation blobs on hard drop. - use fluree_db_binary_index::annotation_arena::format::{ - AnnotationForwardBranch, AnnotationForwardBranchEntry, AnnotationForwardLeaf, - AnnotationReverseBranch, AnnotationReverseBranchEntry, AnnotationReverseLeaf, - }; - - let storage = MemoryStorage::new(); - let store = test_store(&storage); - - let fwd_leaf_bytes = AnnotationForwardLeaf::default().encode(); - let fwd_leaf_cid = store - .put(ContentKind::AnnotationForwardLeaf, &fwd_leaf_bytes) - .await - .unwrap(); - let rev_leaf_bytes = AnnotationReverseLeaf::default().encode(); - let rev_leaf_cid = store - .put(ContentKind::AnnotationReverseLeaf, &rev_leaf_bytes) - .await - .unwrap(); - - // Sample edge / Sid for the branch entries — content doesn't - // matter, the helper only walks branch → leaf links. - use fluree_db_core::{EdgeKey, FlakeValue, Sid}; - let sample = EdgeKey { - g: None, - s: Sid::new(1, "s"), - p: Sid::new(1, "p"), - o: FlakeValue::Ref(Sid::new(1, "o")), - dt: Sid::new(0, "http://www.w3.org/2001/XMLSchema#anyURI"), - lang: None, - list_i: None, - }; - let ann_sid = Sid::new(2, "a"); - - let fwd_branch = AnnotationForwardBranch { - leaves: vec![AnnotationForwardBranchEntry { - first_edge: sample.clone(), - first_ann: ann_sid.clone(), - last_edge: sample.clone(), - last_ann: ann_sid.clone(), - row_count: 0, - leaf_cid: fwd_leaf_cid.clone(), - }], - }; - let fwd_branch_cid = store - .put(ContentKind::AnnotationForwardBranch, &fwd_branch.encode()) - .await - .unwrap(); - let rev_branch = AnnotationReverseBranch { - leaves: vec![AnnotationReverseBranchEntry { - first_ann: ann_sid.clone(), - first_edge: sample.clone(), - last_ann: ann_sid, - last_edge: sample, - row_count: 0, - leaf_cid: rev_leaf_cid.clone(), - }], - }; - let rev_branch_cid = store - .put(ContentKind::AnnotationReverseBranch, &rev_branch.encode()) - .await - .unwrap(); - - let root = minimal_fir6_root( - 1, - None, - None, - Some(fluree_db_core::AnnotationIndexRoot { - version: 1, - max_t: 0, - forward_branch_cid: fwd_branch_cid.clone(), - reverse_branch_cid: rev_branch_cid.clone(), - stats: fluree_db_core::AnnotationStats::default(), - }), - ); - let root_bytes = root.encode(); - let root_cid = ContentId::new(ContentKind::IndexRoot, &root_bytes); - let root_addr = fluree_db_core::content_address( - "memory", - ContentKind::IndexRoot, - LEDGER, - &root_cid.digest_hex(), - ); - storage.write_bytes(&root_addr, &root_bytes).await.unwrap(); - - let cids = collect_ledger_cids(&store, None, Some(&root_cid), None, None) - .await - .unwrap(); - - assert!(cids.contains(&root_cid), "missing root CID"); - assert!( - cids.contains(&fwd_branch_cid), - "missing annotation forward branch CID" - ); - assert!( - cids.contains(&rev_branch_cid), - "missing annotation reverse branch CID" - ); - assert!( - cids.contains(&fwd_leaf_cid), - "missing annotation forward leaf CID — branch was not expanded" - ); - assert!( - cids.contains(&rev_leaf_cid), - "missing annotation reverse leaf CID — branch was not expanded" - ); - } - #[tokio::test] async fn test_collect_deduplicates() { let storage = MemoryStorage::new(); diff --git a/fluree-db-indexer/src/gc/collector.rs b/fluree-db-indexer/src/gc/collector.rs index 2db3d67ae2..eaf5b2803b 100644 --- a/fluree-db-indexer/src/gc/collector.rs +++ b/fluree-db-indexer/src/gc/collector.rs @@ -100,8 +100,8 @@ struct NodePartition { /// those earlier states, reviving a CID an already-consumed manifest named. /// Checked directly against `all_cas_ids()` rather than expanded, since a /// resurrected CID this pass must not delete is exactly one still directly -/// reachable from a surviving root — nothing behind a named-graph or -/// annotation branch manifest changes that. +/// reachable from a surviving root — nothing behind a branch manifest +/// changes that. fn retained_refs( index_chain: &[IndexChainEntry], first_released: usize, @@ -2335,7 +2335,6 @@ mod tests { db_stats: None, db_schema: None, sketch_ref: None, - attachment_events: None, prev_index, term_dict: None, } diff --git a/fluree-db-indexer/src/gc/test_support.rs b/fluree-db-indexer/src/gc/test_support.rs index 62eee1a0eb..5883198dd0 100644 --- a/fluree-db-indexer/src/gc/test_support.rs +++ b/fluree-db-indexer/src/gc/test_support.rs @@ -84,9 +84,8 @@ pub(crate) fn fir6_with_named_graph_for( garbage, sketch_ref: None, has_annotations: false, - annotation_index: None, + legacy_annotation_arena: None, term_dict: None, - had_annotation_arena: false, has_list_meta: None, o_type_table: IndexRoot::build_o_type_table(&[], &[]), ns_split_mode: fluree_db_core::ns_encoding::NsSplitMode::default(), diff --git a/fluree-db-indexer/src/lib.rs b/fluree-db-indexer/src/lib.rs index edc2f9179b..4deb64ffa4 100644 --- a/fluree-db-indexer/src/lib.rs +++ b/fluree-db-indexer/src/lib.rs @@ -47,9 +47,8 @@ pub mod stats; // Re-export main types pub use config::{ - AttachmentEventCoverage, AttachmentEventsProvider, ConfiguredFulltextProperty, - ConfiguredFulltextScope, FulltextConfigProvider, IndexerConfig, WarmCacheSource, - DEFAULT_CATCHUP_INTERVAL_SECS, + ConfiguredFulltextProperty, ConfiguredFulltextScope, FulltextConfigProvider, IndexerConfig, + WarmCacheSource, DEFAULT_CATCHUP_INTERVAL_SECS, }; pub use drop::collect_ledger_cids; pub use error::{IndexerError, Result}; diff --git a/fluree-db-indexer/src/orchestrator.rs b/fluree-db-indexer/src/orchestrator.rs index ed229183e6..1d3a7d4332 100644 --- a/fluree-db-indexer/src/orchestrator.rs +++ b/fluree-db-indexer/src/orchestrator.rs @@ -2190,15 +2190,7 @@ impl BackgroundIndexerWorker { return; } }; - // Per-job config: clone the worker's static config and stamp - // in the running ledger's attachment events when a provider - // is attached. The provider returning `None` (e.g. ledger not - // loaded into the running registry) becomes "delta unknown" - // in the indexer — see `IndexerConfig.attachment_events`. let mut job_config = self.config.clone(); - if let Some(provider) = self.config.attachment_events_provider.as_ref() { - job_config.attachment_events = provider.attachment_events(ledger_id).await; - } // Always create a fuel-enabled, no-limit tracker per build so each // background indexing pass is measured. Indexing never enforces a // limit — measurement only. The tally is logged on success and diff --git a/fluree-db-indexer/src/run_index/build/incremental_root.rs b/fluree-db-indexer/src/run_index/build/incremental_root.rs index 0664a7e681..2aab548fc4 100644 --- a/fluree-db-indexer/src/run_index/build/incremental_root.rs +++ b/fluree-db-indexer/src/run_index/build/incremental_root.rs @@ -218,72 +218,11 @@ impl IncrementalRootBuilder { self.root.sketch_ref = cid; } - /// Replace the on-disk annotation arena pointer. - /// - /// Pass `Some(_)` after building the new arena and writing its - /// branch + leaf blobs to CAS. The encoder enforces the truth- - /// table invariant that any populated `annotation_index` implies - /// `has_annotations = true` on the wire (see - /// `fluree_db_core::annotation_index`), so callers don't need to - /// flip the sticky bit separately. - /// - /// `previous_leaf_cids` must enumerate **every** leaf CID - /// referenced by the arena currently in `root.annotation_index`. - /// `new_leaf_cids` must enumerate every leaf CID referenced by - /// `new_index` (empty when `new_index` is `None`). - /// `ContentStore::release` deletes exact CIDs (not child graphs), - /// so without these lists the old leaves leak when the new root - /// supersedes the chain. The orchestrator computes these sets from - /// [`PersistedArenaResult`](crate::build::annotation_arena::PersistedArenaResult); - /// pass empty `Vec`s when there's no previous arena. Old branch - /// CIDs are reconciled automatically from `root.annotation_index`. - /// - /// Old CIDs (branches + leaves) that the **new** arena still - /// references are NOT recorded as garbage. Content-addressed - /// storage means a re-sealed unchanged arena (e.g. the `Augment` - /// path on a continuously-running ledger whose overlay still holds - /// pre-index events matching the base arena) produces identical - /// CIDs; recording them would let GC delete leaves/branches the - /// new live root still points at. Mirrors `set_dict_refs`. - pub fn set_annotation_index( - &mut self, - new_index: Option, - previous_leaf_cids: Vec, - new_leaf_cids: Vec, - ) { - // CIDs the new arena references (branches + leaves). - let mut new_cids: HashSet = HashSet::new(); - if let Some(new) = new_index.as_ref() { - new_cids.insert(new.forward_branch_cid.clone()); - new_cids.insert(new.reverse_branch_cid.clone()); - } - new_cids.extend(new_leaf_cids); - - // CIDs the old arena referenced (branches + leaves). - let mut old_cids: HashSet = HashSet::new(); - if let Some(prev) = self.root.annotation_index.as_ref() { - old_cids.insert(prev.forward_branch_cid.clone()); - old_cids.insert(prev.reverse_branch_cid.clone()); - } - old_cids.extend(previous_leaf_cids); - - // Only old CIDs the new arena no longer references are garbage. - let mut replaced: Vec = old_cids.difference(&new_cids).cloned().collect(); - // Keep ordering deterministic for garbage manifest stability. - replaced.sort_by_key(std::string::ToString::to_string); - self.replaced_cids.extend(replaced); - // Sticky bit: flip `had_annotation_arena` to `true` the - // moment any arena is sealed, and *never* clear it on - // subsequent calls — including when this call sets - // `new_index = None` (defensive drop). Without this, the - // post-drop root would look identical to a fresh import - // and the provider's bootstrap base-index scan-fallback - // could resurrect a live-only `Authoritative` arena, - // losing historical retract/reassert rows. - if new_index.is_some() { - self.root.had_annotation_arena = true; - } - self.root.annotation_index = new_index; + /// Drop the base root's legacy annotation arena, which the new root does + /// not carry, recording its blobs as garbage. + pub fn release_legacy_annotation_arena(&mut self, arena_cids: Vec) { + self.root.legacy_annotation_arena = None; + self.replaced_cids.extend(arena_cids); } /// Fold this pass's list-row observation into the sticky @@ -310,19 +249,7 @@ impl IncrementalRootBuilder { } /// Consume the builder, returning the final root and all replaced CIDs. - pub fn build(mut self) -> (IndexRoot, Vec) { - // Sticky-bit coercion: see the canonical contract on - // `IndexRoot.had_annotation_arena` in - // `fluree-db-binary-index/src/format/index_root.rs`. Every - // indexer-produced root with `has_annotations = true` - // sets the bit so the provider's base-index scan-fallback - // can't later resurrect a live-only `Authoritative` arena - // from a defensive-drop or no-seal pass. Bulk import is - // the only path that leaves the bit false (it bypasses - // this builder entirely; see `fluree-db-api/src/import.rs`). - if self.root.has_annotations { - self.root.had_annotation_arena = true; - } + pub fn build(self) -> (IndexRoot, Vec) { (self.root, self.replaced_cids) } } @@ -353,10 +280,7 @@ fn collect_dict_cids(refs: &DictRefs) -> Vec { mod tests { use super::*; use fluree_db_binary_index::{DictPackRefs, DictTreeRefs}; - use fluree_db_core::{ - ns_encoding::NsSplitMode, AnnotationIndexRoot, AnnotationStats, ContentKind, - SubjectIdEncoding, - }; + use fluree_db_core::{ns_encoding::NsSplitMode, ContentKind, SubjectIdEncoding}; fn cid(label: &[u8]) -> ContentId { ContentId::new(ContentKind::IndexLeaf, label) @@ -400,51 +324,27 @@ mod tests { garbage: None, sketch_ref: None, has_annotations: false, - annotation_index: None, + legacy_annotation_arena: None, term_dict: None, - had_annotation_arena: false, has_list_meta: None, o_type_table: IndexRoot::build_o_type_table(&[], &[]), ns_split_mode: NsSplitMode::default(), } } - fn arena(fwd: ContentId, rev: ContentId) -> AnnotationIndexRoot { - AnnotationIndexRoot { - version: 1, - max_t: 5, - forward_branch_cid: fwd, - reverse_branch_cid: rev, - stats: AnnotationStats::default(), - } - } - #[test] - fn set_annotation_index_keeps_unchanged_reseal_cids_out_of_garbage() { - // STOR-1: re-sealing an unchanged arena produces identical - // content-addressed CIDs. The live CIDs must NOT enter the - // garbage manifest, or GC deletes data the new root references. - let fwd = cid(b"fwd-branch"); - let rev = cid(b"rev-branch"); - let leaf_a = cid(b"leaf-a"); - let leaf_b = cid(b"leaf-b"); - + fn releasing_a_legacy_arena_drops_it_and_records_its_blobs() { let mut root = minimal_root(); - root.annotation_index = Some(arena(fwd.clone(), rev.clone())); + root.legacy_annotation_arena = Some(fluree_db_binary_index::LegacyAnnotationArena { + forward_branch_cid: cid(b"fwd-branch"), + reverse_branch_cid: cid(b"rev-branch"), + }); let mut b = IncrementalRootBuilder::from_old_root(root, "test"); - // Same branches + same leaves (byte-identical re-seal). - b.set_annotation_index( - Some(arena(fwd.clone(), rev.clone())), - vec![leaf_a.clone(), leaf_b.clone()], - vec![leaf_a.clone(), leaf_b.clone()], - ); - let (_root, garbage) = b.build(); - for c in [&fwd, &rev, &leaf_a, &leaf_b] { - assert!( - !garbage.contains(c), - "unchanged re-seal must not GC live arena CID {c}" - ); - } + let blobs = vec![cid(b"fwd-branch"), cid(b"rev-branch"), cid(b"leaf")]; + b.release_legacy_annotation_arena(blobs.clone()); + let (root, garbage) = b.build(); + assert!(root.legacy_annotation_arena.is_none()); + assert_eq!(garbage, blobs); } #[test] @@ -456,31 +356,4 @@ mod tests { let (root, _) = IncrementalRootBuilder::from_old_root(root, "db:dev").build(); assert_eq!(root.ledger_id, "db:dev"); } - - #[test] - fn set_annotation_index_retires_changed_arena_cids() { - // Control: when the arena genuinely changes, the old now-unused - // CIDs ARE retired to garbage, but CIDs the new arena still - // references (the shared reverse branch) are kept. - let old_fwd = cid(b"fwd-old"); - let rev = cid(b"rev-shared"); - let old_leaf = cid(b"leaf-old"); - - let mut root = minimal_root(); - root.annotation_index = Some(arena(old_fwd.clone(), rev.clone())); - let mut b = IncrementalRootBuilder::from_old_root(root, "test"); - let new_fwd = cid(b"fwd-new"); - b.set_annotation_index( - Some(arena(new_fwd.clone(), rev.clone())), - vec![old_leaf.clone()], - Vec::new(), // new arena references no leaves - ); - let (_root, garbage) = b.build(); - assert!(garbage.contains(&old_fwd), "changed forward branch retired"); - assert!(garbage.contains(&old_leaf), "unused old leaf retired"); - assert!( - !garbage.contains(&rev), - "reverse branch still referenced by new arena must be kept" - ); - } } diff --git a/fluree-db-ledger/src/lib.rs b/fluree-db-ledger/src/lib.rs index 455b8ca702..2a4b97e2d4 100644 --- a/fluree-db-ledger/src/lib.rs +++ b/fluree-db-ledger/src/lib.rs @@ -1208,11 +1208,11 @@ mod tests { /// Used by `test_apply_index_equal_t_noop` to produce two FIR6 /// blobs with the same `index_t` but different CIDs without /// resorting to trailing-byte padding (which the strict - /// `FIR6: trailing bytes after annotation_index` check rejects). + /// `FIR6: trailing bytes` check rejects). fn build_test_fir6_with_base(ledger_id: &str, index_t: i64, base_t: i64) -> Vec { // Mirror `fluree-db-core::db::decode_fir6_metadata` exactly. // The decoder now enforces a strict trailing-byte check - // (`FIR6: trailing bytes after annotation_index`), so any + // (`FIR6: trailing bytes`), so any // section that doesn't exactly match what the decoder reads // will surface either as truncation or trailing-byte error. // Keep this skeleton in lockstep with the decoder. @@ -1274,7 +1274,7 @@ mod tests { // named graph routing buf.extend_from_slice(&0u16.to_le_bytes()); // named_count = 0 - // No optional sections (flags=0); no annotation_index section. + // No optional sections (flags=0). // The strict trailing-byte check requires exactly zero bytes // beyond this point. diff --git a/fluree-db-novelty/src/attachments.rs b/fluree-db-novelty/src/attachments.rs deleted file mode 100644 index 8147e0fe90..0000000000 --- a/fluree-db-novelty/src/attachments.rs +++ /dev/null @@ -1,882 +0,0 @@ -//! Edge-annotation attachment overlay (M1 — novelty only). -//! -//! Mirrors flake-level [`Novelty`](crate::Novelty) for the -//! attachment side: an in-memory, derived index of which annotations -//! are attached to which base edges. Populated by observing -//! `f:reifies*` system flakes that flow through the novelty pipeline, -//! either at apply-commit time (live transactions) or warmup -//! (rehydrating a snapshot from prior commits). -//! -//! `AttachmentNovelty` is **derived state** — never primary truth. The -//! durable encoding lives in the seven `f:reifies*` flakes themselves -//! (see `fluree_db_core::edge::EdgeKey::to_reifies_facts`). If the -//! attachment overlay disagrees with the underlying flakes, the -//! flakes win. M2 will replace this in-memory map with a binary -//! arena, keeping the durable encoding unchanged. -//! -//! Two indexes are maintained in parallel: -//! -//! - `forward: EdgeKey -> Vec` — for edge-rooted lookups. -//! "Given this base edge, which annotations point at it?" -//! - `reverse: Sid -> Vec` — for annotation-rooted -//! lookups. "Given this annotation subject, which edge does it -//! reify?" -//! -//! Each row carries `(t, op)` so history queries can replay the -//! attachment lifecycle without re-decoding the flakes. - -use fluree_db_core::edge::EdgeKey; -use fluree_db_core::namespaces::is_reserved_reifies_predicate; -use fluree_db_core::{Flake, Sid}; -use std::collections::BTreeMap; - -use crate::error::Result; - -/// One forward-direction row: an annotation attached to an edge. -/// -/// Stored under [`AttachmentNovelty::forward`] keyed by the edge. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct ForwardRow { - /// The annotation subject SID. - pub ann: Sid, - /// Transaction time of this attachment event. - pub t: i64, - /// `true` = attachment asserted, `false` = retracted. - pub op: bool, -} - -/// One reverse-direction row: an edge an annotation reifies. -/// -/// Stored under [`AttachmentNovelty::reverse`] keyed by the annotation -/// SID. Carries the full `EdgeKey` (graph + s + p + o + dt + lang) so -/// downstream operators can re-probe the base fact indexes for -/// visibility checks without an additional lookup. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct ReverseRow { - /// The reified base edge. - pub edge: EdgeKey, - /// Transaction time of this attachment event. - pub t: i64, - /// `true` = attachment asserted, `false` = retracted. - pub op: bool, -} - -/// In-memory attachment overlay paralleling [`Novelty`](crate::Novelty). -/// -/// Updated by [`Self::observe_flakes`] from the same flake stream that -/// `Novelty::apply_commit` accepts (post-dedup). Caches the -/// `has_annotations` gate so cascade-retract paths can short-circuit -/// without walking the maps when the ledger has never seen an -/// annotation. -#[derive(Clone, Debug, Default)] -pub struct AttachmentNovelty { - forward: BTreeMap>, - reverse: BTreeMap>, - /// `true` once *any* `f:reifies*` bundle has been observed (asserted - /// or retracted, doesn't matter — the cascade gate cares about - /// "could this snapshot ever have annotations"). - has_annotations: bool, - /// Cumulative count of malformed `f:reifies*` bundles observed by - /// [`Self::observe_flakes`] over the lifetime of this overlay. - /// Per the design contract, malformed bundles are skipped + warned - /// rather than erroring out so a single corrupt event in replay - /// can't block the rest of the ledger from loading. Operators - /// scrape this to detect data-corruption / replay-anomaly - /// signals; a non-zero value indicates either a software bug in - /// the writer or an externally-tampered commit history. - observed_malformed_bundles: u64, -} - -impl AttachmentNovelty { - /// Create an empty overlay. - pub fn new() -> Self { - Self::default() - } - - /// True iff at least one `f:reifies*` flake has been observed. - /// - /// Cascade fast-path: when both `Novelty::attachments.has_annotations()` - /// and the indexed arena both report `false`, plain edge retracts - /// can skip the attachment-cascade lookup entirely. - #[inline] - pub fn has_annotations(&self) -> bool { - self.has_annotations - } - - /// Cumulative count of malformed `f:reifies*` bundles observed - /// over this overlay's lifetime. See the field docs on - /// `observed_malformed_bundles` for the operational signal. - /// Always `0` on a healthy ledger. - #[inline] - pub fn observed_malformed_bundle_count(&self) -> u64 { - self.observed_malformed_bundles - } - - /// Iterator over annotation SIDs **currently attached** to `edge`, - /// where "current" is evaluated against `as_of_t`: only events - /// with `t <= as_of_t` are considered, and the latest such event - /// for each annotation must be `op == true`. - /// - /// This is the time-travel-correct read used by query and - /// hydration paths — the formatter passes `self.db.t` so a - /// historical view sees the attachment state as of that - /// transaction, not the live latest. - pub fn current_annotations_for_at<'a>( - &'a self, - edge: &'a EdgeKey, - as_of_t: i64, - ) -> impl Iterator + 'a { - self.forward - .get(edge) - .map(|rows| latest_assertions_at::<_, _>(rows.iter(), |r| (&r.ann, r.t, r.op), as_of_t)) - .into_iter() - .flatten() - .cloned() - } - - /// Iterator over annotation SIDs currently attached to `edge` - /// against the **live** overlay state (i.e., the latest event - /// over all `t`). - /// - /// Used by transactor staging where the relevant state is - /// always "everything committed before this transaction" — the - /// novelty's attachment rows only carry post-commit events with - /// `t <= ledger.t()` by construction, so the live and as-of - /// reads coincide. - /// - /// **Read paths must use [`Self::current_annotations_for_at`] - /// with an explicit `as_of_t`** — this method is for write-side - /// callers only. - pub fn current_annotations_for<'a>( - &'a self, - edge: &'a EdgeKey, - ) -> impl Iterator + 'a { - self.current_annotations_for_at(edge, i64::MAX) - } - - /// Time-travel-correct counterpart of [`Self::current_targets_for`]. - pub fn current_targets_for_at<'a>( - &'a self, - ann: &'a Sid, - as_of_t: i64, - ) -> impl Iterator + 'a { - self.reverse - .get(ann) - .map(|rows| { - latest_assertions_at::<_, _>(rows.iter(), |r| (&r.edge, r.t, r.op), as_of_t) - }) - .into_iter() - .flatten() - .cloned() - } - - /// Iterator over base [`EdgeKey`]s currently reified by - /// annotation `ann` against the live overlay state. See - /// [`Self::current_annotations_for`] for the time-travel-vs- - /// live distinction. - pub fn current_targets_for<'a>(&'a self, ann: &'a Sid) -> impl Iterator + 'a { - self.current_targets_for_at(ann, i64::MAX) - } - - /// Iterator over the *full* attachment history of `ann` — - /// every `(EdgeKey, t, op)` event, in row-stored order. Used by - /// history-range queries that explicitly want to see attachment - /// lifecycle alongside flake history. - pub fn target_history<'a>(&'a self, ann: &'a Sid) -> impl Iterator + 'a { - self.reverse - .get(ann) - .into_iter() - .flat_map(|rows| rows.iter()) - } - - /// Iterator over the full attachment history of `edge` — every - /// `(ann, t, op)` event for this edge, in row-stored order. - pub fn forward_history<'a>( - &'a self, - edge: &'a EdgeKey, - ) -> impl Iterator + 'a { - self.forward - .get(edge) - .into_iter() - .flat_map(|rows| rows.iter()) - } - - /// Collect every overlay event for `edge` as `(ann, t, op)` triples. - /// - /// Shaped to match the input format of - /// `fluree_db_binary_index::annotation_arena::AnnotationArenaReader::current_annotations_merged` - /// so callers in higher-layer crates can pass the slice directly. - /// Empty when the overlay has no rows for the given edge. - pub fn collect_forward_events(&self, edge: &EdgeKey) -> Vec<(Sid, i64, bool)> { - self.forward - .get(edge) - .map(|rows| rows.iter().map(|r| (r.ann.clone(), r.t, r.op)).collect()) - .unwrap_or_default() - } - - /// Collect every overlay event for `ann` as `(edge, t, op)` triples. - /// Counterpart of [`Self::collect_forward_events`] for the reverse - /// arena. - pub fn collect_reverse_events(&self, ann: &Sid) -> Vec<(EdgeKey, i64, bool)> { - self.reverse - .get(ann) - .map(|rows| rows.iter().map(|r| (r.edge.clone(), r.t, r.op)).collect()) - .unwrap_or_default() - } - - /// Iterator over every overlay event as `(EdgeKey, ann, t, op)` - /// tuples — the input shape of - /// `fluree_db_binary_index::annotation_arena::build_arenas_from_event_pairs`. - /// - /// Used by the indexer when sealing a new arena: collect the full - /// overlay state (or merge with the previous arena's events first) - /// and feed straight into the arena builder. Walks `forward` so - /// each `(edge, ann)` pair is yielded together with all its - /// history rows in row-stored order. - pub fn iter_event_pairs(&self) -> impl Iterator + '_ { - self.forward.iter().flat_map(|(edge, rows)| { - rows.iter() - .map(move |r| (edge.clone(), r.ann.clone(), r.t, r.op)) - }) - } - - /// Observe a slice of accepted flakes and update the overlay. - /// - /// Filters down to `f:reifies*` flakes, groups them by - /// `(ann_sid, t, op)`, and decodes each group via - /// [`EdgeKey::from_reifies_facts`]. A malformed bundle (missing - /// required predicate, duplicate, or deferred shape) is **skipped - /// with a `tracing::warn!` and counted** in - /// `observed_malformed_bundles`. The non-`f:reifies*` flakes for - /// the same annotation subject remain visible as ordinary RDF — - /// only the attachment binding is dropped. - /// - /// **Why skip rather than error:** the replay-validation contract - /// in `docs/design/edge-annotations.md` says "Reject (skip + - /// telemetry counter) any ann_sid that has a partial bundle." - /// A single malformed bundle in commit replay - /// (e.g. legacy data or a tampered commit) would otherwise block - /// the ledger from loading. Operators detect data corruption - /// post-hoc via [`Self::observed_malformed_bundle_count`]. - /// - /// Caller contract: pass the **post-dedup** flake set that - /// `Novelty::apply_commit` ultimately stored in the arena. - /// Observing a deduped duplicate would create a phantom row. - pub fn observe_flakes(&mut self, flakes: &[Flake]) -> Result<()> { - if flakes.is_empty() { - return Ok(()); - } - - // Group `f:reifies*` flakes into bundles keyed by - // `(flake_graph, ann_sid, t, op)`. The flake-level graph is - // part of the key because the writer convention is that - // `f:reifies*` flakes for an edge in graph G are themselves - // asserted in graph G; folding two graphs together at this - // stage would let a pathological cross-graph collision merge - // into a single (wrong-graph) bundle. Mirrors the arena - // builder's grouping in - // `fluree_db_binary_index::annotation_arena::bundle::build_arenas_from_flakes` - // so the two paths agree on what counts as a malformed - // bundle. - let mut bundles: BTreeMap<(Option, Sid, i64, bool), Vec> = BTreeMap::new(); - for f in flakes { - if !is_reserved_reifies_predicate(&f.p) { - continue; - } - bundles - .entry((f.g.clone(), f.s.clone(), f.t, f.op)) - .or_default() - .push(f.clone()); - } - - if bundles.is_empty() { - return Ok(()); - } - - for ((bundle_g, ann, t, op), bundle) in bundles { - let edge = match EdgeKey::from_reifies_facts(&bundle) { - Ok(edge) => edge, - Err(e) => { - self.observed_malformed_bundles = - self.observed_malformed_bundles.saturating_add(1); - tracing::warn!( - ?ann, - t, - op, - error = %e, - cumulative_skipped = self.observed_malformed_bundles, - "skipping malformed f:reifies* bundle in novelty observer" - ); - continue; - } - }; - - // Cross-check: the graph the bundle was *asserted in* - // (flake-level `g`) must match the graph the bundle - // *reifies* (`EdgeKey.g`, derived from the optional - // `f:reifiesGraph` flake). Mismatches indicate a - // malformed bundle (e.g. `f:reifiesGraph` missing on a - // named-graph edge, or a tampered commit) — file under - // the wrong graph and the cascade fast-path can't find - // the bundle on a base-edge retract. - // - // **Belt-and-suspenders.** The invariant now lives in - // the decoder itself: `EdgeKey::from_reifies_facts` - // returns `EdgeKeyDecodeError::GraphMismatch` when - // `f:reifiesGraph` disagrees with the bundle's - // flake-level `g`, and `MixedFlakeGraphs` when the - // bundle's flakes don't share one graph. Since this - // observer groups by `(g, ann, t, op)` before calling - // the decoder, the inner slice is graph-uniform and - // the decoder's `GraphMismatch` check already covers - // every case this external check could fire on. The - // external check is kept as defense-in-depth: if a - // future refactor changes the decoder semantics or - // skips the reconciliation, this branch catches the - // regression with a structured warn that names both - // graphs. - if edge.g != bundle_g { - self.observed_malformed_bundles = self.observed_malformed_bundles.saturating_add(1); - tracing::warn!( - ?ann, - t, - op, - bundle_graph = ?bundle_g, - edge_graph = ?edge.g, - cumulative_skipped = self.observed_malformed_bundles, - "skipping bundle: f:reifiesGraph disagrees with flake-level graph" - ); - continue; - } - - self.forward - .entry(edge.clone()) - .or_default() - .push(ForwardRow { - ann: ann.clone(), - t, - op, - }); - self.reverse - .entry(ann) - .or_default() - .push(ReverseRow { edge, t, op }); - self.has_annotations = true; - } - - Ok(()) - } - - /// Iterator over every `(edge, rows)` pair in the forward map. - /// Diagnostic / test use — walks the entire overlay so callers - /// must keep the cost in mind (linear in distinct edges). - pub fn iter_forward(&self) -> impl Iterator)> { - self.forward.iter() - } - - /// Iterator over every `(ann_sid, rows)` pair in the reverse map. - /// Diagnostic / test counterpart of [`Self::iter_forward`]. - pub fn iter_reverse(&self) -> impl Iterator)> { - self.reverse.iter() - } - - /// Total number of forward rows across all edges. Diagnostic / - /// telemetry-only — not a hot-path metric. - pub fn forward_row_count(&self) -> usize { - self.forward.values().map(Vec::len).sum() - } - - /// Total number of reverse rows across all annotations. Diagnostic - /// / telemetry-only. - pub fn reverse_row_count(&self) -> usize { - self.reverse.values().map(Vec::len).sum() - } - - /// Number of distinct edges with at least one attachment row. - pub fn distinct_edges(&self) -> usize { - self.forward.len() - } - - /// Number of distinct annotation subjects with at least one row. - pub fn distinct_annotations(&self) -> usize { - self.reverse.len() - } -} - -/// Walk a row sequence and yield each "other" position whose latest -/// `(t, op)` event with `t <= as_of_t` is currently asserted. -/// -/// Generic over both row types so both `current_annotations_for_at` -/// and `current_targets_for_at` share the same implementation. Rows -/// with `t > as_of_t` are ignored entirely (so a future retract is -/// invisible to a past view, and vice-versa). Stable: when the same -/// `(other, t)` appears twice (impossible in practice but not -/// enforced by the type), the *last-encountered* `op` wins. -fn latest_assertions_at<'a, R, T>( - rows: impl Iterator, - extract: impl Fn(&'a R) -> (&'a T, i64, bool), - as_of_t: i64, -) -> impl Iterator -where - R: 'a, - T: 'a + Ord + Clone, -{ - // Build a small map of "latest visible (t, op)" per `other`. A - // BTreeMap keyed on the `other` side gives a deterministic - // iteration order (good for tests and replay determinism). - let mut latest: BTreeMap<&'a T, (i64, bool)> = BTreeMap::new(); - for row in rows { - let (other, t, op) = extract(row); - if t > as_of_t { - continue; - } - latest - .entry(other) - .and_modify(|cur| { - // Tie-break on `op` at equal `t` so an assert (op = true) - // deterministically wins over a retract (op = false; false < - // true), matching the arena builder's `(t, op)` sort - // semantics. Without this, a novelty-only read and an - // arena-merged read could disagree when an edge is both - // asserted and retracted at the same `t`. - if (t, op) >= (cur.0, cur.1) { - *cur = (t, op); - } - }) - .or_insert((t, op)); - } - latest - .into_iter() - .filter_map(|(other, (_t, op))| if op { Some(other) } else { None }) -} - -#[cfg(test)] -mod tests { - use super::*; - use fluree_db_core::edge::EdgeKey; - use fluree_db_core::{FlakeMeta, FlakeValue, Sid}; - - #[test] - fn latest_assertions_tie_breaks_assert_over_retract_at_equal_t() { - // STOR-2/BUGS-3: at equal t an assert (op=true) must win over a - // retract (op=false; false < true), matching the arena builder's - // (t, op) latest-wins sort, so novelty-only reads agree with - // arena-merged reads regardless of insertion order. - let target = Sid::new(13, "x"); - // assert then retract, both at t=5 — the discriminating case - // (pre-fix `t >= cur.0` let the trailing retract overwrite). - let rows = [ - (target.clone(), 5_i64, true), - (target.clone(), 5_i64, false), - ]; - let live: Vec<&Sid> = - latest_assertions_at(rows.iter(), |r| (&r.0, r.1, r.2), 100).collect(); - assert_eq!( - live, - vec![&target], - "assert must win at equal t (assert-then-retract)" - ); - - // retract then assert — same answer. - let rows2 = [ - (target.clone(), 5_i64, false), - (target.clone(), 5_i64, true), - ]; - let live2: Vec<&Sid> = - latest_assertions_at(rows2.iter(), |r| (&r.0, r.1, r.2), 100).collect(); - assert_eq!( - live2, - vec![&target], - "assert must win at equal t (retract-then-assert)" - ); - } - - fn sample_edge() -> EdgeKey { - EdgeKey::from_flake(&Flake::new( - Sid::new(13, "alice"), - Sid::new(13, "worksFor"), - FlakeValue::Ref(Sid::new(13, "acme")), - fluree_db_core::edge::id_datatype_sid(), - 42, - true, - None, - )) - } - - fn ann_sid(name: &str) -> Sid { - Sid::new(13, name) - } - - #[test] - fn empty_overlay_reports_no_annotations() { - let overlay = AttachmentNovelty::new(); - assert!(!overlay.has_annotations()); - let edge = sample_edge(); - assert!(overlay.current_annotations_for(&edge).next().is_none()); - } - - #[test] - fn observe_assertion_makes_attachment_visible() { - let edge = sample_edge(); - let ann = ann_sid("ann1"); - let bundle = edge.to_reifies_facts(&ann, 5, true); - - let mut overlay = AttachmentNovelty::new(); - overlay.observe_flakes(&bundle).unwrap(); - assert!(overlay.has_annotations()); - - let attached: Vec = overlay.current_annotations_for(&edge).collect(); - assert_eq!(attached, vec![ann.clone()]); - - let targets: Vec = overlay.current_targets_for(&ann).collect(); - assert_eq!(targets, vec![edge]); - } - - #[test] - fn assert_then_retract_clears_current_attachment_but_keeps_history() { - let edge = sample_edge(); - let ann = ann_sid("ann1"); - - let mut overlay = AttachmentNovelty::new(); - overlay - .observe_flakes(&edge.to_reifies_facts(&ann, 5, true)) - .unwrap(); - overlay - .observe_flakes(&edge.to_reifies_facts(&ann, 7, false)) - .unwrap(); - - assert!( - overlay.current_annotations_for(&edge).next().is_none(), - "retraction at t=7 must hide the t=5 attachment from current view" - ); - // History still sees both events. - let history: Vec<(i64, bool)> = overlay - .target_history(&ann) - .map(|row| (row.t, row.op)) - .collect(); - assert_eq!(history, vec![(5, true), (7, false)]); - // has_annotations remains sticky — the index has been touched. - assert!(overlay.has_annotations()); - } - - #[test] - fn current_annotations_for_at_respects_as_of_t() { - // Time-travel correctness: an annotation asserted at t=5 and - // retracted at t=7 must be visible to a view at t=5 or t=6, - // hidden at t=7+, and not yet visible at t=4. - let edge = sample_edge(); - let ann = ann_sid("ann_a"); - - let mut overlay = AttachmentNovelty::new(); - overlay - .observe_flakes(&edge.to_reifies_facts(&ann, 5, true)) - .unwrap(); - overlay - .observe_flakes(&edge.to_reifies_facts(&ann, 7, false)) - .unwrap(); - - // Before any event: no rows visible. - let before: Vec = overlay.current_annotations_for_at(&edge, 4).collect(); - assert!(before.is_empty(), "view at t=4 must not see t=5 assertion"); - - // At t=5 and t=6: assertion visible. - let at5: Vec = overlay.current_annotations_for_at(&edge, 5).collect(); - assert_eq!(at5, vec![ann.clone()], "view at t=5 must see the assertion"); - let at6: Vec = overlay.current_annotations_for_at(&edge, 6).collect(); - assert_eq!(at6, vec![ann.clone()], "view at t=6 must still see it"); - - // At t=7+: retract takes effect. - let at7: Vec = overlay.current_annotations_for_at(&edge, 7).collect(); - assert!(at7.is_empty(), "view at t=7 must see retraction"); - let at_max: Vec = overlay - .current_annotations_for_at(&edge, i64::MAX) - .collect(); - assert!(at_max.is_empty(), "live view sees retraction too"); - - // Live `current_annotations_for` agrees with as-of MAX. - let live: Vec = overlay.current_annotations_for(&edge).collect(); - assert!(live.is_empty()); - } - - #[test] - fn current_targets_for_at_respects_as_of_t() { - // Counterpart for the reverse map. - let edge = sample_edge(); - let ann = ann_sid("ann_a"); - - let mut overlay = AttachmentNovelty::new(); - overlay - .observe_flakes(&edge.to_reifies_facts(&ann, 5, true)) - .unwrap(); - - let at_4: Vec = overlay.current_targets_for_at(&ann, 4).collect(); - assert!(at_4.is_empty()); - let at_5: Vec = overlay.current_targets_for_at(&ann, 5).collect(); - assert_eq!(at_5, vec![edge]); - } - - #[test] - fn parallel_annotations_on_one_edge_both_visible() { - // Two distinct annotation subjects attached to the same edge. - let edge = sample_edge(); - let ann_a = ann_sid("ann_A"); - let ann_b = ann_sid("ann_B"); - - let mut overlay = AttachmentNovelty::new(); - overlay - .observe_flakes(&edge.to_reifies_facts(&ann_a, 5, true)) - .unwrap(); - overlay - .observe_flakes(&edge.to_reifies_facts(&ann_b, 6, true)) - .unwrap(); - - let mut attached: Vec = overlay.current_annotations_for(&edge).collect(); - attached.sort(); - assert_eq!(attached, vec![ann_a, ann_b]); - } - - #[test] - fn observe_ignores_non_reifies_flakes_silently() { - let edge = sample_edge(); - let ann = ann_sid("ann1"); - - let mut overlay = AttachmentNovelty::new(); - // Annotation metadata flake (e.g. `ann ex:role "Engineer"`) - // accompanies the bundle but must not affect the overlay. - let mut all_flakes = edge.to_reifies_facts(&ann, 5, true); - all_flakes.push(Flake::new( - ann.clone(), - Sid::new(13, "role"), - FlakeValue::String("Engineer".into()), - Sid::new(2, "string"), - 5, - true, - None, - )); - overlay.observe_flakes(&all_flakes).unwrap(); - - // The metadata flake should be silently ignored — only the - // bundle drives the overlay. - assert_eq!(overlay.distinct_edges(), 1); - assert_eq!(overlay.distinct_annotations(), 1); - } - - #[test] - fn observe_no_op_on_empty_input() { - let mut overlay = AttachmentNovelty::new(); - overlay.observe_flakes(&[]).unwrap(); - assert!(!overlay.has_annotations()); - } - - #[test] - fn observe_skips_and_counts_graph_mismatch_bundle() { - // A bundle whose *flake-level* graph (the graph the - // f:reifies* flakes were asserted in) doesn't match the - // bundle's *decoded* graph (from the optional f:reifiesGraph - // flake) is malformed. Without this cross-check, a cross- - // graph collision could merge into a single (wrong-graph) - // bundle and file the attachment under the wrong edge — - // breaking the cascade fast-path on base-edge retract. - // Mirrors the arena builder's guard in - // `fluree_db_binary_index::annotation_arena::bundle::build_arenas_from_flakes`. - let edge = sample_edge(); // default-graph edge - let ann = ann_sid("ann_x"); - // Build a default-graph bundle (correct shape: edge.g == - // None, no f:reifiesGraph flake), then re-graph every flake - // to graph G_a — so flake-level g = Some(G_a) but decoded - // EdgeKey.g = None. Mismatch. - let g_a = Sid::new(13, "graph_a"); - let bundle_default: Vec = edge.to_reifies_facts(&ann, 5, true); - let bundle_mismatch: Vec = bundle_default - .into_iter() - .map(|f| Flake::new_in_graph(g_a.clone(), f.s, f.p, f.o, f.dt, f.t, f.op, f.m)) - .collect(); - - let mut overlay = AttachmentNovelty::new(); - overlay - .observe_flakes(&bundle_mismatch) - .expect("graph-mismatch bundle skipped, not an error"); - assert!(!overlay.has_annotations(), "no rows landed"); - assert_eq!( - overlay.observed_malformed_bundle_count(), - 1, - "graph-mismatch bundle counts as malformed" - ); - } - - #[test] - fn observe_skips_and_counts_malformed_bundle() { - // Strip a required predicate from the bundle — decoder rejects. - // Per the design contract, observe_flakes now SKIPS + warns + - // counts rather than erroring out, so a single corrupt event - // in replay can't block the rest of the ledger from loading. - let edge = sample_edge(); - let ann = ann_sid("ann1"); - let mut bundle = edge.to_reifies_facts(&ann, 5, true); - bundle.retain(|f| !fluree_db_core::namespaces::is_reifies_subject(&f.p)); - - let mut overlay = AttachmentNovelty::new(); - overlay - .observe_flakes(&bundle) - .expect("observe_flakes skips malformed bundles instead of erroring"); - // Overlay's attachment maps stay empty — only the malformed - // bundle was in the input. - assert!(!overlay.has_annotations()); - // The cumulative counter ticks so operators can detect the - // signal post-hoc. - assert_eq!(overlay.observed_malformed_bundle_count(), 1); - } - - #[test] - fn collect_forward_events_returns_arena_reader_input_shape() { - // Two events on the same edge: ann_a attached at t=5, retracted - // at t=7. `collect_forward_events` returns both as - // (ann, t, op) triples in row-stored order — ready to hand - // straight to AnnotationArenaReader::current_annotations_merged. - let edge = sample_edge(); - let ann = ann_sid("ann_a"); - let mut overlay = AttachmentNovelty::new(); - overlay - .observe_flakes(&edge.to_reifies_facts(&ann, 5, true)) - .unwrap(); - overlay - .observe_flakes(&edge.to_reifies_facts(&ann, 7, false)) - .unwrap(); - - let events = overlay.collect_forward_events(&edge); - assert_eq!(events.len(), 2); - assert_eq!(events[0], (ann.clone(), 5, true)); - assert_eq!(events[1], (ann, 7, false)); - } - - #[test] - fn collect_reverse_events_mirrors_forward_collector() { - let edge = sample_edge(); - let ann = ann_sid("ann_a"); - let mut overlay = AttachmentNovelty::new(); - overlay - .observe_flakes(&edge.to_reifies_facts(&ann, 5, true)) - .unwrap(); - - let events = overlay.collect_reverse_events(&ann); - assert_eq!(events.len(), 1); - assert_eq!(events[0].0, edge); - assert_eq!(events[0].1, 5); - assert!(events[0].2); - } - - #[test] - fn collect_events_empty_for_unknown_keys() { - let overlay = AttachmentNovelty::new(); - assert!(overlay.collect_forward_events(&sample_edge()).is_empty()); - assert!(overlay - .collect_reverse_events(&ann_sid("never_seen")) - .is_empty()); - } - - #[test] - fn observe_handles_named_graph_and_lang_bundles() { - let mut base = Flake::new( - Sid::new(13, "alice"), - Sid::new(13, "label"), - FlakeValue::String("Engineer".into()), - Sid::new(2, "string"), - 42, - true, - None, - ); - base.g = Some(Sid::new(13, "graph_a")); - base.m = Some(FlakeMeta { - lang: Some("fr".into()), - i: None, - }); - let edge = EdgeKey::from_flake(&base); - let ann = ann_sid("ann_named"); - - let mut overlay = AttachmentNovelty::new(); - overlay - .observe_flakes(&edge.to_reifies_facts(&ann, 5, true)) - .unwrap(); - - let attached: Vec = overlay.current_annotations_for(&edge).collect(); - assert_eq!(attached, vec![ann]); - } - - #[test] - fn malformed_bundle_skipped_with_warn_and_counter_bump() { - // A bundle missing the required `f:reifiesSubject` flake - // (decoder rejects with `EdgeKeyDecodeError::Missing`) used - // to error out the caller's commit. Per the design contract - // it now skips + warns + bumps the cumulative counter so a - // single corrupt event in replay can't block the rest of - // the ledger from loading. - use fluree_db_core::edge::id_datatype_sid; - use fluree_vocab::db as p; - use fluree_vocab::namespaces::FLUREE_DB; - - let ann = ann_sid("ann_bad"); - let id_dt = id_datatype_sid(); - - // Bundle with f:reifiesPredicate + f:reifiesObject only — - // missing f:reifiesSubject. `EdgeKey::from_reifies_facts` - // returns `Missing("f:reifiesSubject")`. - let malformed: Vec = vec![ - Flake::new( - ann.clone(), - Sid::new(FLUREE_DB, p::REIFIES_PREDICATE), - FlakeValue::Ref(Sid::new(13, "worksFor")), - id_dt.clone(), - 7, - true, - None, - ), - Flake::new( - ann.clone(), - Sid::new(FLUREE_DB, p::REIFIES_OBJECT), - FlakeValue::Ref(Sid::new(13, "acme")), - id_dt, - 7, - true, - None, - ), - ]; - - // A second, well-formed bundle for a different annotation in - // the same call so we can assert "the malformed one is - // skipped, the well-formed one still applies." - let edge = sample_edge(); - let good_ann = ann_sid("ann_good"); - let mut all = malformed; - all.extend(edge.to_reifies_facts(&good_ann, 7, true)); - - let mut overlay = AttachmentNovelty::new(); - overlay - .observe_flakes(&all) - .expect("observe_flakes must NOT error on malformed bundle — skip + count"); - - assert_eq!( - overlay.observed_malformed_bundle_count(), - 1, - "exactly one malformed bundle was observed" - ); - // The well-formed bundle still landed. - let attached: Vec = overlay.current_annotations_for(&edge).collect(); - assert_eq!(attached, vec![good_ann]); - // The malformed bundle's annotation has no live target. - assert!(overlay.current_targets_for(&ann).next().is_none()); - - // A second observe with a fresh malformed bundle bumps the - // counter again; it's cumulative across calls. - let another_bad: Vec = vec![Flake::new( - ann_sid("ann_bad2"), - Sid::new(FLUREE_DB, p::REIFIES_OBJECT), - FlakeValue::Ref(Sid::new(13, "acme")), - fluree_db_core::edge::id_datatype_sid(), - 8, - true, - None, - )]; - overlay.observe_flakes(&another_bad).unwrap(); - assert_eq!(overlay.observed_malformed_bundle_count(), 2); - } -} diff --git a/fluree-db-novelty/src/lib.rs b/fluree-db-novelty/src/lib.rs index 1b79097411..251adc67fa 100644 --- a/fluree-db-novelty/src/lib.rs +++ b/fluree-db-novelty/src/lib.rs @@ -33,7 +33,6 @@ //! let slice = novelty.slice_for_range(g_id, IndexType::Spot, Some(&first), Some(&rhs), false); //! ``` -pub mod attachments; mod commit; mod commit_flakes; pub mod delta; @@ -43,7 +42,6 @@ mod links; mod runtime_stats; mod stats; -pub use attachments::{AttachmentNovelty, ForwardRow, ReverseRow}; pub use commit::{ collect_dag_cids, collect_first_parent_cids, collect_first_parent_cids_with_split_mode, find_common_ancestor, load_commit_by_id, load_commit_envelope_by_id, @@ -561,12 +559,6 @@ pub struct Novelty { /// cached config is current. Stays 0 for ledgers that never write config. pub config_write_t: i64, - /// Edge-annotation attachment overlay (M1 — derived from the - /// `f:reifies*` system flakes flowing through the same pipeline). - /// Updated automatically by [`Self::apply_commit`] / - /// [`Self::bulk_apply_commits`] from the post-dedup flake set. - pub attachments: AttachmentNovelty, - /// Current-state fact index for RDF set-semantics dedup (latest op per /// identity, per graph, within this novelty window). Persistent map, so it /// clones in O(1). The dedup oracle behind the seam; see [`fact_state`]. @@ -577,9 +569,8 @@ pub struct Novelty { link_base: Option, /// Every commit at or before this `t` has its links in novelty. links_through: i64, - /// An `rdf:reifies` link has been applied; sticky, like - /// [`AttachmentNovelty::has_annotations`]. - saw_links: bool, + /// An annotation flake has been applied; sticky. + saw_annotations: bool, } #[inline] @@ -601,18 +592,17 @@ impl Novelty { shacl_epoch: 0, has_list_meta: false, config_write_t: 0, - attachments: AttachmentNovelty::new(), fact_state: NoveltyFactState::new(), link_base: None, links_through: t, - saw_links: false, + saw_annotations: false, } } /// True once novelty has applied an annotation: an `rdf:reifies` link or /// a legacy `f:reifies*` bundle flake. pub fn has_annotations(&self) -> bool { - self.saw_links || self.attachments.has_annotations() + self.saw_annotations } /// Resolve a flake's graph ID from its `Flake.g` field. @@ -958,11 +948,6 @@ impl Novelty { // is updated only after this loop). let mut per_graph: HashMap> = HashMap::new(); let mut deduped = 0u64; - // Capture post-dedup `f:reifies*` flakes so the attachment overlay - // observer sees exactly the flakes that landed in novelty. The - // reserved-predicate test is a single SID compare — negligible even on - // ledgers that never use annotations. - let mut accepted_reifies: Vec = Vec::new(); let mut schema_touched = false; let mut shacl_touched = false; for (flake, g_id) in routed { @@ -970,10 +955,7 @@ impl Novelty { deduped += 1; continue; } - if fluree_db_core::namespaces::is_reserved_reifies_predicate(&flake.p) { - accepted_reifies.push(flake.clone()); - } - self.saw_links |= fluree_db_core::is_rdf_reifies(&flake.p); + self.saw_annotations |= fluree_db_core::is_annotation_predicate(&flake.p); // Asserting OR retracting a hierarchy edge changes the RDFS // schema — invalidate the shared hierarchy cache. Likewise any // SHACL-vocabulary flake invalidates the compiled-shapes cache. @@ -1006,13 +988,6 @@ impl Novelty { ); } - // Update the attachment overlay from the post-dedup set. The observer - // skips quietly when no `f:reifies*` flakes are present — most commits - // never touch annotations. - if !accepted_reifies.is_empty() { - self.attachments.observe_flakes(&accepted_reifies)?; - } - // Build + append one immutable segment per touched graph. No merge into // existing storage — this is the per-commit O(novelty) → O(batch) // collapse. Small per-commit batches sort sequentially (rayon hand-off @@ -1180,16 +1155,10 @@ impl Novelty { } // Maintain the current-state index so later apply_commit calls dedup - // against bulk-loaded facts, and capture `f:reifies*` flakes for the - // attachment overlay observer (cheap clone, rare relative to data - // flakes). `kept` is in (s,p,o,dt,m,t,op) order, so the last record - // per identity is its highest-t (latest) op. - let mut accepted_reifies: Vec = Vec::new(); + // against bulk-loaded facts. `kept` is in (s,p,o,dt,m,t,op) order, so + // the last record per identity is its highest-t (latest) op. for flake in &kept { - if fluree_db_core::namespaces::is_reserved_reifies_predicate(&flake.p) { - accepted_reifies.push(flake.clone()); - } - self.saw_links |= fluree_db_core::is_rdf_reifies(&flake.p); + self.saw_annotations |= fluree_db_core::is_annotation_predicate(&flake.p); self.fact_state.record(g_id, flake); } @@ -1203,15 +1172,6 @@ impl Novelty { if g_id == CONFIG_GRAPH_ID { self.config_write_t = self.config_write_t.max(max_t); } - - // Update attachment overlay after the per-graph batch is committed. - // Malformed bundles are skipped + warned + counted on - // `attachments.observed_malformed_bundle_count` (see - // `AttachmentNovelty::observe_flakes`); `?` only propagates - // infrastructure-level errors. - if !accepted_reifies.is_empty() { - self.attachments.observe_flakes(&accepted_reifies)?; - } } self.t = max_t; diff --git a/fluree-db-query/src/planner.rs b/fluree-db-query/src/planner.rs index c479dcdae9..b6321dc02b 100644 --- a/fluree-db-query/src/planner.rs +++ b/fluree-db-query/src/planner.rs @@ -1228,10 +1228,8 @@ pub fn estimate_pattern( row_count: DEFAULT_SERVICE_ROW_COUNT, }, - // Edge-annotation patterns (M0): treated as a `Source` with the - // wrapped edge's cardinality as a first approximation. Real - // cost-based selection between edge-first and annotation-first - // scans arrives in M3 alongside `AnnotationStats`. + // Edge-annotation patterns expand before planning; estimated by the + // wrapped edge's cardinality as a first approximation. Pattern::EdgeAnnotation { edge, .. } => PatternEstimate::Source { row_count: estimate_triple_row_count(edge, bound_vars, stats), }, @@ -4384,89 +4382,6 @@ mod tests { ); } - #[test] - fn estimate_uses_merged_annotation_stats_for_reifies_predicates() { - // After `StatsView::merge_annotation_stats` runs with per-slot - // NDVs, the planner's classifier should produce arena-aligned - // BoundObject estimates: `count / ndv_values` for the matching - // slot's NDV, not the conservative fallback. - - use fluree_db_core::AnnotationStats; - use fluree_vocab::db as p; - use fluree_vocab::namespaces::FLUREE_DB; - - let mut stats = StatsView::default(); - let ann = AnnotationStats { - forward_rows: 1_000, - reverse_rows: 1_000, - distinct_edges: 200, - distinct_annotations: 800, - live_attachment_pairs: 800, - distinct_reified_subjects: 50, - distinct_reified_predicates: 4, - distinct_reified_objects: 200, - ..Default::default() - }; - let mut ns = std::collections::HashMap::new(); - ns.insert(FLUREE_DB, "https://ns.flur.ee/db#".to_string()); - stats.merge_annotation_stats(&ann, &ns); - - // PropertyScan: `?ann f:reifiesObject ?o` — total annotations. - let scan = TriplePattern::new( - Ref::Var(VarId(0)), - Ref::Sid(Sid::new(FLUREE_DB, p::REIFIES_OBJECT)), - Term::Var(VarId(1)), - ); - let scan_est = estimate_triple_row_count(&scan, &HashSet::new(), Some(&stats)); - assert_eq!( - scan_est, 800.0, - "PropertyScan should equal annotation count" - ); - - // BoundObject: `?ann f:reifiesObject `. With per- - // slot NDV the estimate is `800 / 200 = 4` — annotations per - // pinned object. - let bound_o = TriplePattern::new( - Ref::Var(VarId(0)), - Ref::Sid(Sid::new(FLUREE_DB, p::REIFIES_OBJECT)), - Term::Sid(Sid::new(7, "obj1")), - ); - let bound_o_est = estimate_triple_row_count(&bound_o, &HashSet::new(), Some(&stats)); - assert_eq!( - bound_o_est, 4.0, - "BoundObject on reifiesObject should be distinct_annotations / distinct_reified_objects" - ); - - // BoundSubject: a known annotation subject probing its slot. - let mut bound_subj_ctx = HashSet::new(); - bound_subj_ctx.insert(VarId(0)); - let bound_s = TriplePattern::new( - Ref::Var(VarId(0)), - Ref::Sid(Sid::new(FLUREE_DB, p::REIFIES_SUBJECT)), - Term::Var(VarId(2)), - ); - let bound_s_est = estimate_triple_row_count(&bound_s, &bound_subj_ctx, Some(&stats)); - assert_eq!( - bound_s_est, 1.0, - "BoundSubject on reifiesSubject should be ~1 row per known annotation" - ); - - // BoundObject on reifiesPredicate: 800 / 4 = 200 annotations - // per pinned predicate. Larger than reifiesObject's - // selectivity here, which is realistic — predicates are - // typically a small set even at scale. - let bound_p = TriplePattern::new( - Ref::Var(VarId(0)), - Ref::Sid(Sid::new(FLUREE_DB, p::REIFIES_PREDICATE)), - Term::Sid(Sid::new(7, "worksFor")), - ); - let bound_p_est = estimate_triple_row_count(&bound_p, &HashSet::new(), Some(&stats)); - assert_eq!( - bound_p_est, 200.0, - "BoundObject on reifiesPredicate should be distinct_annotations / distinct_reified_predicates" - ); - } - #[test] fn test_is_property_join() { // Valid property join: ?s :name ?n, ?s :age ?a diff --git a/fluree-db-query/src/stats_cache.rs b/fluree-db-query/src/stats_cache.rs index b9a144d9fc..81c67ec5e5 100644 --- a/fluree-db-query/src/stats_cache.rs +++ b/fluree-db-query/src/stats_cache.rs @@ -192,13 +192,6 @@ pub(crate) fn cached_stats_view_for_db( if !view.class_coverage_trustworthy { view.source = None; } - // Overlay arena-derived stats for `f:reifies*` predicates so the - // join planner gets tight selectivity estimates on snapshots - // with a built annotation index. See - // `StatsView::merge_annotation_stats` for the synthesis rules. - if let Some(ann) = db.snapshot.annotation_index.as_ref() { - view.merge_annotation_stats(&ann.stats, db.snapshot.namespaces()); - } tracing::debug!( stats_view_build_ms = started.elapsed().as_secs_f64() * 1000.0, classes = view.classes.len(), @@ -215,28 +208,15 @@ pub(crate) fn cached_stats_view_for_db( // always have different epoch values. Limitation: if an overlay is replaced by a // wholly new instance (e.g. after ledger reload), epoch resets to 0, but in that // case snapshot.t will also differ, so the key remains unique. - // - // We also fold in the annotation arena's identity (`forward_branch_cid` + - // `reverse_branch_cid`) so a reindex/rebuild that swaps the arena at the - // same `snapshot.t` produces a fresh cache slot — `merge_annotation_stats` - // depends on these contents, and CIDs are content-addressed so they - // rotate on any rebuild. - let arena_key = db - .snapshot - .annotation_index - .as_ref() - .map(|a| format!("{}:{}", a.forward_branch_cid, a.reverse_branch_cid)) - .unwrap_or_else(|| "none".to_string()); let cache_key = xxh3_128( format!( - "stats-view:{}:{}:{}:{}:{}:{}:{}", + "stats-view:{}:{}:{}:{}:{}:{}", db.snapshot.ledger_id, db.snapshot.t, db.t, db.overlay.epoch(), u8::from(db.runtime_small_dicts.is_some() || binary_store.is_some()), u8::from(allow_semantic_elision), - arena_key, ) .as_bytes(), ); diff --git a/fluree-db-server/src/routes/import.rs b/fluree-db-server/src/routes/import.rs index d4ce2c61c8..e4ca240798 100644 --- a/fluree-db-server/src/routes/import.rs +++ b/fluree-db-server/src/routes/import.rs @@ -861,34 +861,13 @@ async fn run_source_import( .await .map_err(|e| e.to_string())?; - // The bulk-imported root carries `annotation_index: None`. One reindex - // through the api's attachment provider runs the bulk-import bootstrap - // scan and seals an authoritative annotation arena, so relationship- - // binding queries take the arena probe instead of scan-fallback — - // mirroring what `fluree create --from` does after a local import. - let sealed_root_id = if result.has_annotations { - Some( - state - .fluree - .reindex(ledger_id, fluree_db_api::ReindexOptions::default()) - .await - .map_err(|e| format!("post-import annotation-arena seal (reindex): {e}"))? - .root_id, - ) - } else { - None - }; - Ok(serde_json::json!({ "kind": "bulk-import", "ledger_id": result.ledger_id, "t": result.t, "flake_count": result.flake_count, "commit_head_id": result.commit_head_id.to_string(), - "root_id": sealed_root_id - .as_ref() - .or(result.root_id.as_ref()) - .map(std::string::ToString::to_string), + "root_id": result.root_id.as_ref().map(std::string::ToString::to_string), "index_t": result.index_t, "has_annotations": result.has_annotations, })) diff --git a/fluree-db-server/tests/flpack_import_integration.rs b/fluree-db-server/tests/flpack_import_integration.rs index e225b98ebe..f481dfff48 100644 --- a/fluree-db-server/tests/flpack_import_integration.rs +++ b/fluree-db-server/tests/flpack_import_integration.rs @@ -915,13 +915,12 @@ async fn discovery_advertises_source_upload_when_presign_enabled() { } #[tokio::test] -async fn source_upload_with_relationships_seals_annotation_arena() { +async fn source_upload_with_relationships_imports_annotated_edges() { let (_tmp, state) = presign_test_state().await; let dst = "src-cypher-rel/data:main"; // Relationship CREATEs reify edges (EdgePolicy::Annotated default), so the - // imported ledger carries f:reifies* facts; the post-import reindex must - // leave a sealed annotation arena, not scan-fallback. + // imported ledger carries annotations. let cypher = br#"CREATE (:Person {name: "Alice"}); CREATE (:Person {name: "Bob"}); MATCH (a:Person {name: "Alice"}), (b:Person {name: "Bob"}) CREATE (a)-[:KNOWS {since: 2020}]->(b); @@ -943,16 +942,12 @@ MATCH (a:Person {name: "Alice"}), (b:Person {name: "Bob"}) CREATE (a)-[:KNOWS {s assert_eq!( final_status["result"]["root_id"].as_str(), expected_root.as_deref(), - "job result must return the post-seal nameservice root" + "job result must return the nameservice root" ); assert!( handle.snapshot.has_annotations, "imported ledger carries annotations" ); - assert!( - handle.snapshot.annotation_index.is_some(), - "post-import reindex must seal the annotation arena" - ); // The relationship is queryable through the Cypher surface. let db = GraphDb::from_ledger_state(&handle); diff --git a/fluree-db-transact/src/import_sink.rs b/fluree-db-transact/src/import_sink.rs index 45cb664fe8..96c2f942fa 100644 --- a/fluree-db-transact/src/import_sink.rs +++ b/fluree-db-transact/src/import_sink.rs @@ -1457,13 +1457,10 @@ mod inner { } #[test] - fn test_reified_triple_streams_jsonld_compatible_bundle() { - // The import path must write the SAME bundle shape as the - // transactional FlakeSink / JSON-LD lowering: base triple + - // S/P/O bundle, no f:reifiesDatatype, decodable to the base - // edge's EdgeKey. - use fluree_db_core::edge::EdgeKey; - use fluree_db_core::namespaces::is_reifies_datatype; + fn test_reified_triple_streams_its_link() { + // Import writes the same record as the transactional FlakeSink: + // the base triple plus `r rdf:reifies <<( s p o )>>`. + use fluree_db_core::FlakeValue; let mut ns = NamespaceRegistry::new(); let mut sink = make_sink_and_parse(&mut ns, 1).unwrap(); @@ -1476,19 +1473,22 @@ mod inner { sink.emit_reified_triple(s, p, o, r).unwrap(); let (writer, _prefix_map, _spool) = sink.into_parts().unwrap(); - assert_eq!(writer.op_count(), 4, "base + 3 bundle flakes"); + assert_eq!(writer.op_count(), 2, "base + link"); let result = writer.finish(&make_envelope(1)).unwrap(); let decoded = read_commit(&result.bytes).unwrap(); - assert_eq!(decoded.flakes.len(), 4); - let base = &decoded.flakes[0]; - let bundle = &decoded.flakes[1..]; - assert!( - !bundle.iter().any(|f| is_reifies_datatype(&f.p)), - "import bundle must omit f:reifiesDatatype: {bundle:?}" + let [base, link] = decoded.flakes.as_slice() else { + panic!("expected base + link: {:?}", decoded.flakes); + }; + assert!(fluree_db_core::is_rdf_reifies(&link.p)); + let FlakeValue::TripleTerm(term) = &link.o else { + panic!("link object is not a triple term: {link:?}"); + }; + assert_eq!( + (&term.s, &term.p, &term.o), + (&base.s, &base.p, &base.o), + "the link names the base edge" ); - let key = EdgeKey::from_reifies_facts(bundle).expect("bundle decodes"); - assert_eq!(key, EdgeKey::from_flake(base)); } #[test] From be086a152d903292bd11074e4d3443390477cad4 Mon Sep 17 00:00:00 2001 From: bplatz Date: Sat, 3 Oct 2026 08:21:42 -0400 Subject: [PATCH 55/92] feat(query): rdf:reifies links are ordinary data in wildcard reads RDF 1.2 makes a reifier's `r rdf:reifies <<( s p o )>>` an ordinary triple, so `?s ?p ?o` scans, property-path closures and wildcard hydration now return it; hydration renders the term as a JSON-LD-star embedded node. An `@annotation` body leaves out its reifier's link, and Cypher property maps keep treating the link as the relationship itself. Only the legacy `f:reifies*` bundle predicates stay hidden. The scan looks up their p_ids once per index store, so a ledger without legacy bundles skips the per-row predicate check, and a variable-predicate COUNT(*) under novelty counts the cursor without binding rows (routing stamp `scan-count-drain`). The whole-graph count fast path now declines on a graph holding legacy bundle rows; it summed them before. Six W3C SPARQL 1.2 triple-term evaluation tests now pass. --- docs/design/edge-annotations.md | 4 +- docs/query/jsonld-query.md | 13 +- fluree-db-api/Cargo.toml | 5 + fluree-db-api/src/format/hydration.rs | 53 +- fluree-db-api/tests/it_edge_annotations.rs | 457 +++++++++--------- fluree-db-api/tests/it_scan_count_drain.rs | 96 ++++ fluree-db-api/tests/it_triple_term_links.rs | 9 +- fluree-db-api/tests/support/mod.rs | 73 +++ .../src/read/binary_index_store.rs | 16 + fluree-db-core/src/namespaces.rs | 10 +- fluree-db-query/src/binary_scan.rs | 72 +-- fluree-db-query/src/eval/metadata.rs | 4 +- fluree-db-query/src/fast_count.rs | 15 +- fluree-db-query/src/fast_whole_graph_agg.rs | 27 +- testsuite-sparql/tests/registers/mod.rs | 16 +- 15 files changed, 542 insertions(+), 328 deletions(-) create mode 100644 fluree-db-api/tests/it_scan_count_drain.rs diff --git a/docs/design/edge-annotations.md b/docs/design/edge-annotations.md index 2094667a02..a5750ce311 100644 --- a/docs/design/edge-annotations.md +++ b/docs/design/edge-annotations.md @@ -55,13 +55,13 @@ Annotation syntax (`s p o ~ ?r {| ... |}`, JSON-LD `@annotation`, Cypher relatio The link names its triple without joining the base edge, so visibility is checked on the term: `QueryPolicyEnforcer` lets a flake whose object is a triple term through only when the triple that term names would be visible, recursively for nested terms. A policy hiding `ex:worksFor` therefore hides the links to `ex:worksFor` edges on every route. -Wildcard scans (`?s ?p ?o`, wildcard hydration, Cypher property maps) hide `rdf:reifies`; `opts.includeSystemFacts` shows it. +A link is ordinary data: wildcard scans (`?s ?p ?o`) and wildcard hydration return it like any triple, and hydration renders its triple term as an embedded node. An `@annotation` body leaves its reifier's link out, since the body hangs from it, and so do Cypher property maps, where the link is the relationship itself. Hydration (`@annotation` in subject expansion) and export read links the same way: hydration probes `rdf:reifies` with the rendered edge's term; export reads the ledger's live links once and writes a `~ ` marker on each base edge it reaches. ## Ledgers written before links -Earlier releases stored an annotation as an `f:reifies*` bundle (`f:reifiesSubject`, `f:reifiesPredicate`, `f:reifiesObject`, …) on the reifier. Those bundles stay in commits, and the link is derived from them wherever it is needed: an index build derives it from each commit's bundle ops (`link_synth`), and novelty derives it for commits the index has not covered (`fluree-db-novelty/src/links.rs`). Readers see only links. A write that retracts a derived link leaves the bundle in place; the retract cancels the derived assert in novelty and at the next build alike. +Earlier releases stored an annotation as an `f:reifies*` bundle (`f:reifiesSubject`, `f:reifiesPredicate`, `f:reifiesObject`, …) on the reifier. Those bundles stay in commits, hidden from wildcard scans and hydration unless `opts.includeSystemFacts` asks for them; the scan finds their predicate ids once per index store (`scan_hidden_p_ids`), so a ledger without bundles checks no row. The link is derived from them wherever it is needed: an index build derives it from each commit's bundle ops (`link_synth`), and novelty derives it for commits the index has not covered (`fluree-db-novelty/src/links.rs`). Readers see only links. A write that retracts a derived link leaves the bundle in place; the retract cancels the derived assert in novelty and at the next build alike. Index roots from those releases may also carry an annotation-arena section. Readers skip it, keeping only the arena's two branch CIDs, and the next index build releases the arena's blobs as garbage. diff --git a/docs/query/jsonld-query.md b/docs/query/jsonld-query.md index f9bdaeb1d6..d44d33863e 100644 --- a/docs/query/jsonld-query.md +++ b/docs/query/jsonld-query.md @@ -317,12 +317,13 @@ a subject: Variable-predicate scans return every stored triple, including data written with the Fluree vocabulary (`https://ns.flur.ee/db#`, e.g. stored -`f:AccessPolicy` definitions). The one exception is the seven `f:reifies*` -predicates — the internal storage encoding of edge annotations. They are -system-written (user transactions cannot assert them), redundant with the -edge and annotation content already in the results, and therefore hidden from -variable-predicate scans. Pass `"opts": {"includeSystemFacts": true}` to -surface them for debugging or inspection. Commit metadata (`f:t`, `f:address`, +`f:AccessPolicy` definitions) and an annotation's `rdf:reifies` link. The one +exception is the seven `f:reifies*` predicates, which ledgers written by +earlier releases used to store edge annotations. They are system-written +(user transactions cannot assert them), redundant with the `rdf:reifies` +links derived from them, and therefore hidden from variable-predicate scans. +Pass `"opts": {"includeSystemFacts": true}` to surface them for debugging or +inspection. Commit metadata (`f:t`, `f:address`, …) lives in the ledger's txn-meta graph, not the default graph, so it never appears in default-graph scans either way. diff --git a/fluree-db-api/Cargo.toml b/fluree-db-api/Cargo.toml index d5fbd9caef..61bef03d55 100644 --- a/fluree-db-api/Cargo.toml +++ b/fluree-db-api/Cargo.toml @@ -419,6 +419,11 @@ path = "tests/it_bm25_sync_scoping.rs" name = "it_minmax_fast_path_fired" path = "tests/it_minmax_fast_path_fired.rs" +# Standalone routing-stamp test, same convention as above. +[[test]] +name = "it_scan_count_drain" +path = "tests/it_scan_count_drain.rs" + # Standalone (not in grp_transact): asserts on the `upsert deletion subject # pre-check` tracing event via the thread-local span capture, so it must be the # only test in its process. Same convention as the span-capture tests above. diff --git a/fluree-db-api/src/format/hydration.rs b/fluree-db-api/src/format/hydration.rs index b99f57210d..a6fd0f4515 100644 --- a/fluree-db-api/src/format/hydration.rs +++ b/fluree-db-api/src/format/hydration.rs @@ -1308,19 +1308,8 @@ impl<'a> HydrationFormatter<'a> { // Format each predicate for (pred, mut pred_flakes) in by_pred { - // System-fact filter (M1b): the seven `f:reifies*` - // predicates encode an annotation's reified edge. They - // are system-controlled — never user-data — and must - // not leak through wildcard subject hydration. Direct - // user mention in queries is already blocked by the - // parser firewall in `fluree-db-query::parse`; this - // filter closes the wildcard-projection path. - // - // Explicitly-listed levels can still reach these via - // a `Pattern::Triple` lookup at the planner layer - // (which is what the `Pattern::EdgeAnnotation` IR - // expansion and the `rdf:reifies` link lowering do), but those - // patterns don't go through hydration. + // The legacy `f:reifies*` bundle predicates are an internal + // encoding of annotations, hidden here as from wildcard scans. if fluree_db_core::is_scan_hidden_predicate(&pred) { continue; } @@ -1616,6 +1605,7 @@ impl<'a> HydrationFormatter<'a> { let mut values = Vec::new(); for flake in pred_ctx.flakes { match &flake.o { + FlakeValue::TripleTerm(term) => values.push(self.format_triple_term(term)?), FlakeValue::Ref(ref_sid) => { if is_rdf_type { // @type special case: compact IRI string, not {"@id": ...} @@ -1830,7 +1820,8 @@ impl<'a> HydrationFormatter<'a> { } /// Render the annotation bodies of `ann_sids` through a wildcard - /// select spec. + /// select spec, without the reifiers' `rdf:reifies` links: a link is + /// the attachment the body hangs from, not part of it. async fn render_annotation_bodies<'b>( &'b self, ann_sids: &[Sid], @@ -1842,13 +1833,15 @@ impl<'a> HydrationFormatter<'a> { refinements: HashMap::new(), reverse: HashMap::new(), }; + let link_key = self.format_predicate_key(fluree_db_core::rdf_reifies_sid())?; let mut bodies: Vec = Vec::with_capacity(ann_sids.len()); for ann_sid in ann_sids { let mut body = self .format_subject(ann_sid, None, &ann_level, depth.descend(), visited, cache) .await?; - if ann_sid.namespace_code == BLANK_NODE { - if let Some(map) = body.as_object_mut() { + if let Some(map) = body.as_object_mut() { + map.remove(&link_key); + if ann_sid.namespace_code == BLANK_NODE { map.remove("@id"); } } @@ -1954,6 +1947,34 @@ impl<'a> HydrationFormatter<'a> { } /// Format a literal flake value + /// A triple term as a JSON-LD-star embedded node; a literal object + /// renders as this formatter renders that literal anywhere else. + fn format_triple_term(&self, term: &fluree_db_core::TripleTermValue) -> Result { + super::triple_term_node(term, self.compactor, |_| match &term.o { + FlakeValue::TripleTerm(inner) => self.format_triple_term(inner), + _ => { + let meta = term.lang.clone().map(|lang| fluree_db_core::FlakeMeta { + lang: Some(lang), + i: None, + }); + let object = Flake::new( + term.s.clone(), + term.p.clone(), + term.o.clone(), + term.dt.clone(), + 0, + true, + meta, + ); + if self.typed { + self.format_typed_literal_value(&object) + } else { + self.format_literal_value(&object) + } + } + }) + } + fn format_literal_value(&self, flake: &Flake) -> Result { let dt_full = self.compactor.decode_sid(&flake.dt)?; let dt_compact = self.compactor.compact_sid(&flake.dt)?; diff --git a/fluree-db-api/tests/it_edge_annotations.rs b/fluree-db-api/tests/it_edge_annotations.rs index b4d955f6fc..a0b8aecb1f 100644 --- a/fluree-db-api/tests/it_edge_annotations.rs +++ b/fluree-db-api/tests/it_edge_annotations.rs @@ -1844,149 +1844,85 @@ async fn cascade_keeps_explicit_iri_annotation_metadata() { assert_eq!(row[1].as_str(), Some("Engineer")); } -#[tokio::test] -async fn variable_predicate_scan_hides_f_reifies_in_named_graph() { - // Annotation bundles are emitted in the reified edge's graph, - // so a variable-predicate scan scoped to a named graph would - // expose `f:reifies*` flakes there too. The filter must apply - // to every graph, not only the default graph. - let fluree = FlureeBuilder::memory().build_memory(); - let ledger_id = "it/edge-annotations:variable-predicate-named-graph"; - let ledger0 = genesis_ledger(&fluree, ledger_id); - - let txn = json!({ - "@context": ctx(), - "@id": "ex:alice", - "@graph": "ex:hr-graph", - "ex:worksFor": { - "@id": "ex:acme", - "@annotation": { - "@id": "ex:emp/alice-acme", - "ex:role": "Engineer" - } - } - }); - let committed = fluree - .insert(ledger0, &txn) - .await - .expect("named-graph annotated insert"); - - // Scope the variable-predicate scan to the named graph via the - // dataset alias. - let named_graph_alias = format!("{ledger_id}#http://example.org/hr-graph"); - let query = json!({ - "@context": ctx(), - "from": &named_graph_alias, - "select": ["?p"], - "where": { "@id": "ex:emp/alice-acme", "?p": "?o" } - }); - - // `query_connection` is the dataset-aware path; pair with a - // formatter against the post-insert snapshot. - let result = fluree - .query_connection(&query) - .await - .expect("named-graph variable-predicate query"); - let ledger = fluree.ledger(ledger_id).await.expect("reload ledger"); - let json = result.to_jsonld(&ledger.snapshot).expect("to_jsonld"); - let arr = json.as_array().expect("array"); - - // Collect predicate bindings and assert no `f:reifies*` leaks. - let predicates: Vec = arr +/// The predicates a `?p` row binds, as rendered strings. +fn predicate_column(rows: &JsonValue) -> Vec { + rows.as_array() + .expect("array") .iter() - .filter_map(|row| row.as_array()) - .filter_map(|cols| cols.first()) + .filter_map(|row| row.as_array().and_then(|cols| cols.first())) .filter_map(|v| { v.as_str() .map(String::from) .or_else(|| v.get("@id").and_then(|i| i.as_str()).map(String::from)) }) - .collect(); - for p in &predicates { - assert!( - !p.contains("reifies"), - "the link must not leak from named-graph variable-predicate scan: {p} \ - (full bindings: {predicates:?})" - ); - } - // The user-authored predicate should still be visible. - assert!( - predicates - .iter() - .any(|p| p == "http://example.org/role" || p == "ex:role"), - "user-authored ex:role must be visible in named-graph scan: {predicates:?}" - ); + .collect() +} - // Drop unused suppression: the test is the assertion. - drop(committed); +/// `rdf:reifies` in any of its rendered forms. +fn is_link_key(p: &str) -> bool { + p == "rdf:reifies" || p == "http://www.w3.org/1999/02/22-rdf-syntax-ns#reifies" } +/// The reifier's link is an ordinary triple: a variable-predicate scan of +/// the reifier returns it beside the body, in the edge's named graph too. #[tokio::test] -async fn variable_predicate_scan_hides_f_reifies() { - // A triple pattern with a variable predicate (`?s ?p ?o`) used - // to surface `f:reifies*` system flakes from the annotation - // subject's overlay rows. The scan-layer filter in - // `flakes_to_bindings` skips Fluree-system-namespace predicates - // when the user's predicate slot is a variable, mirroring the - // existing filter on the binary-cursor path. +async fn variable_predicate_scan_returns_the_link() { let fluree = FlureeBuilder::memory().build_memory(); - let ledger_id = "it/edge-annotations:variable-predicate-no-leak"; - let ledger0 = genesis_ledger(&fluree, ledger_id); - - let txn = json!({ - "@context": ctx(), - "@id": "ex:alice", - "ex:worksFor": { - "@id": "ex:acme", - "@annotation": { - "@id": "ex:emp/alice-acme", - "ex:role": "Engineer" + for (ledger_id, graph) in [ + ("it/edge-annotations:link-default-graph", None), + ("it/edge-annotations:link-named-graph", Some("ex:hr-graph")), + ] { + let ledger0 = genesis_ledger(&fluree, ledger_id); + let mut txn = json!({ + "@context": ctx(), + "@id": "ex:alice", + "ex:worksFor": { + "@id": "ex:acme", + "@annotation": { + "@id": "ex:emp/alice-acme", + "ex:role": "Engineer" + } } + }); + if let Some(graph) = graph { + txn["@graph"] = json!(graph); } - }); - let committed = fluree - .insert(ledger0, &txn) - .await - .expect("annotated insert"); - - // Bind ?p to every predicate the annotation subject carries. - let query = json!({ - "@context": ctx(), - "select": ["?p"], - "where": { "@id": "ex:emp/alice-acme", "?p": "?o" } - }); - let rows = support::query_jsonld_formatted(&fluree, &committed.ledger, &query) - .await - .expect("variable-predicate query"); - let arr = rows.as_array().expect("array"); - - // Collect the predicate bindings as strings. - let predicates: Vec = arr - .iter() - .filter_map(|row| row.as_array()) - .filter_map(|cols| cols.first()) - .filter_map(|v| { - v.as_str() - .map(String::from) - .or_else(|| v.get("@id").and_then(|i| i.as_str()).map(String::from)) - }) - .collect(); + let committed = fluree + .insert(ledger0, &txn) + .await + .expect("annotated insert"); - // No `f:reifies*` predicate may leak. - for p in &predicates { + let mut query = json!({ + "@context": ctx(), + "select": ["?p"], + "where": { "@id": "ex:emp/alice-acme", "?p": "?o" } + }); + let rows = match graph { + None => support::query_jsonld_formatted(&fluree, &committed.ledger, &query) + .await + .expect("default-graph query"), + Some(_) => { + query["from"] = json!(format!("{ledger_id}#http://example.org/hr-graph")); + let result = fluree + .query_connection(&query) + .await + .expect("named-graph query"); + result + .to_jsonld(&committed.ledger.snapshot) + .expect("to_jsonld") + } + }; + let mut predicates = predicate_column(&rows); + predicates.sort(); assert!( - !p.contains("reifies"), - "the link must not leak through variable-predicate scan: {p} \ - (full bindings: {predicates:?})" + predicates.len() == 2 + && predicates.iter().any(|p| is_link_key(p)) + && predicates + .iter() + .any(|p| p == "ex:role" || p == "http://example.org/role"), + "graph {graph:?}: the scan returns the link and the body: {predicates:?}" ); } - // The user-authored `ex:role` must still be visible. - assert!( - predicates - .iter() - .any(|p| p == "http://example.org/role" || p == "ex:role"), - "user-authored ex:role must be visible: {predicates:?}" - ); } #[tokio::test] @@ -2153,75 +2089,44 @@ async fn opts_include_system_facts_propagates_through_dataset_path() { async fn opts_include_system_facts_works_for_ask_queries() { // ASK queries return from the parser before `parse_options()` // runs, so `opts.includeSystemFacts` has to be parsed inline on - // that branch. Without that, an ASK against an annotation - // subject's `?p`-shape would always answer false even with the - // opt-in set. + // that branch. Without that, an ASK against a reifier's + // `?p`-shape would always answer false even with the opt-in set. + // The flag only reveals the legacy `f:reifies*` bundle, so the + // reifier carries one. let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "it/edge-annotations:opts-ask"; - let ledger0 = genesis_ledger(&fluree, ledger_id); - let _ = fluree - .insert( - ledger0, - &json!({ - "@context": ctx(), - "@id": "ex:alice", - "ex:worksFor": { - "@id": "ex:acme", - "@annotation": { - "@id": "ex:emp/alice-acme", - "ex:role": "Engineer" - } - } - }), - ) - .await - .expect("annotated insert"); - - let ledger = fluree.ledger(ledger_id).await.expect("reload"); + let ledger = support::commit_legacy_bundle(&fluree, genesis_ledger(&fluree, ledger_id)).await; - // Ask whether the annotation subject has *any* predicate. With - // the filter on (default) this still answers true via the - // ex:role flake. Pin the discriminating shape: a variable-predicate - // row whose object is a triple term, which only the link is. - let q_default = json!({ - "@context": ctx(), - "ask": [ - { "@id": "ex:emp/alice-acme", "?p": "?o" }, - ["filter", "(istriple ?o)"] - ] - }); - let resp_default = fluree - .query(&support::graphdb_from_ledger(&ledger), &q_default) - .await - .expect("ask default"); - let json_default: JsonValue = resp_default - .to_jsonld(&ledger.snapshot) - .expect("to_jsonld default"); + let ask = |opts: Option| { + let mut q = json!({ + "@context": ctx(), + "ask": [ + { "@id": "ex:emp/alice-acme", "?p": "?o" }, + ["filter", "(strStarts (str ?p) \"https://ns.flur.ee/db#reifies\")"] + ] + }); + if let Some(opts) = opts { + q["opts"] = opts; + } + let (fluree, ledger) = (&fluree, &ledger); + async move { + fluree + .query(&support::graphdb_from_ledger(ledger), &q) + .await + .expect("ask") + .to_jsonld(&ledger.snapshot) + .expect("to_jsonld") + } + }; assert_eq!( - json_default, + ask(None).await, JsonValue::Bool(false), - "without includeSystemFacts, ASK over the hidden link must answer false: {json_default}" + "without includeSystemFacts the bundle stays hidden" ); - - // With the opt-in, the ASK now returns true because the scan - // filter is bypassed and the link binds. - let q_opt = json!({ - "@context": ctx(), - "ask": [ - { "@id": "ex:emp/alice-acme", "?p": "?o" }, - ["filter", "(istriple ?o)"] - ], - "opts": { "includeSystemFacts": true } - }); - let resp_opt = fluree - .query(&support::graphdb_from_ledger(&ledger), &q_opt) - .await - .expect("ask opt"); - let json_opt: JsonValue = resp_opt.to_jsonld(&ledger.snapshot).expect("to_jsonld opt"); assert_eq!( - json_opt, + ask(Some(json!({ "includeSystemFacts": true }))).await, JsonValue::Bool(true), - "ASK + opts.includeSystemFacts must surface the link: {json_opt}" + "includeSystemFacts reveals the bundle to ASK" ); } @@ -2289,19 +2194,12 @@ async fn history_query_surfaces_f_reifies_events() { } #[tokio::test] -async fn wildcard_subject_hydration_hides_f_reifies_predicates() { - // Annotation subjects minted by the M1a transactor lowering carry - // `f:reifies*` system facts in addition to the user-authored body - // properties. Wildcard subject hydration (`select: {"?s": ["*"]}`) - // expands all properties of a subject, which would otherwise leak - // these system facts to the user. - // - // The hydration-layer filter in `format/hydration.rs` skips any - // predicate where `is_reserved_reifies_predicate(&p)` returns - // true. This test pins that contract: the wildcard projection - // sees the user's `ex:role` but not any `f:reifies*` predicate. +async fn wildcard_subject_hydration_shows_the_link_but_not_in_annotation_bodies() { + // A reifier's link is ordinary data, so wildcard hydration of the + // reifier renders it — as a JSON-LD-star embedded node. An + // `@annotation` body hangs from that link and leaves it out. let fluree = FlureeBuilder::memory().build_memory(); - let ledger_id = "it/edge-annotations:wildcard-hides-reifies"; + let ledger_id = "it/edge-annotations:wildcard-link"; let ledger0 = genesis_ledger(&fluree, ledger_id); let txn = json!({ @@ -2320,42 +2218,37 @@ async fn wildcard_subject_hydration_hides_f_reifies_predicates() { .await .expect("annotated insert"); - let query = json!({ + let reifier = json!({ "@context": ctx(), "select": {"?ann": ["*"]}, "where": { "@id": "?ann", "ex:role": "Engineer" } }); - - let rows = support::query_jsonld_formatted(&fluree, &committed.ledger, &query) + let rows = support::query_jsonld_formatted(&fluree, &committed.ledger, &reifier) .await - .expect("wildcard hydration over annotation subject"); - let arr = rows.as_array().expect("array"); - assert!( - !arr.is_empty(), - "wildcard hydration should find the annotation subject" + .expect("wildcard hydration of the reifier"); + let node = rows[0].as_object().expect("hydrated node"); + let link = node + .iter() + .find_map(|(key, value)| is_link_key(key).then_some(value)) + .unwrap_or_else(|| panic!("wildcard hydration renders the link: {node:#?}")); + assert_eq!( + link, + &json!({"@id": {"@id": "ex:alice", "ex:worksFor": {"@id": "ex:acme"}}}), + "the link renders as an embedded node" ); - // The user's `ex:role` is visible. - let node = arr[0] - .as_object() - .expect("hydrated node should be an object"); - let role_visible = node - .get("ex:role") - .or_else(|| node.get("http://example.org/role")) - .is_some(); - assert!( - role_visible, - "user-authored ex:role must remain visible under wildcard hydration: {node:#?}" + let edge = json!({ + "@context": ctx(), + "select": {"ex:alice": ["*"]} + }); + let rows = support::query_jsonld_formatted(&fluree, &committed.ledger, &edge) + .await + .expect("hydration of the annotated edge"); + assert_eq!( + rows[0]["ex:worksFor"]["@annotation"], + json!({"@id": "ex:emp/alice-acme", "ex:role": "Engineer"}), + "the annotation body leaves out the link: {rows:#?}" ); - - // The link may not appear under any namespace form (full IRI or - // compact alias). - for key in node.keys() { - assert!( - !key.contains("reifies"), - "the link '{key}' must not leak through wildcard hydration" - ); - } } #[tokio::test] @@ -6185,10 +6078,18 @@ async fn annotation_matrix_survivors(fluree: &MemoryFluree, ledger_id: &str) -> }; Survivors { base: count("PREFIX : SELECT ?o WHERE { :alice :knows ?o }").await, - claim1_body: count("PREFIX : SELECT ?p ?v WHERE { :claim1 ?p ?v }") - .await, - claim2_body: count("PREFIX : SELECT ?p ?v WHERE { :claim2 ?p ?v }") - .await, + claim1_body: count( + "PREFIX : \ + SELECT ?p ?v WHERE { :claim1 ?p ?v \ + FILTER(?p != ) }", + ) + .await, + claim2_body: count( + "PREFIX : \ + SELECT ?p ?v WHERE { :claim2 ?p ?v \ + FILTER(?p != ) }", + ) + .await, attached: count( "PREFIX : \ SELECT ?c WHERE { :alice :knows :bob ~ ?c {| :confidence ?f |} }", @@ -6469,3 +6370,97 @@ async fn base_edge_retraction_detaches_sibling_reifiers_the_delete_never_named() docs must warn about" ); } + +/// Wildcard reads of a legacy bundle show the link derived from it, never +/// the bundle — from novelty, from an index, and from novelty over an index +/// that has not seen the bundle's predicates — and a whole-graph count agrees +/// with the rows the scan returns. +#[tokio::test] +async fn variable_predicate_scan_hides_legacy_bundles() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/edge-annotations:legacy-bundle-scan"; + let ledger = support::commit_legacy_bundle(&fluree, genesis_ledger(&fluree, ledger_id)).await; + + let check = |ledger: MemoryLedger, label: &'static str, triples: usize| { + let fluree = &fluree; + async move { + let reifier = json!({ + "@context": ctx(), + "select": ["?p"], + "where": { "@id": "ex:emp/alice-acme", "?p": "?o" } + }); + let rows = support::query_jsonld_formatted(fluree, &ledger, &reifier) + .await + .expect("reifier scan"); + let mut predicates = predicate_column(&rows); + predicates.sort(); + assert!( + predicates.len() == 2 + && predicates.iter().any(|p| is_link_key(p)) + && predicates + .iter() + .any(|p| p == "ex:role" || p == "http://example.org/role"), + "[{label}] the scan shows the derived link and the body: {predicates:?}" + ); + + let all = support::query_sparql_formatted( + fluree, + &ledger, + "SELECT ?s ?p ?o WHERE { ?s ?p ?o }", + ) + .await + .expect("whole-graph scan"); + let rows = all.as_array().expect("rows").len(); + let counted = support::query_sparql_formatted( + fluree, + &ledger, + "SELECT (COUNT(*) AS ?n) WHERE { ?s ?p ?o }", + ) + .await + .expect("whole-graph count"); + let counted = counted[0].as_array().map_or(&counted[0], |row| &row[0]); + assert_eq!(rows, triples, "[{label}] base edge, link, body: {all:#}"); + assert_eq!( + counted, + &json!(triples), + "[{label}] the count agrees with the scan" + ); + + let hydrated = support::query_jsonld_formatted( + fluree, + &ledger, + &json!({ + "@context": ctx(), + "select": {"?ann": ["*"]}, + "where": { "@id": "?ann", "ex:role": "Engineer" } + }), + ) + .await + .expect("wildcard hydration"); + let keys: Vec<&String> = hydrated[0].as_object().expect("node").keys().collect(); + assert!( + keys.iter().any(|k| is_link_key(k)) + && !keys.iter().any(|k| k.contains("reifiesSubject")), + "[{label}] hydration shows the link, not the bundle: {keys:?}" + ); + } + }; + + check(ledger, "novelty", 3).await; + support::rebuild_and_publish_index(&fluree, ledger_id).await; + check(fluree.ledger(ledger_id).await.expect("load"), "indexed", 3).await; + + let ledger_id = "it/edge-annotations:legacy-bundle-over-index"; + let plain = fluree + .insert( + genesis_ledger(&fluree, ledger_id), + &json!({"@context": ctx(), "@id": "ex:x", "ex:y": {"@id": "ex:z"}}), + ) + .await + .expect("plain insert"); + drop(plain); + support::rebuild_and_publish_index(&fluree, ledger_id).await; + let indexed = fluree.ledger(ledger_id).await.expect("load"); + let ledger = support::commit_legacy_bundle(&fluree, indexed).await; + check(ledger, "novelty over an index", 4).await; +} diff --git a/fluree-db-api/tests/it_scan_count_drain.rs b/fluree-db-api/tests/it_scan_count_drain.rs new file mode 100644 index 0000000000..71ec4894bb --- /dev/null +++ b/fluree-db-api/tests/it_scan_count_drain.rs @@ -0,0 +1,96 @@ +//! Routing for a variable-predicate `COUNT(*)` under novelty: the scan +//! counts its cursor without binding rows unless the ledger holds legacy +//! `f:reifies*` bundles. +//! +//! The only test in its own `[[test]]` binary, as the other routing-stamp +//! tests are: tracing's callsite interest and the `set_default` capture are +//! process- and thread-shared, and a bundled sibling that installs a global +//! subscriber leaves this capture empty. +#![cfg(feature = "native")] + +mod support; +use fluree_db_api::FlureeBuilder; +use serde_json::{json, Value as JsonValue}; +use support::genesis_ledger; + +fn ctx() -> JsonValue { + json!({"ex": "http://example.org/"}) +} + +/// The `scan-count-drain` outcomes `query` stamps on `ledger`. +async fn scan_count_drain_outcomes( + fluree: &fluree_db_api::Fluree, + ledger: &fluree_db_api::LedgerState, + query: &str, +) -> (JsonValue, Vec) { + // Register the stamp callsite before this thread's subscriber reads it. + let _ = support::query_sparql_formatted(fluree, ledger, query).await; + let (store, _guard) = support::span_capture::init_test_tracing(); + tracing::callsite::rebuild_interest_cache(); + let result = support::query_sparql_formatted(fluree, ledger, query) + .await + .expect("count"); + let outcomes = store + .find_events("fast-path outcome") + .iter() + .filter(|e| e.fields.get("site").map(String::as_str) == Some("scan-count-drain")) + .filter_map(|e| e.fields.get("outcome").cloned()) + .collect(); + (result, outcomes) +} + +/// A variable-predicate `COUNT(*)` under novelty counts the scan's cursor +/// without binding a row unless the ledger holds legacy bundles, whose rows +/// the scan has to drop one by one. The index knows `ex:worksFor`, so the +/// novelty link's term translates. +#[tokio::test(flavor = "current_thread")] +async fn variable_predicate_count_counts_the_cursor_without_legacy_bundles() { + let fluree = FlureeBuilder::memory().build_memory(); + let count = "SELECT (COUNT(*) AS ?n) WHERE { ?s ?p ?o }"; + + let ledger_id = "it/edge-annotations:count-drain"; + let indexed = fluree + .insert( + genesis_ledger(&fluree, ledger_id), + &json!({"@context": ctx(), "@id": "ex:x", "ex:worksFor": {"@id": "ex:z"}}), + ) + .await + .expect("plain insert"); + drop(indexed); + support::rebuild_and_publish_index(&fluree, ledger_id).await; + let ledger = fluree + .insert( + fluree.ledger(ledger_id).await.expect("load"), + &json!({ + "@context": ctx(), + "@id": "ex:alice", + "ex:worksFor": {"@id": "ex:acme", "@annotation": {"ex:role": "Engineer"}} + }), + ) + .await + .expect("annotated insert into novelty") + .ledger; + let (result, outcomes) = scan_count_drain_outcomes(&fluree, &ledger, count).await; + assert_eq!(result, json!([[4]]), "plain edge, base edge, link, body"); + assert_eq!(outcomes, ["proceed"], "{outcomes:?}"); + + let ledger_id = "it/edge-annotations:count-drain-legacy"; + let indexed = fluree + .insert( + genesis_ledger(&fluree, ledger_id), + &json!({"@context": ctx(), "@id": "ex:x", "ex:worksFor": {"@id": "ex:z"}}), + ) + .await + .expect("plain insert"); + drop(indexed); + support::rebuild_and_publish_index(&fluree, ledger_id).await; + let ledger = + support::commit_legacy_bundle(&fluree, fluree.ledger(ledger_id).await.expect("load")).await; + let (result, outcomes) = scan_count_drain_outcomes(&fluree, &ledger, count).await; + assert_eq!( + result, + json!([[4]]), + "plain edge, base edge, derived link, body" + ); + assert_eq!(outcomes, ["fallback:gate_declined"], "{outcomes:?}"); +} diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 0f1b836e63..f76efe8f6c 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -104,8 +104,7 @@ async fn imported_reifiers_carry_a_decodable_triple_term_link() { "claim2 must reify carol age 42 (literal object): {t2}" ); - // While the link form is index-internal, wildcard scans keep hiding it - // exactly as they hide the `f:reifies*` bundle. + // The link is ordinary data to a wildcard scan. let sparql = "PREFIX ex: \n\ SELECT ?p WHERE { ex:claim1 ?p ?o } ORDER BY ?p"; let result = support::query_sparql_formatted(&fluree, &ledger, sparql) @@ -113,8 +112,10 @@ async fn imported_reifiers_carry_a_decodable_triple_term_link() { .expect("wildcard predicate scan"); let preds: Vec = rows(&result).into_iter().map(|r| r[0].clone()).collect(); assert!( - !preds.iter().any(|p| p.contains("reifies")), - "wildcard scan must hide rdf:reifies and f:reifies*: {preds:?}" + preds + .iter() + .any(|p| p.ends_with("#reifies") || p == "rdf:reifies"), + "the wildcard scan returns the link: {preds:?}" ); assert!( preds.iter().any(|p| p.ends_with("confidence")), diff --git a/fluree-db-api/tests/support/mod.rs b/fluree-db-api/tests/support/mod.rs index a13d55398c..0dd47beb0e 100644 --- a/fluree-db-api/tests/support/mod.rs +++ b/fluree-db-api/tests/support/mod.rs @@ -895,3 +895,76 @@ pub async fn run_collector_and_fork_drop_scenario( round(main_id.clone(), 7).await; assert_dicts_present(main_id.clone(), "after a build following the drop").await; } + +/// Commits `ex:alice ex:worksFor ex:acme`, annotated the way releases before +/// `rdf:reifies` links stored it: an `f:reifies*` bundle on +/// `ex:emp/alice-acme`, beside the body `ex:role "Engineer"`. No writer +/// produces bundles any more. +pub async fn commit_legacy_bundle( + fluree: &fluree_db_api::Fluree, + ledger: LedgerState, +) -> LedgerState { + use fluree_db_core::namespaces::{ + reifies_object_sid, reifies_predicate_sid, reifies_subject_sid, + }; + use fluree_db_core::{Flake, FlakeValue, Sid}; + + let mut ns = fluree_db_transact::NamespaceRegistry::from_db(&ledger.snapshot); + let mut ex = |name: &str| ns.sid_for_iri(&format!("http://example.org/{name}")); + let (alice, works_for, acme, ann, role) = ( + ex("alice"), + ex("worksFor"), + ex("acme"), + ex("emp/alice-acme"), + ex("role"), + ); + let t = ledger.t() + 1; + let id_dt = fluree_db_core::edge::id_datatype_sid(); + let flake = |s: &Sid, p: &Sid, o: FlakeValue, dt: Sid| { + Flake::new(s.clone(), p.clone(), o, dt, t, true, None) + }; + let flakes = vec![ + flake( + &alice, + &works_for, + FlakeValue::Ref(acme.clone()), + id_dt.clone(), + ), + flake( + &ann, + reifies_subject_sid(), + FlakeValue::Ref(alice.clone()), + id_dt.clone(), + ), + flake( + &ann, + reifies_predicate_sid(), + FlakeValue::Ref(works_for.clone()), + id_dt.clone(), + ), + flake(&ann, reifies_object_sid(), FlakeValue::Ref(acme), id_dt), + flake( + &ann, + &role, + FlakeValue::String("Engineer".into()), + fluree_db_core::edge::xsd_string_datatype_sid(), + ), + ]; + let view = + fluree_db_transact::stage_flakes(ledger, flakes, fluree_db_transact::StageOptions::new()) + .await + .expect("stage the legacy bundle"); + fluree + .commit_staged( + view, + ns, + &fluree_db_ledger::IndexConfig { + reindex_min_bytes: 100_000, + reindex_max_bytes: 1_000_000_000, + }, + fluree_db_transact::CommitOpts::default(), + ) + .await + .expect("commit the legacy bundle") + .1 +} diff --git a/fluree-db-binary-index/src/read/binary_index_store.rs b/fluree-db-binary-index/src/read/binary_index_store.rs index 265cd5badf..f13c5c9963 100644 --- a/fluree-db-binary-index/src/read/binary_index_store.rs +++ b/fluree-db-binary-index/src/read/binary_index_store.rs @@ -327,6 +327,7 @@ pub struct BinaryIndexStore { /// configuration (`set_ns_split_mode`, namespace augmentation), which /// cannot occur once the store is behind `Arc`. p_sid_table: std::sync::OnceLock>, + scan_hidden_p_ids: std::sync::OnceLock>, /// Conclusive per-`(graph, predicate)` decimal-only proofs. Index contents /// are immutable per store, so a proof holds for the store's lifetime. decimal_only_proofs: RwLock>, @@ -397,6 +398,7 @@ impl BinaryIndexStore { ns_split_mode: NsSplitMode::default(), ns_split_mode_set: true, p_sid_table: std::sync::OnceLock::new(), + scan_hidden_p_ids: std::sync::OnceLock::new(), decimal_only_proofs: RwLock::new(HashMap::new()), } } @@ -579,6 +581,7 @@ impl BinaryIndexStore { ns_split_mode: root.ns_split_mode, ns_split_mode_set: true, p_sid_table: std::sync::OnceLock::new(), + scan_hidden_p_ids: std::sync::OnceLock::new(), decimal_only_proofs: RwLock::new(HashMap::new()), }) } @@ -1896,6 +1899,19 @@ impl BinaryIndexStore { }) } + /// Persisted p_ids of the predicates wildcard scans hide + /// ([`fluree_db_core::is_scan_hidden_predicate`]). Empty unless the + /// ledger's history holds legacy annotation bundles, so a + /// variable-predicate scan of any other ledger checks no row. + pub fn scan_hidden_p_ids(&self) -> &[u32] { + self.scan_hidden_p_ids.get_or_init(|| { + fluree_vocab::reifies_iris::ALL + .iter() + .filter_map(|iri| self.find_predicate_id(iri)) + .collect() + }) + } + /// Lookup a predicate IRI → p_id. pub fn find_predicate_id(&self, iri: &str) -> Option { self.dicts.predicate_reverse.get(iri).copied() diff --git a/fluree-db-core/src/namespaces.rs b/fluree-db-core/src/namespaces.rs index a4f399d358..3a899af199 100644 --- a/fluree-db-core/src/namespaces.rs +++ b/fluree-db-core/src/namespaces.rs @@ -399,13 +399,13 @@ pub fn is_annotation_predicate(sid: &Sid) -> bool { is_rdf_reifies(sid) || is_reserved_reifies_predicate(sid) } -/// True for the predicates wildcard scans hide from users: the seven -/// `f:reifies*` bundle predicates and, while the RDF 1.2 link form is -/// index-internal, `rdf:reifies`. Read-side only; the write firewall is -/// [`is_reserved_reifies_predicate`]. +/// True for the predicates wildcard scans hide from users: the seven legacy +/// `f:reifies*` bundle predicates, an internal encoding whose links are read +/// instead. `rdf:reifies` is ordinary data. Read-side only; the write +/// firewall is [`is_reserved_reifies_predicate`]. #[inline] pub fn is_scan_hidden_predicate(sid: &Sid) -> bool { - is_reserved_reifies_predicate(sid) || is_rdf_reifies(sid) + is_reserved_reifies_predicate(sid) } /// The cached `f:tripleTerm` datatype SID carried by a triple-term object, diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index 4aee635163..e44e0658c0 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -201,6 +201,10 @@ pub struct BinaryScanOperator { /// Novelty-only predicate overrides keyed by ephemeral p_id, populated /// during overlay translation (ephemeral ids sit above the persisted range). p_sids_ephemeral: HashMap, + /// p_ids a variable-predicate scan hides (see `is_internal_predicate`). + /// Empty for a bound predicate, under `include_system_facts`, and on + /// ledgers without legacy annotation bundles. + hidden_p_ids: Vec, /// Cached s_id → Sid for amortized IRI resolution. sid_cache: HashMap, /// Whether predicate is a variable (for internal predicate filtering). @@ -761,6 +765,7 @@ impl BinaryScanOperator { pending_cursors: VecDeque::new(), p_sids: Vec::new().into(), p_sids_ephemeral: HashMap::new(), + hidden_p_ids: Vec::new(), sid_cache: HashMap::new(), p_is_var, include_system_facts: false, @@ -1198,40 +1203,22 @@ impl BinaryScanOperator { .unwrap_or_else(|| Sid::new(0, "")) } - /// Filter: skip `f:reifies*` system predicates when predicate is a - /// variable. + /// Filter: skip the legacy `f:reifies*` bundle predicates when the + /// predicate is a variable. /// - /// The seven edge-annotation bundle predicates are hidden in - /// **every** graph context: the annotation bundle is emitted in the - /// reified edge's graph, so a per-graph `?p` scan would otherwise - /// leak the bundle into the user's results. They are the internal - /// encoding of `@annotation` — user transactions cannot write them, - /// so surfacing them would break export/re-import round-trips. + /// They are the internal encoding of annotations written before + /// `rdf:reifies` links, hidden in **every** graph context since a bundle + /// lives in its reified edge's graph. User transactions cannot write + /// them, so surfacing them would break export/re-import round-trips. /// /// Other Fluree-namespace predicates are NOT filtered: system /// metadata lives in its own graphs (txn-meta at `g_id == 1`, /// config at `g_id == 2`), and `f:`-vocabulary data users author in /// the default graph (e.g. `f:AccessPolicy` definitions) must stay - /// visible to wildcard scans. A default-graph namespace-wide hide - /// existed historically, from the era when commit metadata was - /// stored in the main graph. + /// visible to wildcard scans. #[inline] fn is_internal_predicate(&self, p_id: u32) -> bool { - if !self.p_is_var { - return false; - } - // Opt-in escape for debug / inspection workflows. - if self.include_system_facts { - return false; - } - let sid = match self.p_sids_ephemeral.get(&p_id) { - Some(sid) => sid, - None => match self.p_sids.get(p_id as usize) { - Some(sid) => sid, - None => return false, - }, - }; - fluree_db_core::is_scan_hidden_predicate(sid) + !self.hidden_p_ids.is_empty() && self.hidden_p_ids.contains(&p_id) } /// Enforce within-pattern repeated-variable constraints. @@ -1345,7 +1332,7 @@ impl BinaryScanOperator { /// can drop or transform a row after the cursor yields it, and there are no /// overlay-only fallback rows. Each clause below maps to exactly one /// `continue`/drop site in `batch_to_bindings`: - /// - bound predicate (`!p_is_var`) → `is_internal_predicate` never skips + /// - no hidden predicates (`hidden_p_ids`) → `is_internal_predicate` never skips /// - no encoded pre-filters, datatype constraint, repeated-var checks, /// bound object, object bounds, or unresolved-bound-subject IRI check /// - no inline ops (which may carry FILTER/BIND that drop rows) @@ -1357,7 +1344,7 @@ impl BinaryScanOperator { /// returns `Ok(None)` and the caller falls back to the streaming drain. fn count_only_eligible(&self) -> bool { matches!(self.mode, crate::temporal_mode::TemporalMode::Current) - && !self.p_is_var + && self.hidden_p_ids.is_empty() && self.inline_ops.is_empty() && self.encoded_pre_filters.is_empty() && self.object_bounds.is_none() @@ -2177,6 +2164,11 @@ impl Operator for BinaryScanOperator { let store_ref = store.as_ref(); // Persisted p_id → Sid table, built once per store instance. self.p_sids = Arc::clone(store_ref.p_sid_table()); + self.hidden_p_ids.clear(); + if self.p_is_var && !self.include_system_facts { + self.hidden_p_ids + .extend_from_slice(store_ref.scan_hidden_p_ids()); + } // Extract bound terms in snapshot namespace space and build the persisted-ID filter // by translating through full IRIs into store namespace space. @@ -2974,6 +2966,12 @@ impl Operator for BinaryScanOperator { // Record novelty-only predicates so that ephemeral p_ids from // overlay ops can be decoded back to Sids during row binding. for (sid, ep_id) in &translated.ephemeral_preds { + if self.p_is_var + && !self.include_system_facts + && fluree_db_core::is_scan_hidden_predicate(sid) + { + self.hidden_p_ids.push(*ep_id); + } self.p_sids_ephemeral.insert(*ep_id, sid.clone()); } @@ -3174,8 +3172,10 @@ impl Operator for BinaryScanOperator { return Ok(None); } if !self.count_only_eligible() { + stamp_scan_count_drain(false); return Ok(None); } + stamp_scan_count_drain(true); // Residency mode: handle a store's ContentStore + retry budget for // the drain/fetch/retry arm in the cursor loop below. @@ -3230,6 +3230,7 @@ impl Operator for BinaryScanOperator { self.sid_cache.clear(); self.p_sids = Vec::new().into(); self.p_sids_ephemeral.clear(); + self.hidden_p_ids.clear(); self.unresolved_bound_subject_iri = None; self.state = OperatorState::Closed; } @@ -4294,6 +4295,21 @@ fn datatype_sid_for_untyped_value( } } +/// Routing stamp for `drain_count`: `proceed` when it counts the cursor +/// without binding rows, `fallback:gate_declined` when a per-row check +/// (`count_only_eligible`) sends `COUNT(*)` to the streaming drain. +fn stamp_scan_count_drain(counts: bool) { + use crate::fast_path_outcome::{stamp_fast_path, FastPathFallback, FastPathOutcome}; + stamp_fast_path( + "scan-count-drain", + if counts { + FastPathOutcome::Proceed + } else { + FastPathOutcome::Fallback(FastPathFallback::GateDeclined) + }, + ); +} + /// Routing stamp for a bare-number object: `proceed` when the scan seeks its /// slices, `fallback:gate_declined` when it walks unnarrowed. const BARE_NUMBER_SEEK_SITE: &str = "bare_number_seek"; diff --git a/fluree-db-query/src/eval/metadata.rs b/fluree-db-query/src/eval/metadata.rs index 0507cedd98..c42d640b02 100644 --- a/fluree-db-query/src/eval/metadata.rs +++ b/fluree-db-query/src/eval/metadata.rs @@ -519,10 +519,10 @@ fn data_properties_from_flakes(mut flakes: Vec) -> Vec { let mut live: Vec = Vec::new(); for flake in flakes { // Data properties only: skip references (relationships), rdf:type, and - // the reifier sidecar. + // a reifier's link, which is the relationship itself. if matches!(flake.o, FlakeValue::Ref(_)) || fluree_db_core::is_rdf_type(&flake.p) - || fluree_db_core::is_scan_hidden_predicate(&flake.p) + || fluree_db_core::is_annotation_predicate(&flake.p) { continue; } diff --git a/fluree-db-query/src/fast_count.rs b/fluree-db-query/src/fast_count.rs index bbddd92676..c865620f5a 100644 --- a/fluree-db-query/src/fast_count.rs +++ b/fluree-db-query/src/fast_count.rs @@ -1025,6 +1025,11 @@ pub fn count_triples_operator(out_var: VarId, fallback: Option) - let Some(store) = fast_path_store(ctx) else { return Ok(None); }; + // Leaf row counts include the legacy `f:reifies*` rows the + // variable-predicate scan hides. + if crate::fast_whole_graph_agg::graph_has_scan_hidden_predicates(ctx, store)? { + return Ok(None); + } let count = count_triples_from_branch_manifest(store, ctx.binary_g_id)?; let count_i64 = count_to_i64(count, "COUNT triples")?; Ok(Some(build_count_batch(out_var, count_i64)?)) @@ -1082,12 +1087,10 @@ pub fn count_distinct_position_operator( let Some(store) = fast_path_store(ctx) else { return Ok(None); }; - // The variable-predicate scan hides `f:reifies*` (and default-graph - // `f:`) facts, but the SPOT/PSOT/OPST directories these folds read - // include them — a graph with a reified edge would over-count in - // every position (its reifier subject, the reifies predicates, and - // the annotation objects are all pipeline-invisible). Decline to - // the general pipeline when any such predicate exists. + // The variable-predicate scan hides legacy `f:reifies*` facts, but + // the SPOT/PSOT/OPST directories these folds read include them — a + // graph holding a legacy bundle would over-count in every position. + // Decline to the general pipeline when any such predicate has rows. if crate::fast_whole_graph_agg::graph_has_scan_hidden_predicates(ctx, store)? { return Ok(None); } diff --git a/fluree-db-query/src/fast_whole_graph_agg.rs b/fluree-db-query/src/fast_whole_graph_agg.rs index a4afc842e3..7bc2823de3 100644 --- a/fluree-db-query/src/fast_whole_graph_agg.rs +++ b/fluree-db-query/src/fast_whole_graph_agg.rs @@ -628,9 +628,9 @@ fn overlay_all_subjects_count( if declined { return; } - // Mirror the pipeline's `?n ?p ?o` visibility: `f:reifies*` is - // invisible to the scan but present in SPOT — its subjects must - // not be counted. + // Mirror the pipeline's `?n ?p ?o` visibility: legacy + // `f:reifies*` is invisible to the scan but present in SPOT — its + // subjects must not be counted. if fluree_db_core::is_scan_hidden_predicate(&flake.p) { declined = true; return; @@ -1067,27 +1067,22 @@ fn compute_task( /// reads, so their presence makes the fold inexact. /// /// Two deliberate choices: -/// - Candidates come from the store's predicate **dictionary**, not the -/// per-graph stats: the incremental index build can persist delta-only -/// per-graph property stats (base entries lost), so a stats-driven check -/// can silently pass on a graph that does carry hidden facts. +/// - Candidates come from the store's predicate **dictionary** +/// (`scan_hidden_p_ids`), not the per-graph stats: the incremental index +/// build can persist delta-only per-graph property stats (base entries +/// lost), so a stats-driven check can silently pass on a graph that does +/// carry hidden facts. /// - Each hidden candidate is confirmed by its **row count in the queried /// graph** (`count_rows_for_predicate_psot`, directory-only): the /// dictionary is global across graphs, so a `f:reifies*` predicate minted /// by annotations in another graph is always present in it — bare -/// membership would permanently decline every annotated ledger. +/// membership would decline every graph of the ledger. pub(crate) fn graph_has_scan_hidden_predicates( ctx: &crate::context::ExecutionContext<'_>, store: &BinaryIndexStore, ) -> Result { - for p_id in 0..store.predicate_count() { - let Some(sid) = store.predicate_sid(p_id) else { - // Unresolvable dictionary entry — err toward declining. - return Ok(true); - }; - if fluree_db_core::is_scan_hidden_predicate(&sid) - && count_rows_for_predicate_psot(store, ctx.binary_g_id, p_id)? > 0 - { + for &p_id in store.scan_hidden_p_ids() { + if count_rows_for_predicate_psot(store, ctx.binary_g_id, p_id)? > 0 { return Ok(true); } } diff --git a/testsuite-sparql/tests/registers/mod.rs b/testsuite-sparql/tests/registers/mod.rs index 8a798a545a..94f1839305 100644 --- a/testsuite-sparql/tests/registers/mod.rs +++ b/testsuite-sparql/tests/registers/mod.rs @@ -409,22 +409,14 @@ pub const SPARQL12_EVAL_TRIPLE_TERMS: &[&str] = &[ "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#order-1", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#order-2", // data load: blocked on TriG GRAPH-block parsing, orthogonal to star - // (D-8) — "expected subject, found 'GRAPH'" (5) + // (D-8) — "expected subject, found 'GRAPH'" (4) "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#expr-1", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#graphs-1", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#graphs-2", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#update-1", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#update-2", - // results not isomorphic: the star data now LOADS (the collector sink - // accepts `<< >>` / `{| |}` / `~`), but Fluree's model asserts the base - // triple of every `<< s p o >>` and binds reifiers, not triple terms — - // RDF 1.2's non-asserting reified triples and `?t` triple-term bindings - // are the Option-1 epic (8) - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#basic-2", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#basic-3", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#basic-7", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#pattern-1", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#pattern-2", + // results not isomorphic: the star data LOADS, but Fluree's model + // asserts the base triple of every `<< s p o >>` — RDF 1.2's + // non-asserting reified triples (3) "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#pattern-8-nomatch", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#results-reifiedtriples-1j", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#results-reifiedtriples-1x", From 653f66c1ce5e2d0d6a24acc4960838f0a5ec93a1 Mon Sep 17 00:00:00 2001 From: bplatz Date: Sat, 3 Oct 2026 09:15:03 -0400 Subject: [PATCH 56/92] feat: reified triples no longer assert their triple (RDF 1.2) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit As RDF 1.2 defines them, only the annotation syntax (`s p o ~ r`, `s p o {| … |}`) asserts the triple it reifies. `<< s p o >>` and `r rdf:reifies <<( s p o )>>` now attach the reifier without asserting `s p o` on every Turtle, TriG and N-Quads write path. - JSON-LD: a node's `@reifies` (`{"@id": s, p: o}`, or an array) makes it a reifier of that triple without asserting it. The Turtle→JSON-LD conversion behind upsert and graph sync writes unasserted reifications this way; it used to drop them in release builds. - Queries: the annotation syntax (SPARQL `{| |}`/`~`, JSON-LD `@annotation`, Cypher relationship properties) joins the base edge again, last in the chain so it is a bound existence probe where estimates tie. Reified-triple patterns read the link alone. - Export writes a link whose triple its graph does not assert as the link itself (`@reifies` in JSON-LD) instead of dropping it. - CONSTRUCT: `?r rdf:reifies <<( … )>>` and `?r rdf:reifies ?t` write the reification without the triple; the JSON-LD formatter writes such a reification as `@reifies` rather than dropping it. - The W3C harness compares CONSTRUCT reifications on both sides; four more SPARQL 1.2 evaluation tests pass. --- docs/concepts/edge-annotations.md | 22 ++- docs/design/edge-annotations.md | 12 +- docs/guides/cookbook-edge-annotations.md | 6 +- docs/query/construct.md | 8 +- docs/reference/compatibility.md | 6 +- docs/transactions/insert.md | 5 +- docs/transactions/turtle.md | 2 +- fluree-db-api/src/export.rs | 145 +++++++++++--- fluree-db-api/src/export_annotations.rs | 92 +++++++-- fluree-db-api/src/format/construct.rs | 23 ++- .../tests/it_edge_annotations_parse.rs | 71 ++++++- fluree-db-api/tests/it_import_turtle_star.rs | 9 +- fluree-db-api/tests/it_query_construct.rs | 13 +- fluree-db-api/tests/it_triple_term_links.rs | 18 +- .../tests/it_turtle_star_write_paths.rs | 47 ++++- fluree-db-cli/tests/integration.rs | 86 +++++++-- fluree-db-query/src/execute/where_plan.rs | 32 ++-- fluree-db-query/src/ir/query.rs | 17 ++ fluree-db-sparql/src/lower/construct.rs | 4 +- .../src/parse/edge_annotations.rs | 179 ++++++++++-------- fluree-db-transact/src/parse/trig_meta.rs | 55 +++--- fluree-graph-format/src/jsonld.rs | 29 +++ fluree-graph-ir/src/sink.rs | 8 +- fluree-graph-turtle/src/adapter.rs | 63 ++++-- fluree-graph-turtle/src/parser.rs | 51 +++-- fluree-graph-turtle/src/splitter.rs | 2 + testsuite-sparql/src/rdf_handlers.rs | 20 +- testsuite-sparql/src/result_format.rs | 109 +++++++++-- testsuite-sparql/tests/registers/mod.rs | 18 +- 29 files changed, 824 insertions(+), 328 deletions(-) diff --git a/docs/concepts/edge-annotations.md b/docs/concepts/edge-annotations.md index 12683b9bc8..b41ab24782 100644 --- a/docs/concepts/edge-annotations.md +++ b/docs/concepts/edge-annotations.md @@ -25,9 +25,9 @@ If a fact is naturally about a *node* (Alice's birthdate, Acme's industry), put | Surface | How | Notes | |---|---|---| -| **JSON-LD insert / upsert / update** | `@annotation` (or alias `@edge`) on a value object | Most ergonomic. Covers literal-valued edges (with explicit `@type` / `@language`), parallel annotations, named reifiers, body cascades, and **named-graph edges** (an annotation on an edge inside a named graph is written into that same graph, keeping the edge's graph identity). `@reifies` is a query-side construct, **not** an insert form (user-authored `@reifies` on a write is rejected). | +| **JSON-LD insert / upsert / update** | `@annotation` (or alias `@edge`) on a value object | Most ergonomic. Covers literal-valued edges (with explicit `@type` / `@language`), parallel annotations, named reifiers, body cascades, and **named-graph edges** (an annotation on an edge inside a named graph is written into that same graph, keeping the edge's graph identity). A node's `@reifies` (`{"@id": s, p: o}`, or an array of them) makes the node a reifier of that triple **without asserting it**, as `r rdf:reifies <<( s p o )>>` does. | | **SPARQL 1.2 UPDATE** | `INSERT DATA { :s :p :o {\| ... \|} }`, `~ `, optional `INSERT { } WHERE { }` templates | Use this when integrating with SPARQL pipelines or when porting from RDF 1.2 / SPARQL-star. **Default graph only:** an annotation tail inside an explicit `GRAPH { }` block, or under a `WITH ` template, is rejected — use the JSON-LD surface or TriG-star for named-graph edge annotations. See [SPARQL 1.2 surface](#sparql-12--rdf-12-surface) below for the per-operation rules. | -| **Turtle / N-Triples / TriG / N-Quads ingest** (`insert`, `upsert`, bulk `import`, graph sync (`fluree sync`, `/sync`), memory import; TriG via `insert` / `upsert` / `import` / `fluree sync` / `/sync`) | RDF 1.2 asserting forms: `:s :p :o ~ {\| ... \|}`, `<< :s :p :o ~ :r >>` in subject or object position, and the canonical `:r rdf:reifies <<( :s :p :o )>>` (the only star spelling N-Triples and N-Quads have) — in the default graph and inside TriG `GRAPH { }` blocks alike | Same stored link as `@annotation`. An annotation inside a `GRAPH { }` block, or on an N-Quads statement with a graph label, is written into that graph with the edge's graph identity, exactly as JSON-LD `@graph` + `@annotation` does. Every form asserts the base edge (Fluree reifies asserted edges, so `<< s p o >>` and `rdf:reifies <<( s p o )>>` are asserting here). Rejected with a specific error: `<<( ... )>>` anywhere other than the object of `rdf:reifies`, nested triple terms, star constructs inside an annotation body, annotations on collections, and annotations in a TriG `<#txn-meta>` block. Paths that convert to JSON-LD first (`upsert`, `graph sync`, memory import) also reject an annotation on an `rdf:type` edge. See [Turtle ingest](../transactions/turtle.md#edge-annotations-rdf-12--turtle-star). | +| **Turtle / N-Triples / TriG / N-Quads ingest** (`insert`, `upsert`, bulk `import`, graph sync (`fluree sync`, `/sync`), memory import; TriG via `insert` / `upsert` / `import` / `fluree sync` / `/sync`) | RDF 1.2 forms: the annotation syntax `:s :p :o ~ {\| ... \|}`, `<< :s :p :o ~ :r >>` in subject or object position, and the canonical `:r rdf:reifies <<( :s :p :o )>>` (the only star spelling N-Triples and N-Quads have) — in the default graph and inside TriG `GRAPH { }` blocks alike | Same stored link as `@annotation`. An annotation inside a `GRAPH { }` block, or on an N-Quads statement with a graph label, is written into that graph with the edge's graph identity, exactly as JSON-LD `@graph` + `@annotation` does. As in RDF 1.2, only the annotation syntax asserts the triple: `<< s p o >>` and `rdf:reifies <<( s p o )>>` reify it without asserting it, so N-Triples and N-Quads state an annotated triple as the triple plus its reifier's link. Rejected with a specific error: `<<( ... )>>` anywhere other than the object of `rdf:reifies`, nested triple terms, star constructs inside an annotation body, annotations on collections, and annotations in a TriG `<#txn-meta>` block. Paths that convert to JSON-LD first (`upsert`, `graph sync`, memory import) also reject an annotation on an `rdf:type` edge. See [Turtle ingest](../transactions/turtle.md#edge-annotations-rdf-12--turtle-star). | Mint annotations through `@annotation` / `@edge` (JSON-LD) or the RDF 1.2 forms (`~`, `{| |}`, `<< >>`, `rdf:reifies <<( )>>`) in SPARQL UPDATE and Turtle. The [`f:reifies*` predicates](../reference/vocabulary.md#edge-annotation-predicates-reserved) earlier releases stored annotations under are reserved, and the write surfaces reject them; bulk import reads them, as an export written before links carries them, and stores each annotation's link instead. @@ -395,7 +395,7 @@ In LPG mode, an empty block mints a fresh annotation subject — a property-less ## SPARQL 1.2 / RDF 1.2 surface -`@annotation` lowers to the same on-disk model as the RDF 1.2 annotation tail. The equivalent **write** forms below produce identical storage; pick whichever is ergonomic for your input. (`@reifies` / `rdf:reifies` are **query-side** constructs — see [Querying annotation-rooted](#querying-annotation-rooted-metadata-first-edge-second) — and are rejected on the write side.) +`@annotation` lowers to the same on-disk model as the RDF 1.2 annotation tail. The equivalent **write** forms below produce identical storage; pick whichever is ergonomic for your input. Each asserts the triple and links its reifier to it; `@reifies` and `rdf:reifies <<( … )>>` write the link alone (see [Reifiers of unasserted triples](#reifiers-of-unasserted-triples)). ### Equivalent forms @@ -523,9 +523,19 @@ Anonymous annotation blocks (`{| |}` without `~`) lower to a fresh non-distingui As in RDF 1.2, an annotation subject may reify several triples: a given `@id` on two edges is linked to both, and its properties describe each. To move an explicit-IRI annotation from one edge to another, retract the old attachment and assert the new one; a JSON-LD upsert of the annotation does that for you. -#### Lifecycle coupling +#### Reifiers of unasserted triples -Fluree's annotation is *lifecycle-coupled* to an asserted edge: the annotation describes a triple that's currently in the graph. RDF 1.2 also allows reifiers for unasserted propositions ("X claims Alice works for Acme without us asserting it"). That mode is **not supported in v1** — see *Current limits* below. +A reifier may describe a triple that is not in the graph ("X claims Alice works for Acme", without asserting it). Write one with a reified triple — Turtle `<< :alice :worksFor :acme ~ :claim >> :source :x`, `:claim rdf:reifies <<( :alice :worksFor :acme )>>`, or JSON-LD: + +```json +{ + "@id": "ex:claim", + "ex:source": {"@id": "ex:x"}, + "@reifies": {"@id": "ex:alice", "ex:worksFor": {"@id": "ex:acme"}} +} +``` + +Reified-triple patterns (`<< :alice :worksFor ?o ~ ?r >>`, `?r rdf:reifies <<( … )>>`, JSON-LD `@reifies`) find such a reifier; the annotation syntax (`:alice :worksFor ?o {| … |}`, JSON-LD `@annotation`, a Cypher relationship) matches only asserted triples, as RDF 1.2 defines it. Export writes the reifier as its link (`@reifies` in JSON-LD), so it round-trips. ### Deferred SPARQL shapes (rejected at parse time) @@ -563,10 +573,8 @@ The bare-quoted-triple form combined with an annotation tail (`<< :s :p :o >> :p Today's surface covers the common LPG / RDF-star use cases. The following are not yet supported and produce a clear validation error rather than silent partial behavior: - **Annotations on list-occurrence triples.** `@list` membership is in scope as a future extension; the on-disk format already reserves space for it. Today, annotating a list element is rejected at parse time. -- **Reifiers for unasserted triples.** `@reifies` must point at an asserted edge. Pure-proposition reification (claims about triples that are not in the graph) is deferred. - **Triple terms as stored values.** `ex:doc ex:mentions <<( ex:s ex:p ex:o )>>` is not a representable value on any write surface (JSON-LD, SPARQL, Turtle): the only accepted position for `<<( ... )>>` is the object of `rdf:reifies`, where it names an edge rather than storing a triple. Use a separate annotation subject. A query can still build and return one: `TRIPLE`, `SUBJECT`, `PREDICATE`, `OBJECT`, `isTRIPLE` and `BIND(<<( ?s ?p ?o )>> AS ?t)` work on terms, and a bound term renders in every result format (see [Output formats](../query/output-formats.md#triple-terms)). JSON-LD queries name them `triple`, `subject`, `predicate`, `object` and `isTriple` (see [JSON-LD query](../query/jsonld-query.md#triple-term-functions)). - **SPARQL UPDATE annotations inside named graphs.** Annotation tails under `GRAPH { }` / `WITH ` in SPARQL UPDATE are rejected; write named-graph annotations with JSON-LD `@annotation` or TriG-star. -- **Unasserted reified triples.** RDF 1.2's `<< s p o >>` and `r rdf:reifies <<( s p o )>>` do not assert `s p o`; Fluree's do (the reifier is lifecycle-coupled to a live edge). A W3C test that depends on a reifier existing for a triple that is *not* in the graph therefore diverges. The mandated SPARQL 1.2 `VERSION "1.2"` prologue declaration is **accepted** (lex-and-skipped): the RDF 1.2 surface runs ungated, so a conformant 1.2 client that emits the declaration parses normally. diff --git a/docs/design/edge-annotations.md b/docs/design/edge-annotations.md index a5750ce311..ddc0cfbf8a 100644 --- a/docs/design/edge-annotations.md +++ b/docs/design/edge-annotations.md @@ -18,12 +18,12 @@ A triple term carries its object's datatype and language tag, so `"chat"@fr` and ## Writers -Every write surface produces the link directly: +Every write surface produces the link directly. As in RDF 1.2, only the annotation syntax (`s p o ~ r`, `s p o {| … |}`) asserts the triple it reifies; a reified triple (`<< s p o >>`, `r rdf:reifies <<( s p o )>>`, JSON-LD `@reifies`) does not. -- **Turtle / TriG / N-Quads** (`FlakeSink`, `ImportSink`, TriG import): `reified_triple_link` builds the link flake for each `~ r`, `{| … |}`, `<< s p o ~ r >>` and `r rdf:reifies <<( s p o )>>`; the parser emits the base triple itself. +- **Turtle / TriG / N-Quads** (`FlakeSink`, `ImportSink`, TriG import): `reified_triple_link` builds the link flake for each `~ r`, `{| … |}`, `<< s p o ~ r >>` and `r rdf:reifies <<( s p o )>>`; the parser emits the base triple itself for the annotation syntax. The Turtle→JSON-LD conversion behind upsert and graph sync writes a reification whose triple the document does not assert as a `@reifies` node. - **SPARQL UPDATE**: `expand_annotated_triples` desugars an annotation tail as the spec does, to the base triple plus `r rdf:reifies <<( s p o )>>`. Template lowering turns the triple term into `TemplateTerm::TripleTerm`, whose positions resolve per solution; WHERE lowering turns it into the reifier pattern queries use. - **Cypher**: `CREATE (a)-[r:T]->(b)` writes `a T b` and `r rdf:reifies <<( a T b )>>` with `r` a fresh reifier. -- **JSON-LD**: JSON-LD has no triple-term syntax, so `@annotation` / `@edge` lower (before expansion) to `f:reifies*` slot keys on a sibling node. After parsing, `fold_slots_into_links` turns each annotation's slots into its link template, and bulk import's `ImportSink` does the same with the slot triples it receives. The slots are an intermediate form and are never stored. A delete-by-selector matches the existing link with the query-side `@reifies` form. +- **JSON-LD**: JSON-LD has no triple-term syntax, so `@annotation` / `@edge` lower (before expansion) to `f:reifies*` slot keys on a sibling node, and a node's `@reifies` (`{"@id": s, p: o}`, or an array of them) to slot keys on the node itself. After parsing, `fold_slots_into_links` turns each reifier's slots into its link template, and bulk import's `ImportSink` does the same with the slot triples it receives. The slots are an intermediate form and are never stored. A delete-by-selector matches the existing link with the query-side `@reifies` form. The `f:reifies*` predicates are reserved: the write surfaces reject user-authored ones. @@ -51,13 +51,13 @@ A reifier may reify several triples. Re-pointing one is a retract of the old lin Every annotation pattern reads the link. Reified-triple patterns (`<< s p o ~ ?r >>`, `?r rdf:reifies <<( s p o )>>`, JSON-LD `@reifies`) lower to it directly: `lower_reified_link` (`fluree-db-query/src/ir/term_components.rs`), shared by the SPARQL and JSON-LD lowerings, emits the reifier's `rdf:reifies` link, `?r rdf:reifies ?t`, plus `TermComponents(?t, s, p, o)` relating the term to its components and a `sameTerm` filter per constant component, which the planner turns into the scan's handle interval. A fully constant edge composes to a constant term instead. -Annotation syntax (`s p o ~ ?r {| ... |}`, JSON-LD `@annotation`, Cypher relationship properties) keeps its `Pattern::EdgeAnnotation` through lowering, and `expand_edge_annotation_patterns_for` (`where_plan.rs`) expands it at planning into the body and the link with its term components (`link_patterns`). The base edge is not joined: every write that attaches a reifier asserts its edge in the same graph and commit, and retracting the edge retracts the link, so a live link names a live edge. A reifier written as `r rdf:reifies <<( s p o )>>` without asserting `s p o` ends that invariant, and the check then has to come back as a join. The chain is wrapped in `Pattern::DefaultGraphSource` only when the default graph is a union of two or more graphs (`PlanningContext::default_graph_union`). +Annotation syntax (`s p o ~ ?r {| ... |}`, JSON-LD `@annotation`, Cypher relationship properties) keeps its `Pattern::EdgeAnnotation` through lowering, and `expand_edge_annotation_patterns_for` (`where_plan.rs`) expands it at planning into the body, the link with its term components (`link_patterns`), and the base edge: the annotation syntax asserts its triple, so a reifier of an unasserted triple must not match it. The base edge comes last in the chain, so where estimates tie it is a bound existence probe. The chain is wrapped in `Pattern::DefaultGraphSource` only when the default graph is a union of two or more graphs (`PlanningContext::default_graph_union`). -The link names its triple without joining the base edge, so visibility is checked on the term: `QueryPolicyEnforcer` lets a flake whose object is a triple term through only when the triple that term names would be visible, recursively for nested terms. A policy hiding `ex:worksFor` therefore hides the links to `ex:worksFor` edges on every route. +A reified-triple pattern names its triple without joining it, so visibility is checked on the term: `QueryPolicyEnforcer` lets a flake whose object is a triple term through only when the triple that term names would be visible, recursively for nested terms. A policy hiding `ex:worksFor` therefore hides the links to `ex:worksFor` edges on every route. A link is ordinary data: wildcard scans (`?s ?p ?o`) and wildcard hydration return it like any triple, and hydration renders its triple term as an embedded node. An `@annotation` body leaves its reifier's link out, since the body hangs from it, and so do Cypher property maps, where the link is the relationship itself. -Hydration (`@annotation` in subject expansion) and export read links the same way: hydration probes `rdf:reifies` with the rendered edge's term; export reads the ledger's live links once and writes a `~ ` marker on each base edge it reaches. +Hydration (`@annotation` in subject expansion) and export read links the same way: hydration probes `rdf:reifies` with the rendered edge's term; export reads the ledger's live links once and writes a `~ ` marker (`@annotation` in JSON-LD) on each base edge it reaches. A link whose triple its graph does not assert has no edge to carry a marker, so export writes it as the link itself: `r rdf:reifies <<( s p o )>>`, or `@reifies` in JSON-LD. CONSTRUCT does the same: `?r rdf:reifies <<( … )>>` in a template writes the reification without the triple (`ConstructTemplate::push_reified_pattern`). ## Ledgers written before links diff --git a/docs/guides/cookbook-edge-annotations.md b/docs/guides/cookbook-edge-annotations.md index 0cc652c34e..6d18a6a515 100644 --- a/docs/guides/cookbook-edge-annotations.md +++ b/docs/guides/cookbook-edge-annotations.md @@ -12,7 +12,7 @@ Throughout, the running example is employment: a `worksFor` edge that needs a `r |---|---|---| | JSON-LD writes, or you need named-graph edges, or literal-valued edges | **JSON-LD `@annotation`** | Most complete surface — covers everything below. | | A SPARQL 1.1/1.2 pipeline, or you're porting RDF-star data | **SPARQL 1.2 annotation tail** (`{\| \|}`, `~`, `rdf:reifies`) | Standards syntax. Default-graph only today. | -| A Turtle / N-Triples / TriG / N-Quads file with RDF 1.2 annotations | **Ingest it as-is** — `insert`, `upsert`, `import` and `graph sync` all accept `{\| \|}`, `~`, `<< >>` and `rdf:reifies <<( )>>`, and TriG `GRAPH { }` blocks accept them too (TriG via `insert` / `upsert` / `import` / `/sync`) | Same on-disk shape as `@annotation`; the reified triple is asserted; re-`upsert` the file to update claim bodies (see [Turtle ingest](../transactions/turtle.md#edge-annotations-rdf-12--turtle-star)). | +| A Turtle / N-Triples / TriG / N-Quads file with RDF 1.2 annotations | **Ingest it as-is** — `insert`, `upsert`, `import` and `graph sync` all accept `{\| \|}`, `~`, `<< >>` and `rdf:reifies <<( )>>`, and TriG `GRAPH { }` blocks accept them too (TriG via `insert` / `upsert` / `import` / `/sync`) | Same on-disk shape as `@annotation`; the annotation syntax asserts its triple, `<< >>` and `rdf:reifies` do not; re-`upsert` the file to update claim bodies (see [Turtle ingest](../transactions/turtle.md#edge-annotations-rdf-12--turtle-star)). | ## Attach metadata to a relationship @@ -312,11 +312,11 @@ The edge and every other claim on it stay live. In RDF mode the named claim's bo ## Gotchas -- **An annotation reifies exactly one live edge.** A single edge carries many parallel annotations, but one annotation `@id` can't point at two edges at once. To re-home an explicit-IRI annotation, retract the old attachment and assert the new one in the same transaction. +- **One annotation `@id` may reify several triples**, and a single edge carries many parallel annotations. To re-home an explicit-IRI annotation, retract the old attachment and assert the new one in the same transaction; a JSON-LD upsert of the annotation does that for you. - **Deleting a claim with `DELETE DATA { … ~ :claim {| … |} }` deletes the edge** and detaches every other claim on it. Retract one claim with the JSON-LD by-id form (see [above](#retract-one-claim-and-keep-the-edge)). - **Don't write `f:reifies*` predicates by hand.** They're reserved and rejected on every write surface; they're also hidden from `?p` scans and `select: "*"`. Use `@annotation` / the annotation tail. (See [Vocabulary](../reference/vocabulary.md#edge-annotation-predicates-reserved).) - **Empty `@annotation: {}`** is a no-op in RDF mode (no subject minted); in LPG mode it mints a property-less relationship with identity. -- **Not yet supported** (all reject cleanly, no silent partial results): annotations on `@list` elements, reifiers for unasserted triples, triple terms as object values, annotation output in Turtle/CONSTRUCT, and the SPARQL 1.2 triple-term functions (`TRIPLE`, `isTRIPLE`, …). See [Current limits](../concepts/edge-annotations.md#current-limits). +- **Not yet supported** (all reject cleanly, no silent partial results): annotations on `@list` elements, triple terms as object values, annotation output in Turtle/CONSTRUCT, and the SPARQL 1.2 triple-term functions (`TRIPLE`, `isTRIPLE`, …). See [Current limits](../concepts/edge-annotations.md#current-limits). ## See also diff --git a/docs/query/construct.md b/docs/query/construct.md index 3dc05802bf..e72ed5c3ef 100644 --- a/docs/query/construct.md +++ b/docs/query/construct.md @@ -362,10 +362,10 @@ details of each. template is a basic graph pattern, per SPARQL 1.1). - A triple term in a template is accepted only as the object of `rdf:reifies`; nested triple terms and property paths inside a template annotation block are rejected. -- `?r rdf:reifies <<( s p o )>>` in a template also writes `s p o`, the same as the annotation - tail `s p o ~ ?r`. Fluree reifies asserted edges only, as every write form does (see - [Edge annotations](../concepts/edge-annotations.md)), so a result never carries a reifier - without its triple. `?r rdf:reifies ?t`, with `?t` bound to a triple term, writes the same. +- `?r rdf:reifies <<( s p o )>>` in a template writes the reification without `s p o`, as RDF + 1.2 defines it; the annotation tail `s p o ~ ?r` writes both. `?r rdf:reifies ?t`, with `?t` + bound to a triple term, writes the same as the first. A JSON-LD result writes a reification + whose triple it does not carry as the reifier's `@reifies`. Under any other predicate, a bound triple term is written as a literal holding its N-Triples text: a result graph holds triple terms only as reifications. - A SPARQL datalog rule whose head (the template) annotates an edge or writes into a named diff --git a/docs/reference/compatibility.md b/docs/reference/compatibility.md index 0ced65d7fe..b2fff3a570 100644 --- a/docs/reference/compatibility.md +++ b/docs/reference/compatibility.md @@ -34,9 +34,9 @@ Turtle 1.2 annotation syntax is accepted on ingest — `{| ... |}` annotation tails, the `~` reifier, `<< s p o >>` reified triples and `r rdf:reifies <<( s p o )>>` — on every Turtle write path (insert, upsert, import, graph sync over the CLI or `/sync`), inside TriG `GRAPH { }` blocks, and in N-Triples and -N-Quads files. All forms assert the base triple: Fluree reifies asserted -edges, so `<< s p o >>` is asserting here where RDF 1.2 makes it -non-asserting. The `VERSION "1.2"` / `@version` +N-Quads files. As in RDF 1.2, only the annotation syntax asserts the triple; +`<< s p o >>` and `rdf:reifies <<( s p o )>>` reify it without asserting it. +The `VERSION "1.2"` / `@version` directive and `--ltr` / `--rtl` base-direction language tags are accepted. The vendored W3C RDF 1.1 and RDF 1.2 Turtle suites run in CI (`testsuite-sparql/tests/w3c_rdf.rs`), with known gaps in the skip register. diff --git a/docs/transactions/insert.md b/docs/transactions/insert.md index 13edff1f18..4777d1ea78 100644 --- a/docs/transactions/insert.md +++ b/docs/transactions/insert.md @@ -377,10 +377,9 @@ Inline `@annotation` queries return one row per occurrence. **Deferred shapes** error with explicit messages: - Annotations on list-occurrence triples (`@list` membership). -- Reifiers for unasserted triples (`@reifies` must point at an asserted edge). -- Multi-triple `@reifies` (more than one predicate-object pair under `@reifies`). +- More than one predicate-object pair in one `@reifies` block (use an array of blocks to + reify several triples). - Annotation-of-annotation (nested `@annotation` inside an annotation body). -- `@reifies` on the insert side (use the inline `@annotation` form instead). - Hand-authored mention of the [reserved system predicates](../reference/vocabulary.md#edge-annotation-predicates-reserved) that back annotations (compact or full IRI form). For the full surface — including SPARQL 1.2 / RDF 1.2 annotation tails (`{| |}`), the named reifier (`~`), the cardinality / multiplicity contract, anonymous vs explicit-IRI lifecycle, and named-graph behavior — see the [Edge annotations concept doc](../concepts/edge-annotations.md). For cascade semantics when a base edge or annotation metadata is removed, see [Retractions](retractions.md). diff --git a/docs/transactions/turtle.md b/docs/transactions/turtle.md index 79039b7c92..e0e6dbed0b 100644 --- a/docs/transactions/turtle.md +++ b/docs/transactions/turtle.md @@ -482,7 +482,7 @@ ex:emp1 rdf:reifies <<( ex:alice ex:worksFor ex:acme )>> . Two rules to know: -- **The reified triple is asserted.** RDF 1.2 says `<< s p o >>` and `r rdf:reifies <<( s p o )>>` do *not* put `s p o` in the graph; Fluree's annotations describe a live edge, so ingest asserts the base triple as well and attaches the reifier to it. The reifier's own triples (the annotation body) are ordinary RDF about the reifier. Each anonymous `<< s p o >>` / `{| |}` occurrence mints a fresh reifier — two textual occurrences are two annotations. +- **Only the annotation syntax asserts the triple.** As RDF 1.2 defines them, `s p o ~ r` and `s p o {| … |}` put `s p o` in the graph and attach the reifier to it, while `<< s p o >>` and `r rdf:reifies <<( s p o )>>` attach the reifier without asserting `s p o`. The reifier's own triples (the annotation body) are ordinary RDF about the reifier. Each anonymous `<< s p o >>` / `{| |}` occurrence mints a fresh reifier — two textual occurrences are two annotations. - **`<<( ... )>>` is accepted only as the object of `rdf:reifies`.** As a plain value (`ex:doc ex:mentions <<( ... )>>`), nested inside another triple term, or inside an annotation body, it is rejected with a specific "deferred" error rather than silently dropped. TriG and N-Quads accept the same forms inside `GRAPH { }` blocks (and on N-Quads statements with a graph label). The annotation is written into that graph and carries the edge's graph identity, exactly as JSON-LD `@graph` + `@annotation` does: diff --git a/fluree-db-api/src/export.rs b/fluree-db-api/src/export.rs index 976f472f03..2f29774bf0 100644 --- a/fluree-db-api/src/export.rs +++ b/fluree-db-api/src/export.rs @@ -425,6 +425,26 @@ impl<'a> AnnotationContext<'a> { fn is_reifies_row(&self, p_id: u32) -> bool { self.reifies_p_ids.contains(&p_id) } + + /// Whether a link row gives way to the `~ ` marker on its asserted + /// edge. A link whose triple is not asserted has no edge to carry it, + /// so it is written as a row. + fn link_becomes_marker( + &self, + resolver: &ExportResolver<'_>, + s_id: u64, + (o_type, o_key, p_id): (u16, u64, u32), + g_id: GraphId, + ) -> io::Result { + if !self.probe.has_unasserted() { + return Ok(true); + } + let FlakeValue::TripleTerm(term) = resolver.decode_value(o_type, o_key, p_id, g_id)? else { + return Ok(true); + }; + let reifier = resolver.resolve_subject_sid(s_id)?; + Ok(!self.probe.link_is_unasserted(g_id, &reifier, &term)) + } } /// An `rdf:reifies` link row: replaced by annotation syntax, and written as @@ -631,8 +651,12 @@ async fn resolve_untranslated( let mut base: Vec = Vec::with_capacity(rows.len()); for f in rows { if fluree_db_core::is_rdf_reifies(&f.p) { - ann.probe.note_link_sid(f.s.clone()); - continue; + let unasserted = matches!(&f.o, FlakeValue::TripleTerm(term) + if ann.probe.link_is_unasserted(g_id, &f.s, term)); + if !unasserted { + ann.probe.note_link_sid(f.s.clone()); + continue; + } } if fluree_db_core::namespaces::is_reserved_reifies_predicate(&f.p) { continue; @@ -681,7 +705,9 @@ fn write_turtle_batch( // Annotation syntax replaces each link with the `~ ` marker // emitted below; a legacy `f:reifies*` bundle is read as its link. if let Some(ann) = ann { - if is_link_row(o_type) { + if is_link_row(o_type) + && ann.link_becomes_marker(resolver, s_id, (o_type, o_key, p_id), g_id)? + { ann.probe.note_link_in_scope(resolver, s_id); continue; } @@ -874,7 +900,14 @@ pub async fn export_graph_jsonld( let o_type = batch.o_type.get_or(row, 0); let o_key = batch.o_key.get(row); if let Some(ann) = ann.as_ref() { - if is_link_row(o_type) { + if is_link_row(o_type) + && ann.link_becomes_marker( + &resolver, + s_id, + (o_type, o_key, p_id), + config.g_id, + )? + { ann.probe.note_link_in_scope(&resolver, s_id); continue; } @@ -897,22 +930,28 @@ pub async fn export_graph_jsonld( continue; } - // Convert to JSON-LD value - let jval = flake_to_jsonld(&value, store, o_type, prefixes); - // One value per reifier, each carrying `@annotation` — the JSON-LD - // shape `parse/edge_annotations.rs` ingests. Repeating the base - // value is how the keyword attaches to an edge: two reifiers on - // one edge are two annotated occurrences of the same triple, which - // re-ingest to one triple and two bundles. - let jvals = match ann.as_ref() { - Some(ann) => annotated_jsonld_values( - &resolver, - ann, - &jval, - row_reifiers(&reifiers, row), - prefixes, - ), - None => vec![jval], + let (key, jvals) = match reifies_jsonld(&p_iri, &value, store, prefixes) { + Some(block) => (REIFIES_KEY.to_string(), vec![block]), + None => { + let jval = flake_to_jsonld(&value, store, o_type, prefixes); + // One value per reifier, each carrying `@annotation` — the + // JSON-LD shape `parse/edge_annotations.rs` ingests. + // Repeating the base value is how the keyword attaches to + // an edge: two reifiers on one edge are two annotated + // occurrences of the same triple, which re-ingest to one + // triple and two links. + let jvals = match ann.as_ref() { + Some(ann) => annotated_jsonld_values( + &resolver, + ann, + &jval, + row_reifiers(&reifiers, row), + prefixes, + ), + None => vec![jval], + }; + (compact_iri(&p_iri, prefixes), jvals) + } }; // Check if we've moved to a new subject @@ -939,11 +978,10 @@ pub async fn export_graph_jsonld( } // Append value to the right predicate bucket - let compact_p = compact_iri(&p_iri, prefixes); - if let Some(entry) = current_props.iter_mut().find(|(k, _)| *k == compact_p) { + if let Some(entry) = current_props.iter_mut().find(|(k, _)| *k == key) { entry.1.extend(jvals); } else { - current_props.push((compact_p, jvals)); + current_props.push((key, jvals)); } stats.triples_written += 1; @@ -1011,10 +1049,19 @@ fn merge_untranslated_jsonld( ) { let store = resolver.store; for flake in flakes { - let (Some(p_iri), Some(jval)) = ( - store.sid_to_iri(&flake.p), - flake_to_jsonld_raw(flake, store, prefixes), - ) else { + let Some(p_iri) = store.sid_to_iri(&flake.p) else { + stats.rows_skipped += 1; + continue; + }; + if let Some(block) = reifies_jsonld(&p_iri, &flake.o, store, prefixes) { + match props.iter_mut().find(|(k, _)| k == REIFIES_KEY) { + Some(entry) => entry.1.push(block), + None => props.push((REIFIES_KEY.to_string(), vec![block])), + } + stats.triples_written += 1; + continue; + } + let Some(jval) = flake_to_jsonld_raw(flake, store, prefixes) else { stats.rows_skipped += 1; continue; }; @@ -1040,6 +1087,46 @@ fn merge_untranslated_jsonld( } } +const REIFIES_KEY: &str = "@reifies"; + +/// An `rdf:reifies` link's triple as the JSON-LD `@reifies` block, +/// `{"@id": s, p: o}` — the form JSON-LD inserts take for a reification. +/// `None` for any other row, and for a nested term, which has no block. +fn reifies_jsonld( + p_iri: &str, + value: &FlakeValue, + store: &BinaryIndexStore, + prefixes: &PrefixMap, +) -> Option { + let FlakeValue::TripleTerm(term) = value else { + return None; + }; + if p_iri != fluree_vocab::rdf::REIFIES { + return None; + } + let meta = term.lang.clone().map(|lang| fluree_db_core::FlakeMeta { + lang: Some(lang), + i: None, + }); + let object = Flake::new( + term.s.clone(), + term.p.clone(), + term.o.clone(), + term.dt.clone(), + 0, + true, + meta, + ); + let object = flake_to_jsonld_raw(&object, store, prefixes)?; + let mut block = serde_json::Map::new(); + block.insert( + "@id".to_string(), + serde_json::Value::String(compact_iri(&store.sid_to_iri(&term.s)?, prefixes)), + ); + block.insert(compact_iri(&store.sid_to_iri(&term.p)?, prefixes), object); + Some(serde_json::Value::Object(block)) +} + /// JSON-LD value for an untranslated overlay flake, deriving the language tag /// from `flake.m` and the datatype from `flake.dt`. Returns `None` for value /// variants that should never reach the untranslated set. @@ -1559,7 +1646,9 @@ fn write_batch( let o_type = batch.o_type.get_or(row, 0); let o_key = batch.o_key.get(row); if let Some(ann) = ann { - if is_link_row(o_type) { + if is_link_row(o_type) + && ann.link_becomes_marker(resolver, s_id, (o_type, o_key, p_id), g_id)? + { ann.probe.note_link_in_scope(resolver, s_id); continue; } diff --git a/fluree-db-api/src/export_annotations.rs b/fluree-db-api/src/export_annotations.rs index c1c6a82635..80965ce098 100644 --- a/fluree-db-api/src/export_annotations.rs +++ b/fluree-db-api/src/export_annotations.rs @@ -23,9 +23,13 @@ use crate::{LedgerState, Result}; /// on the public `ExportConfig`. An external caller building an `ExportConfig` /// by hand passes `annotations: None` and gets the links as ordinary triples. pub struct AnnotationProbe<'a> { - /// Live links at the export's `t`, per graph, by the edge they name - /// (its `g` cleared). + /// Live links at the export's `t` whose triple is asserted, per graph, + /// by the edge they name (its `g` cleared). Each becomes a marker on + /// its edge. links: HashMap>>, + /// Live links whose triple is not asserted in their graph: no edge + /// carries their marker, so they are written as rows. + unasserted: HashSet<(GraphId, Sid, EdgeKey)>, /// Reifiers named by a `~ ` marker somewhere in this export. named: Mutex>, /// Reifiers whose link the scan passed — i.e. whose own subject is inside @@ -95,6 +99,8 @@ impl<'a> AnnotationProbe<'a> { /// Read `ledger`'s live links as of `as_of_t`, or establish that it has /// none. `Ok(None)` when neither the index nor novelty has ever held an /// annotation, so a ledger without annotations pays two boolean reads. + /// Each link's triple is looked up once per `(graph, subject, + /// predicate)` to tell markers from rows. pub(crate) async fn for_ledger(ledger: &'a LedgerState, as_of_t: i64) -> Result> { if !ledger.snapshot.has_annotations && !ledger.novelty.has_annotations() { return Ok(None); @@ -107,6 +113,7 @@ impl<'a> AnnotationProbe<'a> { .map(|(g_id, _)| g_id), ); let mut links: HashMap>> = HashMap::new(); + let mut unasserted: HashSet<(GraphId, Sid, EdgeKey)> = HashSet::new(); for g_id in graphs { let flakes = range_with_overlay( &ledger.snapshot, @@ -119,35 +126,83 @@ impl<'a> AnnotationProbe<'a> { ) .await?; for flake in flakes { - let FlakeValue::TripleTerm(term) = flake.o else { + let FlakeValue::TripleTerm(term) = &flake.o else { continue; }; - let edge = EdgeKey { - g: None, - s: term.s, - p: term.p, - o: term.o, - dt: term.dt, - lang: term.lang, - list_i: None, - }; + let edge = term_edge(term); let reifiers = links.entry(g_id).or_default().entry(edge).or_default(); if !reifiers.contains(&flake.s) { reifiers.push(flake.s); } } } + for (&g_id, graph_links) in &mut links { + let mut by_subject_predicate: HashMap<(Sid, Sid), Vec> = HashMap::new(); + for edge in graph_links.keys() { + by_subject_predicate + .entry((edge.s.clone(), edge.p.clone())) + .or_default() + .push(edge.clone()); + } + for ((s, p), edges) in by_subject_predicate { + let asserted: HashSet = range_with_overlay( + &ledger.snapshot, + g_id, + ledger.novelty.as_ref(), + IndexType::Spot, + RangeTest::Eq, + RangeMatch::subject_predicate(s, p), + RangeOptions::new().with_to_t(as_of_t), + ) + .await? + .iter() + .map(|flake| EdgeKey { + g: None, + list_i: None, + ..EdgeKey::from_flake(flake) + }) + .collect(); + for edge in edges { + if !asserted.contains(&edge) { + for reifier in graph_links.remove(&edge).unwrap_or_default() { + unasserted.insert((g_id, reifier, edge.clone())); + } + } + } + } + } for reifiers in links.values_mut().flat_map(HashMap::values_mut) { reifiers.sort(); } Ok(Some(Self { links, + unasserted, named: Mutex::new(HashSet::new()), in_scope: Mutex::new(HashSet::new()), _ledger: PhantomData, })) } + /// Whether any live link names a triple its graph does not assert. + pub(crate) fn has_unasserted(&self) -> bool { + !self.unasserted.is_empty() + } + + /// Whether `reifier`'s link to `term` in graph `g_id` names a triple the + /// graph does not assert, so the link is written as a row rather than + /// replaced by a marker. + pub(crate) fn link_is_unasserted( + &self, + g_id: GraphId, + reifier: &Sid, + term: &fluree_db_core::TripleTermValue, + ) -> bool { + !self.unasserted.is_empty() + && self + .unasserted + .contains(&(g_id, reifier.clone(), term_edge(term))) + } + /// Live reifiers for each edge of graph `g_id`, index-aligned with /// `edges`; entry `i` is empty when `edges[i]` carries no annotation. pub(crate) fn live_reifiers(&self, g_id: GraphId, edges: &[EdgeKey]) -> Vec> { @@ -167,6 +222,19 @@ impl<'a> AnnotationProbe<'a> { } } +/// The edge a triple term names, keyed as the probe keys it. +fn term_edge(term: &fluree_db_core::TripleTermValue) -> EdgeKey { + EdgeKey { + g: None, + s: term.s.clone(), + p: term.p.clone(), + o: term.o.clone(), + dt: term.dt.clone(), + lang: term.lang.clone(), + list_i: None, + } +} + /// Resolves a subject id to the `Sid` a reifier was stored under. /// /// A trait so this module does not have to know about the export writers' diff --git a/fluree-db-api/src/format/construct.rs b/fluree-db-api/src/format/construct.rs index c9937ca200..edc9b9f567 100644 --- a/fluree-db-api/src/format/construct.rs +++ b/fluree-db-api/src/format/construct.rs @@ -118,7 +118,13 @@ pub(super) fn instantiate_construct_graph( Some(g) => Some(slot_of(&mut terms, g, Position::Graph)?), None => None, }; - patterns.push(([s, p, o], graph, Vec::new(), const_term)); + patterns.push(( + [s, p, o], + graph, + Vec::new(), + const_term, + template.is_asserted(i), + )); } for r in template.reifications() { let reifier = slot_of(&mut terms, &r.reifier, Position::Subject)?; @@ -158,10 +164,11 @@ pub(super) fn instantiate_construct_graph( }, }) }; - 'pattern: for (slots, graph_slot, reifier_slots, const_term) in &patterns { + 'pattern: for (slots, graph_slot, reifier_slots, const_term, asserted) in &patterns { // `?r rdf:reifies ?t` with a triple term bound to ?t (or a // constant one) writes what `?r rdf:reifies <<( s p o )>>` - // does: the term's triple and ?r's reification of it. + // does: ?r's reification of the term's triple, which it does + // not assert. let components = match &slots[2] { Slot::Var(v) => batch .get(row, *v) @@ -186,9 +193,9 @@ pub(super) fn instantiate_construct_graph( None => None, }; let [ts, tp, to] = components; - let g = dataset.graph_mut(graph.as_ref()); - g.add_reification(ts.clone(), tp.clone(), to.clone(), reifier); - g.add(Triple::new(ts, tp, to)); + dataset + .graph_mut(graph.as_ref()) + .add_reification(ts, tp, to, reifier); continue 'pattern; } } @@ -226,7 +233,9 @@ pub(super) fn instantiate_construct_graph( for r in reifiers.drain(..) { g.add_reification(s.clone(), p.clone(), o.clone(), r); } - g.add(Triple::new(s, p, o)); + if *asserted { + g.add(Triple::new(s, p, o)); + } } } } diff --git a/fluree-db-api/tests/it_edge_annotations_parse.rs b/fluree-db-api/tests/it_edge_annotations_parse.rs index d5f82684e9..d1237b9e9e 100644 --- a/fluree-db-api/tests/it_edge_annotations_parse.rs +++ b/fluree-db-api/tests/it_edge_annotations_parse.rs @@ -72,7 +72,9 @@ async fn insert_with_edge_alias_succeeds_under_m1() { } #[tokio::test] -async fn insert_with_reifies_unsupported() { +async fn insert_with_reifies_links_without_asserting() { + // `@reifies` makes the node a reifier of the triple it describes, + // without asserting that triple (RDF 1.2 `r rdf:reifies <<( s p o )>>`). let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "it/edge-annotations:reifies-insert"; let ledger0 = genesis_ledger(&fluree, ledger_id); @@ -81,17 +83,66 @@ async fn insert_with_reifies_unsupported() { "@context": ctx(), "@id": "ex:employment-1", "ex:role": "Engineer", - "@reifies": { - "@id": "ex:alice", - "ex:worksFor": { "@id": "ex:acme" } - } + "@reifies": [ + { "@id": "ex:alice", "ex:worksFor": { "@id": "ex:acme" } }, + { "@id": "ex:alice", "ex:age": 42 } + ] }); + let committed = fluree.insert(ledger0, &txn).await.expect("@reifies insert"); - let err = fluree - .insert(ledger0, &txn) - .await - .expect_err("M0: @reifies on insert is rejected"); - assert!(err.to_string().contains("@reifies")); + let rows = |query: serde_json::Value| { + let (fluree, ledger) = (&fluree, &committed.ledger); + async move { + support::query_jsonld_formatted(fluree, ledger, &query) + .await + .expect("query") + .as_array() + .expect("rows") + .clone() + } + }; + let reified = rows(json!({ + "@context": ctx(), + "select": ["?r", "?role"], + "where": { + "@id": "?r", + "@reifies": { "@id": "ex:alice", "ex:worksFor": { "@id": "ex:acme" } }, + "ex:role": "?role" + } + })) + .await; + assert_eq!(reified, vec![json!(["ex:employment-1", "Engineer"])]); + let literal = rows(json!({ + "@context": ctx(), + "select": ["?r"], + "where": { "@id": "?r", "@reifies": { "@id": "ex:alice", "ex:age": 42 } } + })) + .await; + assert_eq!(literal, vec![json!(["ex:employment-1"])]); + + let asserted = rows(json!({ + "@context": ctx(), + "select": ["?p", "?o"], + "where": { "@id": "ex:alice", "?p": "?o" } + })) + .await; + assert!( + asserted.is_empty(), + "the reified triples are not asserted: {asserted:?}" + ); + let annotated = rows(json!({ + "@context": ctx(), + "select": ["?role"], + "where": { + "@id": "ex:alice", + "ex:worksFor": { "@id": "ex:acme", "@annotation": { "ex:role": "?role" } } + } + })) + .await; + assert!( + annotated.is_empty(), + "annotation syntax needs the asserted edge: {annotated:?}" + ); } #[tokio::test] diff --git a/fluree-db-api/tests/it_import_turtle_star.rs b/fluree-db-api/tests/it_import_turtle_star.rs index 87bafff899..2139923e38 100644 --- a/fluree-db-api/tests/it_import_turtle_star.rs +++ b/fluree-db-api/tests/it_import_turtle_star.rs @@ -177,13 +177,16 @@ async fn knows_claims( .collect() } -/// N-Quads has only the `rdf:reifies <<( … )>>` spelling. The importer -/// regroups labeled statements into TriG blocks, so the claim follows its +/// N-Quads has only the `rdf:reifies <<( … )>>` spelling, so an annotated +/// triple is the triple plus its reifier's link. The importer regroups +/// labeled statements into TriG blocks, so the claim follows its /// statement's graph label. #[tokio::test] async fn imported_nquads_claims_land_in_their_statement_graph() { - const NQUADS: &str = r#"_:r1 <<( )>> . + const NQUADS: &str = r#" . +_:r1 <<( )>> . _:r1 "0.9" . + . _:r2 <<( )>> . _:r2 "0.5" . "#; diff --git a/fluree-db-api/tests/it_query_construct.rs b/fluree-db-api/tests/it_query_construct.rs index 2edd56c98e..d90c191435 100644 --- a/fluree-db-api/tests/it_query_construct.rs +++ b/fluree-db-api/tests/it_query_construct.rs @@ -1473,7 +1473,7 @@ mod annotations_and_graphs { } /// `?r rdf:reifies <<( s p o )>>` in a template is the same attachment as - /// the `~ ?r` spelling. + /// the `~ ?r` spelling, without the triple the annotation tail asserts. #[tokio::test] async fn reifies_spelling_matches_annotation_tail() { let (fluree, ledger) = annotated().await; @@ -1490,10 +1490,13 @@ mod annotations_and_graphs { &format!("CONSTRUCT {{ ?r rdf:reifies <<( ?s ex:worksFor ?o )>> }} {where_clause}"), ) .await; - assert_eq!( - sorted_lines(&render(&tail, &ledger, FormatterConfig::ntriples())), - sorted_lines(&render(&reifies, &ledger, FormatterConfig::ntriples())) - ); + let tail = render(&tail, &ledger, FormatterConfig::ntriples()); + let reifies = render(&reifies, &ledger, FormatterConfig::ntriples()); + let (links, triples): (Vec<&str>, Vec<&str>) = sorted_lines(&tail) + .into_iter() + .partition(|line| line.contains("22-rdf-syntax-ns#reifies>")); + assert!(!triples.is_empty(), "the tail asserts its triples"); + assert_eq!(links, sorted_lines(&reifies)); } /// SPARQL and JSON-LD share the template IR: an annotated edge in either diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index f76efe8f6c..b5134f87fb 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -459,14 +459,15 @@ async fn link_lowering_joins_component_variables_in_every_scope() { let got = run("SELECT ?s WHERE { ?s ex:age ?age . << ?s ?p ?o >> ex:source ?src }").await; assert!(got.is_empty(), "carol's claim has no source: {got:#?}"); + // `<< ex:bob ex:knows ex:dave >>` reifies a triple it does not assert. let got = run( "SELECT ?s ?o WHERE { ?s ex:knows ?o . ?r rdf:reifies <<( ?s ex:knows ?o )>> } \ ORDER BY ?s ?o", ) .await; - assert_eq!(got.len(), 3, "{got:#?}"); + assert_eq!(got.len(), 2, "{got:#?}"); assert!( - got[2][0].ends_with("bob") && got[2][1].ends_with("dave"), + got[1][0].ends_with("alice") && got[1][1].ends_with("carol"), "{got:#?}" ); } @@ -2093,8 +2094,8 @@ async fn triple_terms_render_in_every_result_format() { } /// `?r rdf:reifies ?t` in a CONSTRUCT template, with a triple term bound to -/// ?t, writes what the explicit `<<( s p o )>>` template writes. Under any -/// other predicate the term (which the graph model cannot hold as an object) +/// ?t, writes what the explicit `<<( s p o )>>` template writes: the +/// reification, without asserting its triple. Under any other predicate the term (which the graph model cannot hold as an object) /// is written as its N-Triples text. #[tokio::test] async fn construct_writes_triple_terms_as_reifications() { @@ -2144,11 +2145,7 @@ async fn construct_writes_triple_terms_as_reifications() { let by_term = construct("?r rdf:reifies ?t", term_form).await; let by_template = construct("?r rdf:reifies <<( ?s ?p ?o )>>", template_form).await; let lines = sorted(&by_term); - assert_eq!( - lines.len(), - 2, - "the base triple and its reification: {lines:#?}" - ); + assert_eq!(lines.len(), 1, "the reification alone: {lines:#?}"); assert_eq!(lines, sorted(&by_template), "{term_form}"); assert_eq!( by_term.to_construct(&ledger.snapshot).expect("JSON-LD"), @@ -2159,7 +2156,7 @@ async fn construct_writes_triple_terms_as_reifications() { let hr = construct("?r rdf:reifies ?t", "?r rdf:reifies ?t ; ex:source ex:hr").await; assert_eq!( hr.to_construct(&ledger.snapshot).expect("JSON-LD")["@graph"], - json!([{"@id": "ex:alice", "ex:knows": [{"@id": "ex:bob", "@annotation": {"@id": "ex:claim1"}}]}]) + json!([{"@id": "ex:claim1", "@reifies": [{"@id": "ex:alice", "ex:knows": {"@id": "ex:bob"}}]}]) ); // The shorthand's template is the WHERE clause: a constant quoted triple @@ -2171,7 +2168,6 @@ async fn construct_writes_triple_terms_as_reifications() { .expect("construct where"); let ex = |l: &str| format!(""); let mut expected = vec![ - format!("{} {} {} .", ex("alice"), ex("knows"), ex("bob")), format!( "{} {REIFIES} <<( {} {} {} )>> .", ex("claim1"), diff --git a/fluree-db-api/tests/it_turtle_star_write_paths.rs b/fluree-db-api/tests/it_turtle_star_write_paths.rs index 8bb959e28d..9d08068422 100644 --- a/fluree-db-api/tests/it_turtle_star_write_paths.rs +++ b/fluree-db-api/tests/it_turtle_star_write_paths.rs @@ -210,13 +210,33 @@ async fn upsert_turtle_and_insert_turtle_agree_on_the_claim_graph() { ); } +/// Rows of `sparql` over `ledger`'s `graph` (or default graph). +async fn select( + fluree: &fluree_db_api::Fluree, + ledger: &fluree_db_api::LedgerState, + graph: Option<&str>, + pattern: &str, +) -> Vec { + let body = match graph { + Some(g) => format!("GRAPH <{g}> {{ {pattern} }}"), + None => pattern.to_string(), + }; + let sparql = format!("PREFIX ex: \nSELECT * WHERE {{ {body} }}"); + let result = support::query_sparql_formatted(fluree, ledger, &sparql) + .await + .expect("query"); + rows(&result).to_vec() +} + #[tokio::test] -async fn rdf_reifies_triple_term_is_accepted_by_upsert_and_sync() { - // The canonical RDF 1.2 spelling reaches the converted paths through the - // same collector events as `~ r`, so it must land the same claim. +async fn rdf_reifies_triple_term_reifies_without_asserting_on_upsert_and_sync() { + // `r rdf:reifies <<( s p o )>>` reaches the converted paths as a + // reification of a triple the graph does not assert (RDF 1.2): the + // reifier links to it, and the triple stays unasserted. let turtle = "@prefix rdf: .\n\ ex:claim1 rdf:reifies <<( ex:alice ex:knows ex:bob )>> .\n\ ex:claim1 ex:confidence 0.9 .\n"; + let reified = "<< ex:alice ex:knows ex:bob ~ ?r >> ex:confidence ?conf"; let fluree = FlureeBuilder::memory().build_memory(); let upserted = fluree @@ -226,7 +246,18 @@ async fn rdf_reifies_triple_term_is_accepted_by_upsert_and_sync() { ) .await .expect("upsert_turtle with rdf:reifies <<( )>>"); - assert_eq!(confidences(&fluree, &upserted.ledger, None).await, ["0.9"]); + assert_eq!( + select(&fluree, &upserted.ledger, None, reified).await.len(), + 1 + ); + assert!( + select(&fluree, &upserted.ledger, None, "ex:alice ex:knows ?o") + .await + .is_empty() + ); + assert!(confidences(&fluree, &upserted.ledger, None) + .await + .is_empty()); let ledger_id = "it/turtle-star-sync:rdf-reifies"; fluree @@ -239,10 +270,14 @@ async fn rdf_reifies_triple_term_is_accepted_by_upsert_and_sync() { assert!(sync(&fluree, ledger_id, turtle) .await .expect("sync with rdf:reifies <<( )>>")); + let synced = fluree.ledger(ledger_id).await.expect("reload"); assert_eq!( - annotated_knows(&fluree, ledger_id).await, - [("alice".into(), "bob".into(), "0.9".into())] + select(&fluree, &synced, Some(CLAIMS_GRAPH), reified) + .await + .len(), + 1 ); + assert!(annotated_knows(&fluree, ledger_id).await.is_empty()); } #[tokio::test] diff --git a/fluree-db-cli/tests/integration.rs b/fluree-db-cli/tests/integration.rs index c615634f9e..f1466a72c2 100644 --- a/fluree-db-cli/tests/integration.rs +++ b/fluree-db-cli/tests/integration.rs @@ -2571,6 +2571,69 @@ fn export_annotations_round_trip_in_every_format() { } } +/// A reifier of a triple the ledger does not assert has no edge to carry a +/// `~` marker: it is written as its `rdf:reifies` link (`@reifies` in +/// JSON-LD), and re-importing it reifies the triple without asserting it. +#[test] +fn export_round_trips_reifications_of_unasserted_triples() { + let src = TempDir::new().unwrap(); + fluree_cmd(&src).arg("init").assert().success(); + let data = src.path().join("unasserted-src"); + std::fs::create_dir_all(&data).unwrap(); + std::fs::write( + data.join("a.ttl"), + "@prefix ex: .\n\ + ex:alice ex:knows ex:bob ~ ex:claim1 {| ex:confidence 0.8 |} .\n\ + << ex:alice ex:knows ex:carol ~ ex:claim2 >> ex:confidence 0.5 .\n", + ) + .unwrap(); + fluree_cmd(&src) + .args(["create", "unasserted", "--from"]) + .arg(&data) + .assert() + .success(); + + for (fmt, ext) in [("turtle", "ttl"), ("ntriples", "nt"), ("jsonld", "jsonld")] { + let out = src.path().join(format!("unasserted.{ext}")); + fluree_cmd(&src) + .args(["export", "unasserted", "--format", fmt, "-o"]) + .arg(&out) + .assert() + .success(); + + let dst = TempDir::new().unwrap(); + fluree_cmd(&dst).arg("init").assert().success(); + fluree_cmd(&dst) + .args(["create", "unasserted", "--from"]) + .arg(&out) + .assert() + .success(); + fluree_cmd(&dst) + .args([ + "query", + "unasserted", + "--sparql", + "PREFIX ex: \ + SELECT ?r ?c WHERE { << ex:alice ex:knows ?o ~ ?r >> ex:confidence ?c }", + ]) + .assert() + .success() + .stdout(predicate::str::contains("claim1")) + .stdout(predicate::str::contains("claim2")); + fluree_cmd(&dst) + .args([ + "query", + "unasserted", + "--sparql", + "PREFIX ex: SELECT ?o WHERE { ex:alice ex:knows ?o }", + ]) + .assert() + .success() + .stdout(predicate::str::contains("bob")) + .stdout(predicate::str::contains("carol").not()); + } +} + /// A link's object keeps its datatype and language tag, so each object shape /// — a ref, a plain literal, a language-tagged one — has to reach the output. #[test] @@ -2810,17 +2873,12 @@ fn an_untranslated_annotation_exports_with_its_marker() { .stdout(predicate::str::contains("")); } -/// The counter still fires when an annotation genuinely cannot be resolved. -/// -/// Paired with the tests above on purpose. "No warning" is satisfied by a -/// counter that has stopped working, so a `MustNotFire` assertion alone -/// cannot distinguish "nothing was dropped" from "the accounting is dead". -/// -/// The fixture is an export written before links whose `f:reifies*` bundle -/// names a triple the file never asserts. Import stores its link, and the -/// export has no base edge to hang the `~` marker on. +/// An export written before links whose `f:reifies*` bundle names a triple +/// the file never asserts imports as a reifier of that unasserted triple. The +/// export has no edge to hang a `~` marker on, so it writes the link itself, +/// and nothing goes unresolved. #[test] -fn the_unresolved_counter_still_fires_when_it_should() { +fn an_old_bundle_naming_an_unasserted_triple_exports_as_its_link() { let tmp = TempDir::new().unwrap(); fluree_cmd(&tmp).arg("init").assert().success(); let src = tmp.path().join("mf-src"); @@ -2843,9 +2901,11 @@ fn the_unresolved_counter_still_fires_when_it_should() { .args(["export", "mf", "--format", "turtle"]) .assert() .success() - .stderr(predicate::str::contains( - "1 edge annotations could not be resolved", - )); + .stdout(predicate::str::contains( + "rdf:reifies <<( \ + )>>", + )) + .stderr(predicate::str::contains("could not be resolved").not()); } /// Every object shape keeps its annotation, including the big-numeric ones. diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index f8bff6982b..ca6ba24a06 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -132,25 +132,22 @@ fn expand_one_into(pattern: Pattern, out: &mut Vec, one_graph: bool) { body, term, } => { - // Build the chain (the body and the annotation's link to the - // edge) into a local vector. It is wrapped in + // Build the chain (the body, the annotation's link to the edge, + // and the edge) into a local vector. It is wrapped in // `Pattern::DefaultGraphSource` when the default graph is a union, // so the per-source iteration correlates them; otherwise one graph // is in scope and the chain joins its enclosing block. // // The order is the planner's tie-break: the body and the link are - // probes by reifier, and the term's components then decode from - // the bound term. Estimates still decide where they differ (a + // probes by reifier, the term's components then decode from the + // bound term, and the edge is a bound existence probe last. Estimates still decide where they differ (a // constant subject anchors the components through the term // dictionary). // - // The base edge is not joined: every write that attaches a - // reifier asserts its edge in the same graph and commit, and - // retracting the edge retracts the attachment, so a live link - // names a live edge; a policy that hides the edge hides the link. - // Joining it anyway probed the edge once per annotation. A - // reifier that does not assert its triple would end that - // invariant, and with it this elision. + // Annotation syntax asserts its triple (RDF 1.2), so the edge is + // joined: a reifier may reify a triple that is not asserted, and + // such a link must not match `s p o {| … |}`. + let base_edge = Pattern::Triple(edge.clone()); let mut chain: Vec = Vec::new(); // 1. Body patterns (recursively expanded so nested annotations — @@ -174,6 +171,7 @@ fn expand_one_into(pattern: Pattern, out: &mut Vec, one_graph: bool) { &|_| None, &mut chain, ); + chain.push(base_edge); // Under a default-graph union the `DefaultGraphSource` wrapper // switches the execution context to one member at a time, so the @@ -2584,9 +2582,9 @@ pub fn build_where_operators_seeded_with_needed( return Ok(seed.unwrap_or_else(|| Box::new(EmptyOperator::new()))); } - // Edge-annotation expansion (M1b): Pattern::EdgeAnnotation is - // flattened into the equivalent base edge plus the `f:reifies*` - // triple lookups plus the body. The standard scan/join machinery + // Edge-annotation expansion: Pattern::EdgeAnnotation is flattened into + // the equivalent base edge plus the reifier's `rdf:reifies` link plus + // the body. The standard scan/join machinery // handles the rest. The base-edge triple is always emitted, which // gives the annotation its visibility check for free: the base edge // must be currently asserted under the snapshot's normal @@ -6615,11 +6613,11 @@ mod tests { let expanded = expand_edge_annotation_patterns(&patterns); let chain = unwrap_default_graph_source(&expanded[0]); // body triple + link + its predicate filter + term components + - // the sunk FILTER - assert_eq!(chain.len(), 5); + // the base edge + the sunk FILTER + assert_eq!(chain.len(), 6); assert_eq!(filter_count(chain), 1); assert!( - matches!(chain[4], Pattern::Filter(_)), + matches!(chain[5], Pattern::Filter(_)), "the FILTER sinks to the end of the chain, where the body is: {chain:?}" ); // Copied, not relocated: the original keeps feeding diff --git a/fluree-db-query/src/ir/query.rs b/fluree-db-query/src/ir/query.rs index 106a87a142..dac48cb2ce 100644 --- a/fluree-db-query/src/ir/query.rs +++ b/fluree-db-query/src/ir/query.rs @@ -53,6 +53,9 @@ pub struct ConstructTemplate { /// `patterns[triple]` instantiates to. A row that leaves either unbound /// contributes no attachment. reifications: Vec, + /// Patterns that are only reified (`r rdf:reifies <<( s p o )>>`), so + /// their triple is not written. Empty when every pattern is asserted. + reified_only: HashSet, } /// A reifier attachment in a CONSTRUCT template (see @@ -80,6 +83,7 @@ impl ConstructTemplate { bnode_vars, graphs: Vec::new(), reifications: Vec::new(), + reified_only: HashSet::new(), } } @@ -95,6 +99,19 @@ impl ConstructTemplate { self.patterns.len() - 1 } + /// Append a pattern whose triple is reified but not written (see + /// [`push_pattern`](Self::push_pattern)). + pub fn push_reified_pattern(&mut self, pattern: TriplePattern, graph: Option) -> usize { + let i = self.push_pattern(pattern, graph); + self.reified_only.insert(i); + i + } + + /// Whether `patterns[i]`'s triple is written, rather than only reified. + pub fn is_asserted(&self, i: usize) -> bool { + !self.reified_only.contains(&i) + } + /// Attach `reifier` to `patterns[triple]`, an index /// [`push_pattern`](Self::push_pattern) returned. pub fn push_reification(&mut self, triple: usize, reifier: Ref) { diff --git a/fluree-db-sparql/src/lower/construct.rs b/fluree-db-sparql/src/lower/construct.rs index f49da18889..37c6ad5abc 100644 --- a/fluree-db-sparql/src/lower/construct.rs +++ b/fluree-db-sparql/src/lower/construct.rs @@ -105,7 +105,7 @@ impl LoweringContext<'_, E> { let p = self.lower_predicate(&tp.predicate)?; // `?r rdf:reifies <<( s p o )>>`: the triple term is the reified - // triple and `?r` its reifier. + // triple, not asserted, and `?r` its reifier. if let Term::TripleTerm(term) = &tp.object { if p != reifies || tp.annotation.is_some() { return Err(LowerError::not_implemented( @@ -122,7 +122,7 @@ impl LoweringContext<'_, E> { tp.span, )); } - let triple = out.push_pattern(reified, graph.clone()); + let triple = out.push_reified_pattern(reified, graph.clone()); out.push_reification(triple, s); continue; } diff --git a/fluree-db-transact/src/parse/edge_annotations.rs b/fluree-db-transact/src/parse/edge_annotations.rs index 910e11047a..ab7784c2c3 100644 --- a/fluree-db-transact/src/parse/edge_annotations.rs +++ b/fluree-db-transact/src/parse/edge_annotations.rs @@ -1416,96 +1416,113 @@ fn build_annotation_sibling( Ok(Some(Value::Object(ann_map))) } -/// Lower a `@reifies` block on the enclosing node. The enclosing node -/// IS the annotation; `@reifies` names the base edge. +/// Lower a `@reifies` block on the enclosing node: the node is a reifier, +/// and each triple the block describes becomes its `rdf:reifies` link +/// without being asserted (RDF 1.2). An array reifies several triples; the +/// first rides on the node, the rest on siblings with the same `@id`. fn lower_reifies_block( map: &mut Map, + reifier: &str, reifies_val: Value, - _ctx: &mut LowerCtx, + graph: Option<&str>, + ctx: &mut LowerCtx, ) -> Result<()> { - let Value::Object(reifies_map) = reifies_val else { + let blocks = match reifies_val { + Value::Array(blocks) if !blocks.is_empty() => blocks, + Value::Array(_) => { + return Err(TransactError::Parse( + "@reifies must describe at least one triple".to_string(), + )) + } + block => vec![block], + }; + for (i, block) in blocks.into_iter().enumerate() { + let slots = reifies_slots(block)?; + if i == 0 { + map.extend(slots); + } else { + let mut sibling = slots; + sibling.insert("@id".to_string(), json!(reifier)); + if let Some(graph) = graph { + sibling.insert("@graph".to_string(), json!(graph)); + } + ctx.siblings.push(Value::Object(sibling)); + } + } + Ok(()) +} + +/// The `f:reifies*` slots naming the one triple a `@reifies` block +/// describes: `{"@id": s, p: o}`. +fn reifies_slots(block: Value) -> Result> { + let Value::Object(block) = block else { return Err(TransactError::Parse( - "@reifies value must be a JSON object describing the base triple".to_string(), + "@reifies value must be a JSON object describing a triple".to_string(), )); }; - - // Reject nested annotations inside @reifies (v1 deferral). - for (k, _) in &reifies_map { + for k in block.keys() { if is_annotation_key(k) || k == REIFIES_KEY { return Err(TransactError::UnsupportedFeature(format!( - "{k} inside @reifies is the deferred nested-triple-term shape (v1)" + "{k} inside @reifies is the deferred nested-triple-term shape" ))); } } - - // Subject of the base edge: @id of the @reifies node-map. - let Some(Value::String(base_subject)) = reifies_map.get("@id") else { + let Some(Value::String(subject)) = block.get("@id") else { return Err(TransactError::Parse( - "@reifies must include an @id naming the base subject".to_string(), + "@reifies must include an @id naming the triple's subject".to_string(), )); }; - - // Find the single predicate-object pair (non-`@`-keyword key). - let pred_obj_pairs: Vec<(&String, &Value)> = reifies_map - .iter() - .filter(|(k, _)| !k.starts_with('@')) - .collect(); - if pred_obj_pairs.len() != 1 { + let pairs: Vec<(&String, &Value)> = block.iter().filter(|(k, _)| !k.starts_with('@')).collect(); + let [(predicate, object)] = pairs.as_slice() else { + return Err(TransactError::Parse(format!( + "@reifies must describe exactly one triple (got {} predicates); \ + use an array to reify several", + pairs.len() + ))); + }; + if reifies_iris::ALL.contains(&predicate.as_str()) || *predicate == fluree_vocab::rdf::REIFIES { return Err(TransactError::UnsupportedFeature(format!( - "@reifies must describe exactly one base triple (got {} predicates); \ - multi-triple reifiers are deferred to v2", - pred_obj_pairs.len() + "'{predicate}' is a system-controlled predicate" ))); } - let (predicate, object_val) = pred_obj_pairs[0]; - - // Resolve the object: must be an IRI string, `{"@id": "..."}`, or a - // blank node; literal-valued reifiers are deferred. - let object_id = match object_val { - Value::String(s) => s.clone(), - Value::Object(ov) => match ov.get("@id") { - Some(Value::String(s)) => s.clone(), + let shape = match object { + Value::Object(object) + if object.contains_key("@value") || object.contains_key("@language") => + { + classify_reified_object(object)? + } + Value::Object(object) => match object.get("@id") { + Some(Value::String(id)) if object.len() == 1 => ReifiedObjectShape::Iri(id.clone()), _ => { - return Err(TransactError::UnsupportedFeature( - "@reifies object position: literal-valued or multi-property objects are deferred (v1); \ - reify only IRI-typed (or @id-shaped) objects" - .to_string(), - )); + return Err(TransactError::Parse( + "@reifies object must be an @id reference or a value".to_string(), + )) } }, - _ => { + Value::Array(_) => { return Err(TransactError::Parse( - "@reifies object must be an IRI string, @id reference, or variable".to_string(), - )); + "@reifies must describe exactly one triple; use an array of @reifies \ + blocks to reify several" + .to_string(), + )) + } + scalar => { + classify_reified_object(&Map::from_iter([("@value".to_string(), (*scalar).clone())]))? } }; - // Inject f:reifies* predicates onto the enclosing map. - map.insert( - reifies_iris::SUBJECT.to_string(), - json!({"@id": base_subject}), - ); - map.insert( + let mut slots = Map::new(); + slots.insert(reifies_iris::SUBJECT.to_string(), json!({"@id": subject})); + slots.insert( reifies_iris::PREDICATE.to_string(), json!({"@id": predicate}), ); - map.insert(reifies_iris::OBJECT.to_string(), json!({"@id": object_id})); - - // The base edge is asserted by the user including @reifies, so - // we don't synthesize a sibling for it: presence of f:reifiesSubject / - // f:reifiesPredicate / f:reifiesObject IS the assertion intent at - // the system level; the actual base flake is asserted via the - // `f:reifies*` mechanism plus the AttachmentNovelty observer in - // M1's runtime path. M2 layers an arena on top. - // - // Wait — actually no. `@reifies` is *only* a query-side construct - // in v1 per the design doc. On the insert path, `@reifies` is - // currently rejected as the deferred unasserted-reifier shape. - Err(TransactError::UnsupportedFeature( - "@reifies on inserts is deferred (v1); use @annotation on the inline form instead, \ - or split the insert into the base edge plus a separate annotation node" - .to_string(), - )) + let (object, lang) = emit_reifies_object_payload(&shape); + slots.insert(reifies_iris::OBJECT.to_string(), object); + if let Some(lang) = lang { + slots.insert(reifies_iris::LANG.to_string(), json!(lang)); + } + Ok(slots) } /// Append synthetic sibling nodes to the document so the standard @@ -1845,16 +1862,14 @@ fn lower_object_with_subject( graph: effective_graph, }; - // 1. Honor `@reifies` on this node (rejected in v1 — see above). - // Subject minting must use the merged context so a node-local - // `@id` alias is recognized. + // 1. Honor `@reifies` on this node. Subject minting must use the + // merged context so a node-local `@id` alias is recognized. if map.contains_key(REIFIES_KEY) { let val = map.remove(REIFIES_KEY).unwrap(); - // `@reifies` is one of the cases that requires a subject id; the - // lower function reads `map`'s `@id` directly, but the mint must - // run first so the value is present. - let _ = ensure_subject_id(map, &child_walk, ctx); - lower_reifies_block(map, val, ctx)?; + // The lower function reads `map`'s `@id` directly, so the mint + // must run first. + let reifier = ensure_subject_id(map, &child_walk, ctx); + lower_reifies_block(map, &reifier, val, child_walk.graph, ctx)?; } // 2. Walk predicate-value pairs. Skip JSON-LD keywords plus their @@ -3212,19 +3227,27 @@ mod tests { } #[test] - fn rejects_reifies_on_insert() { + fn reifies_on_insert_names_its_triple_without_asserting_it() { let doc = json!({ "@id": "ex:employment-1", "ex:role": "Engineer", - "@reifies": { - "@id": "ex:alice", - "ex:worksFor": { "@id": "ex:acme" } - } + "@reifies": [ + { "@id": "ex:alice", "ex:worksFor": { "@id": "ex:acme" } }, + { "@id": "ex:alice", "ex:age": 42 } + ] }); - let err = lower(doc).unwrap_err(); + let lowered = lower(doc).expect("lower"); + let nodes = lowered["@graph"].as_array().expect("reifier + sibling"); + assert_eq!(nodes.len(), 2, "{lowered:#}"); + assert_eq!(nodes[0]["ex:role"], "Engineer"); + assert_eq!(nodes[0][reifies_iris::SUBJECT], json!({"@id": "ex:alice"})); + assert_eq!(nodes[0][reifies_iris::OBJECT], json!({"@id": "ex:acme"})); + assert_eq!(nodes[1]["@id"], "ex:employment-1"); + assert_eq!(nodes[1][reifies_iris::PREDICATE], json!({"@id": "ex:age"})); + assert_eq!(nodes[1][reifies_iris::OBJECT], json!({"@value": 42})); assert!( - err.to_string().contains("@reifies on inserts"), - "expected @reifies-on-insert deferral message, got: {err}" + nodes.iter().all(|n| n.get("ex:worksFor").is_none()), + "the triple is not asserted" ); } diff --git a/fluree-db-transact/src/parse/trig_meta.rs b/fluree-db-transact/src/parse/trig_meta.rs index 2caf10dd02..39c9829cc6 100644 --- a/fluree-db-transact/src/parse/trig_meta.rs +++ b/fluree-db-transact/src/parse/trig_meta.rs @@ -88,9 +88,9 @@ pub struct NamedGraphBlock { pub iri: String, /// Triples in this graph. pub triples: Vec, - /// RDF 1.2 reifier attachments (TriG-star) in this graph. The reified - /// base triple is also present in `triples` (Fluree asserts it), so - /// consumers emit the `f:reifies*` bundle from here and nothing else. + /// RDF 1.2 reifier attachments (TriG-star) in this graph. Each becomes + /// the reifier's `rdf:reifies` link; the reified triple is in `triples` + /// only when the annotation syntax asserted it. pub reified: Vec, /// The document's prefix mappings where the block appears (for IRI /// expansion). Blocks with no directive between them share one map. @@ -998,21 +998,6 @@ impl<'a> TrigMetaParser<'a> { }); } - /// Assert the reified base triple (Fluree's documented divergence from - /// RDF 1.2's non-asserting `<< … >>` / `rdf:reifies`). - fn assert_base_triple( - &mut self, - subject: &TermValue, - predicate: &TermValue, - object: &ObjectValue, - ) { - self.stmt_triples.push(ParsedTriple { - subject: subject.clone(), - predicate: predicate.clone(), - objects: vec![object.clone()], - }); - } - /// `<< rtSubject predicate rtObject ( ~ reifier )? >>` — returns the /// reifier term, which is what the construct denotes in its position. fn parse_reified_triple(&mut self) -> Result { @@ -1051,7 +1036,6 @@ impl<'a> TrigMetaParser<'a> { ))); } self.advance(); - self.assert_base_triple(&subject, &predicate, &object); self.attach_reifier(&subject, &predicate, &object, &reifier); Ok(reifier) } @@ -1099,7 +1083,6 @@ impl<'a> TrigMetaParser<'a> { .to_string(), )); } - self.assert_base_triple(&subject, &predicate, &object); self.attach_reifier(&subject, &predicate, &object, reifier); Ok(()) } @@ -2680,16 +2663,29 @@ ex:alice ex:note "value with a { brace" . @prefix rdf: .\n"; #[test] - fn test_trig_star_every_spelling_yields_one_attachment_and_asserts_the_base() { - for (label, body) in [ - ("annotation block", "ex:s ex:p ex:o {| ex:q ex:z |} ."), - ("tilde reifier", "ex:s ex:p ex:o ~ ex:r ."), - ("tilde + block", "ex:s ex:p ex:o ~ ex:r {| ex:q ex:z |} ."), - ("reified subject", "<< ex:s ex:p ex:o ~ ex:r >> ex:q ex:z ."), - ("reified object", "ex:z ex:q << ex:s ex:p ex:o ~ ex:r >> ."), + fn test_trig_star_every_spelling_yields_one_attachment_and_annotations_assert_the_base() { + for (label, body, asserts) in [ + ("annotation block", "ex:s ex:p ex:o {| ex:q ex:z |} .", true), + ("tilde reifier", "ex:s ex:p ex:o ~ ex:r .", true), + ( + "tilde + block", + "ex:s ex:p ex:o ~ ex:r {| ex:q ex:z |} .", + true, + ), + ( + "reified subject", + "<< ex:s ex:p ex:o ~ ex:r >> ex:q ex:z .", + false, + ), + ( + "reified object", + "ex:z ex:q << ex:s ex:p ex:o ~ ex:r >> .", + false, + ), ( "rdf:reifies triple term", "ex:r rdf:reifies <<( ex:s ex:p ex:o )>> .", + false, ), ] { let block = star_block(&format!("{STAR_PREFIX}GRAPH ex:g {{ {body} }}\n")); @@ -2699,9 +2695,10 @@ ex:alice ex:note "value with a { brace" . assert_eq!(raw_iri(&r.predicate), "ex:p", "[{label}]"); assert!(matches!(&r.object, RawObject::PrefixedName { local, .. } if local == "o")); let triples = triple_strs(&block); - assert!( + assert_eq!( triples.iter().any(|(s, p, _)| s == "ex:s" && p == "ex:p"), - "[{label}] base triple must be asserted: {triples:?}" + asserts, + "[{label}] only the annotation syntax asserts the triple: {triples:?}" ); assert!( !triples.iter().any(|(_, p, _)| p == "rdf:reifies"), diff --git a/fluree-graph-format/src/jsonld.rs b/fluree-graph-format/src/jsonld.rs index b6f28e0270..ba786511dc 100644 --- a/fluree-graph-format/src/jsonld.rs +++ b/fluree-graph-format/src/jsonld.rs @@ -344,6 +344,35 @@ fn graph_nodes( nodes.insert(subj_key, node); } + // A reification of a triple the graph does not assert has no edge to + // annotate: its reifier names the triple with `@reifies`. + let asserted: std::collections::HashSet<&fluree_graph_ir::Triple> = graph.iter().collect(); + for reification in graph.reifications() { + let triple = &reification.triple; + if asserted.contains(triple) { + continue; + } + let Term::Iri(p) = &triple.p else { + continue; + }; + let mut block = Map::new(); + block.insert( + "@id".to_string(), + JsonValue::String(term_to_subject_key(&triple.s, config, bnode_renamer)?), + ); + block.insert( + config.compact_vocab_iri(p), + term_to_object(&triple.o, config, bnode_renamer), + ); + let reifier = term_to_subject_key(&reification.reifier, config, bnode_renamer)?; + let node = nodes.entry(reifier.clone()).or_insert_with(|| { + let mut node = Map::new(); + node.insert("@id".to_string(), JsonValue::String(reifier)); + node + }); + add_property(node, "@reifies", JsonValue::Object(block)); + } + // Post-process: wrap single values in arrays if multicardinal_arrays is enabled // Note: @list values should NOT be wrapped if config.multicardinal_arrays { diff --git a/fluree-graph-ir/src/sink.rs b/fluree-graph-ir/src/sink.rs index 5f87c21f3b..fa92c2bad2 100644 --- a/fluree-graph-ir/src/sink.rs +++ b/fluree-graph-ir/src/sink.rs @@ -305,9 +305,11 @@ pub trait GraphSink { /// triple `(subject, predicate, object)`. /// /// Contract: - /// - The parser has ALREADY emitted the base triple via - /// [`Self::emit_triple`] (Fluree's edge-annotation model reifies an - /// asserted edge). This event only records the reifier attachment. + /// - This event records only the reifier attachment. The base triple is + /// asserted only by the annotation syntax (`s p o ~ r` / `{| … |}`), + /// whose parser emits it via [`Self::emit_triple`] first; a reified + /// triple (`<< s p o >>`, `r rdf:reifies <<( s p o )>>`) does not + /// assert it. /// - The parser mints a FRESH blank-node reifier per anonymous /// occurrence (`<< s p o >>` / `{| … |}` without `~ reifier`) and /// never deduplicates reifiers by base-triple identity; sinks must diff --git a/fluree-graph-turtle/src/adapter.rs b/fluree-graph-turtle/src/adapter.rs index 3a7b4d8b41..3b6728edb1 100644 --- a/fluree-graph-turtle/src/adapter.rs +++ b/fluree-graph-turtle/src/adapter.rs @@ -51,7 +51,9 @@ use std::collections::{BTreeMap, HashMap}; /// Several reifiers on one edge produce one annotated object value each /// (the parallel-annotation shape). A reification is attached to the first /// occurrence of its base triple only, so a document that states an edge -/// twice does not mint its bundle twice. +/// twice does not mint its bundle twice. A reification whose triple the +/// graph does not assert becomes a node naming it with `@reifies`: +/// `{ "@id": r, "@reifies": { "@id": s, p: o } }`. /// /// # Errors /// @@ -64,7 +66,7 @@ pub fn graph_to_transaction_json(graph: &Graph) -> Result { if let Some(r) = graph .reifications() .iter() - .find(|r| r.triple.p.as_iri() == Some(RDF_TYPE_IRI)) + .find(|r| r.triple.p.as_iri() == Some(RDF_TYPE_IRI) && graph.triples().contains(&r.triple)) { return Err(TurtleError::Unsupported(format!( "an annotation on an rdf:type edge (<{}> a <{}> ~ {}) cannot be expressed on \ @@ -152,12 +154,26 @@ pub fn graph_to_transaction_json(graph: &Graph) -> Result { nodes.push(JsonValue::Object(node)); } - // Every attachment names a base triple the producer also emitted - // (`GraphSink::emit_reified_triple` contract), so nothing is left over. - debug_assert!( - reifiers.is_empty(), - "reifications without a base triple in the graph: {reifiers:?}" - ); + // A reification whose triple the graph does not assert (`<< s p o >>`, + // `r rdf:reifies <<( s p o )>>`) names the triple through `@reifies`. + let mut unasserted: Vec<(String, &Triple)> = reifiers + .into_iter() + .flat_map(|(triple, attached)| { + attached + .into_iter() + .map(move |reifier| (term_to_subject_key(reifier), triple)) + }) + .collect(); + unasserted.sort_by(|a, b| a.0.cmp(&b.0)); + for (reifier, triple) in unasserted { + let mut reified = Map::new(); + reified.insert( + "@id".to_string(), + JsonValue::String(term_to_subject_key(&triple.s)), + ); + reified.insert(term_to_iri(&triple.p), term_to_object_value(&triple.o)); + nodes.push(json!({ "@id": reifier, "@reifies": reified })); + } Ok(JsonValue::Array(nodes)) } @@ -417,11 +433,34 @@ mod tests { } #[test] - fn reified_triple_in_subject_position_annotates_the_base_edge() { + fn reified_triple_names_its_unasserted_triple_with_reifies() { let json = parse("<< ex:alice ex:knows ex:bob >> ex:certainty 0.5 ."); - let alice = node(&json, "http://example.org/alice"); - let ann = &alice["http://example.org/knows"][0]["@annotation"]; - let reifier = ann["@id"].as_str().expect("blank reifier id"); + assert!( + json.as_array() + .unwrap() + .iter() + .all(|n| n["@id"] != "http://example.org/alice"), + "the reified triple is not asserted: {json:#}" + ); + let reifier = json + .as_array() + .unwrap() + .iter() + .find_map(|n| n.get("@reifies").and(n["@id"].as_str())) + .expect("a @reifies node"); + let reifies = json + .as_array() + .unwrap() + .iter() + .find_map(|n| n.get("@reifies")) + .unwrap(); + assert_eq!( + reifies, + &json!({ + "@id": "http://example.org/alice", + "http://example.org/knows": {"@id": "http://example.org/bob"} + }) + ); assert_eq!( node(&json, reifier)["http://example.org/certainty"][0]["@value"], "0.5" diff --git a/fluree-graph-turtle/src/parser.rs b/fluree-graph-turtle/src/parser.rs index 54537693af..a2172d65b5 100644 --- a/fluree-graph-turtle/src/parser.rs +++ b/fluree-graph-turtle/src/parser.rs @@ -1301,11 +1301,11 @@ impl<'a, 'input, S: GraphSink> Parser<'a, 'input, S> { ) } - /// `r rdf:reifies <<( s p o )>>` — the RDF 1.2 spelling every asserting + /// `r rdf:reifies <<( s p o )>>` — the RDF 1.2 spelling every reifying /// form desugars to, and the only star construct N-Triples/N-Quads have. /// The `<<(` token is current and `subject` is the reifier. Emits exactly - /// what `<< s p o ~ r >>` emits (base triple asserted, then the reifier - /// attachment), so both spellings produce one on-disk shape. + /// what `<< s p o ~ r >>` emits: the reifier attachment, without + /// asserting `s p o`. /// /// Grammar: `tripleTerm ::= '<<(' ttSubject predicate ttObject ')>>'`, /// `ttSubject ::= iri | BlankNode`, `ttObject ::= iri | BlankNode | @@ -1319,7 +1319,6 @@ impl<'a, 'input, S: GraphSink> Parser<'a, 'input, S> { let predicate = p.parse_predicate()?; let object = p.parse_tt_object()?; p.expect(&TokenKind::TripleTermEnd)?; - p.sink_emit_triple(subject, predicate, object)?; p.sink_emit_reified_triple(subject, predicate, object, reifier) })?; // An annotation tail here would reify the `rdf:reifies` triple @@ -1451,11 +1450,8 @@ impl<'a, 'input, S: GraphSink> Parser<'a, 'input, S> { p.expect(&TokenKind::ReifiedTripleEnd)?; - // Fluree's edge-annotation model reifies an asserted edge: emit the - // base triple, then the reifier attachment (documented divergence - // from RDF 1.2's non-asserting `<< >>`; see the roadmap's construct - // inventory). - p.sink_emit_triple(subject, predicate, object)?; + // A reified triple does not assert `s p o` (RDF 1.2); only the + // annotation syntax does. p.sink_emit_reified_triple(subject, predicate, object, reifier)?; Ok(reifier) @@ -2353,16 +2349,14 @@ mod tests { #[test] fn star_reified_triple_subject_position() { - // data-1 shape: assert base, mint anon reifier, reifier gets props. + // data-1 shape: mint anon reifier, reifier gets props; the reified + // triple is not asserted. let sink = parse_star(&format!("{P}<<:a :b :c>> :q :z .")); assert_eq!(sink.reified.len(), 1); let (s, p, o, r) = &sink.reified[0]; assert_eq!((s, p, o), (&iri("a"), &iri("b"), &iri("c"))); assert!(matches!(r, RecTerm::Blank(_)), "anon reifier: {r:?}"); - // Base triple asserted + reifier property triple. - assert!(sink.triples.contains(&(iri("a"), iri("b"), iri("c")))); - assert!(sink.triples.contains(&(r.clone(), iri("q"), iri("z")))); - assert_eq!(sink.triples.len(), 2); + assert_eq!(sink.triples, vec![(r.clone(), iri("q"), iri("z"))]); } #[test] @@ -2477,8 +2471,8 @@ mod tests { let (s, p, o, r) = &sink.reified[0]; assert_eq!((s, p, o), (&iri("a"), &iri("b"), &iri("c"))); assert_eq!(r, &iri("r")); - // Base triple asserted; no ordinary `rdf:reifies` triple emitted. - assert_eq!(sink.triples, vec![(iri("a"), iri("b"), iri("c"))]); + // Neither the triple nor an ordinary `rdf:reifies` triple emitted. + assert!(sink.triples.is_empty(), "{:?}", sink.triples); } #[test] @@ -2501,9 +2495,14 @@ mod tests { fn star_rdf_reifies_matches_tilde_spelling() { let prefix = format!("{P}PREFIX rdf: \n"); let a = parse_star(&format!("{prefix}:r rdf:reifies <<( :a :b :c )>> .")); - let b = parse_star(&format!("{prefix}:a :b :c ~ :r .")); + let b = parse_star(&format!("{prefix}<< :a :b :c ~ :r >> .")); + let annotated = parse_star(&format!("{prefix}:a :b :c ~ :r .")); assert_eq!(a.reified, b.reified); assert_eq!(a.triples, b.triples); + assert_eq!(a.reified, annotated.reified); + // Only the annotation syntax asserts the triple. + assert!(a.triples.is_empty()); + assert_eq!(annotated.triples, vec![(iri("a"), iri("b"), iri("c"))]); } #[test] @@ -2671,19 +2670,19 @@ mod tests { #[test] fn star_bare_reified_triple_statement() { - // `<< s p o >> .` with no predicate-object list: asserts the base - // triple and attaches a fresh anonymous reifier, nothing more. - let sink = parse_star(&format!("{P}:s :p :o .\n<<:s :p :o>> .")); + // `<< s p o >> .` with no predicate-object list attaches a fresh + // anonymous reifier, nothing more. + let sink = parse_star(&format!("{P}<<:s :p :o>> .")); assert_eq!(sink.reified.len(), 1); assert!(matches!(sink.reified[0].3, RecTerm::Blank(_))); - assert_eq!(sink.triples.len(), 2, "{:?}", sink.triples); + assert!(sink.triples.is_empty(), "{:?}", sink.triples); } #[test] fn star_collector_sink_records_reifications() { // The collector (the Turtle→JSON-LD path behind upsert, graph sync - // and memory import) accepts every asserting star form and keeps - // the reifier attachments alongside the triples. + // and memory import) accepts every star form and keeps the reifier + // attachments alongside the triples. let mut sink = GraphCollectorSink::new(); parse( &format!( @@ -2714,9 +2713,9 @@ mod tests { anon.iter().all(|t| matches!(t, Term::BlankNode(_))), "{anon:?}" ); - // Every base triple is asserted exactly once; body triples about - // the reifiers are ordinary triples. - assert_eq!(graph.len(), 5, "{:?}", graph.triples()); + // The two annotated triples are asserted, the reified one is not; + // body triples about the reifiers are ordinary triples. + assert_eq!(graph.len(), 4, "{:?}", graph.triples()); } #[test] diff --git a/fluree-graph-turtle/src/splitter.rs b/fluree-graph-turtle/src/splitter.rs index 093c6d9283..c6b810ba60 100644 --- a/fluree-graph-turtle/src/splitter.rs +++ b/fluree-graph-turtle/src/splitter.rs @@ -2020,6 +2020,7 @@ ex:carol ex:age 42 ~ ex:claim2 . panic!("chunk {i} must be valid Turtle-star: {e}\n{chunk_text}") }); for node in json.as_array().unwrap() { + annotated_edges += usize::from(node.get("@reifies").is_some()); for (key, values) in node.as_object().unwrap() { if key.starts_with('@') { continue; @@ -2444,6 +2445,7 @@ ex:t ex:u ex:v . let json = crate::parse_to_json(&text) .unwrap_or_else(|e| panic!("chunk {idx} must be valid Turtle-star: {e}\n{text}")); for node in json.as_array().unwrap() { + annotated += usize::from(node.get("@reifies").is_some()); for (key, values) in node.as_object().unwrap() { if key.starts_with('@') { continue; diff --git a/testsuite-sparql/src/rdf_handlers.rs b/testsuite-sparql/src/rdf_handlers.rs index f547c2cb40..51c4c06bbb 100644 --- a/testsuite-sparql/src/rdf_handlers.rs +++ b/testsuite-sparql/src/rdf_handlers.rs @@ -26,19 +26,16 @@ use crate::evaluator::TestEvaluator; use crate::files::read_file_to_string; use crate::manifest::Test; use crate::result_comparison::{are_results_isomorphic, format_results_diff}; -use crate::result_format::{ir_term_to_rdf_term, RdfTerm, SparqlResults, Triple}; +use crate::result_format::{ + ir_term_to_rdf_term, RdfTerm, SparqlResults, Triple, REIFIES_OBJECT, REIFIES_PREDICATE, + REIFIES_SUBJECT, +}; use crate::vocab::rdft; const RDF_FIRST: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#first"; const RDF_REST: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#rest"; const RDF_NIL: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#nil"; -/// Stand-in predicates that spell a reifier attachment as ordinary triples, -/// so the isomorphism check sees which triple each reifier names. -const REIFIES_SUBJECT: &str = "urn:fluree:testsuite:reifies-subject"; -const REIFIES_PREDICATE: &str = "urn:fluree:testsuite:reifies-predicate"; -const REIFIES_OBJECT: &str = "urn:fluree:testsuite:reifies-object"; - /// Register handlers for every `rdft:` test type the Turtle parser can serve. pub fn register_rdf_tests(evaluator: &mut TestEvaluator) { evaluator.register(rdft::TEST_TURTLE_POSITIVE_SYNTAX, evaluate_positive_syntax); @@ -107,9 +104,8 @@ fn evaluate_negative_syntax(test: &Test) -> Result<()> { /// Both documents go through the same parser, reifier attachments included: /// the expected `.nt` spells each one `r rdf:reifies <<( s p o )>>`, which /// the parser reads as the same attachment an action's `<< s p o >>` or -/// `{| |}` produces. The parser also asserts `s p o` on both sides (Fluree -/// reifies asserted edges), so a pass means the action desugars to the -/// expected attachments under that model, not that the base triple is absent. +/// `{| |}` produces. Only the annotation syntax asserts `s p o`, on either +/// side. fn evaluate_eval(test: &Test) -> Result<()> { let url = action_url(test)?; let result_url = test @@ -218,9 +214,7 @@ fn graph_to_rdf_triples(graph: &Graph) -> Vec { }); } } - // An RDF graph is a set. The parser emits a base triple again for each - // `rdf:reifies <<( s p o )>>`, so an expected graph that also states - // `s p o .` would otherwise differ from the action by a duplicate. + // An RDF graph is a set. let mut seen = HashSet::new(); out.retain(|t| seen.insert(t.clone())); out diff --git a/testsuite-sparql/src/result_format.rs b/testsuite-sparql/src/result_format.rs index db9d496de3..d62da400ae 100644 --- a/testsuite-sparql/src/result_format.rs +++ b/testsuite-sparql/src/result_format.rs @@ -695,15 +695,53 @@ pub fn parse_expected_graph(url: &str) -> Result> { let mut sink = GraphCollectorSink::new(); parse_turtle(&with_base, &mut sink) .with_context(|| format!("Parsing expected graph: {url}"))?; - let graph = sink.into_graph(); - Ok(graph + Ok(graph_triples(&sink.into_graph())) +} + +/// Stand-in predicates that spell a reifier attachment as ordinary triples, +/// so the isomorphism check sees which triple each reifier names. +pub(crate) const REIFIES_SUBJECT: &str = "urn:fluree:testsuite:reifies-subject"; +pub(crate) const REIFIES_PREDICATE: &str = "urn:fluree:testsuite:reifies-predicate"; +pub(crate) const REIFIES_OBJECT: &str = "urn:fluree:testsuite:reifies-object"; + +/// `reifier`'s attachment to `(s, p, o)` as its three stand-in triples. +pub(crate) fn reification_triples( + reifier: RdfTerm, + s: RdfTerm, + p: RdfTerm, + o: RdfTerm, +) -> [Triple; 3] { + [ + (REIFIES_SUBJECT, s), + (REIFIES_PREDICATE, p), + (REIFIES_OBJECT, o), + ] + .map(|(stand_in, term)| Triple { + subject: reifier.clone(), + predicate: RdfTerm::Iri(stand_in.to_string()), + object: term, + }) +} + +/// A parsed graph's triples, with each reification as its stand-in triples. +fn graph_triples(graph: &IrGraph) -> Vec { + let mut out: Vec = graph .iter() .map(|t| Triple { subject: ir_term_to_rdf_term(&t.s), predicate: ir_term_to_rdf_term(&t.p), object: ir_term_to_rdf_term(&t.o), }) - .collect()) + .collect(); + for r in graph.reifications() { + out.extend(reification_triples( + ir_term_to_rdf_term(&r.reifier), + ir_term_to_rdf_term(&r.triple.s), + ir_term_to_rdf_term(&r.triple.p), + ir_term_to_rdf_term(&r.triple.o), + )); + } + out } // --------------------------------------------------------------------------- @@ -733,15 +771,7 @@ fn parse_ttl_result(content: &str, url: &str) -> Result { if is_result_set { parse_dawg_result_set_from_graph(&graph) } else { - let triples: Vec = graph - .iter() - .map(|t| Triple { - subject: ir_term_to_rdf_term(&t.s), - predicate: ir_term_to_rdf_term(&t.p), - object: ir_term_to_rdf_term(&t.o), - }) - .collect(); - Ok(SparqlResults::Graph(triples)) + Ok(SparqlResults::Graph(graph_triples(&graph))) } } @@ -1145,7 +1175,9 @@ impl JsonLdContext { /// /// Expects a JSON-LD `@graph` array (or a single node object). Each node has /// `@id` as the subject; every other key is a predicate whose values are objects. -/// Compact IRIs are expanded against the result's `@context`. +/// Compact IRIs are expanded against the result's `@context`. A value's +/// `@annotation` and a node's `@reifies` become reification stand-in triples, +/// as the expected graph's reifications do. pub fn fluree_construct_to_sparql_results(json: &serde_json::Value) -> Result { let ctx = JsonLdContext::parse(json); let nodes = if let Some(graph) = json.get("@graph").and_then(|g| g.as_array()) { @@ -1178,6 +1210,31 @@ pub fn fluree_construct_to_sparql_results(json: &serde_json::Value) -> Result Result Result Vec { + match value { + serde_json::Value::Array(values) => values.clone(), + other => vec![other.clone()], + } +} + +/// A node identifier: a blank node label or an expanded IRI. +fn id_term(id: &str, ctx: &JsonLdContext) -> RdfTerm { + match id.strip_prefix("_:") { + Some(label) => RdfTerm::BlankNode(label.to_string()), + None => RdfTerm::Iri(ctx.expand_id(id)), + } +} + /// Convert a JSON-LD value node to an [`RdfTerm`]. /// /// Handles `{"@id": "..."}`, `{"@value": "...", "@type": "...", "@language": "..."}`, diff --git a/testsuite-sparql/tests/registers/mod.rs b/testsuite-sparql/tests/registers/mod.rs index 94f1839305..ad413db875 100644 --- a/testsuite-sparql/tests/registers/mod.rs +++ b/testsuite-sparql/tests/registers/mod.rs @@ -409,27 +409,19 @@ pub const SPARQL12_EVAL_TRIPLE_TERMS: &[&str] = &[ "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#order-1", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#order-2", // data load: blocked on TriG GRAPH-block parsing, orthogonal to star - // (D-8) — "expected subject, found 'GRAPH'" (4) + // (D-8) — "expected subject, found 'GRAPH'" (3) "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#expr-1", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#graphs-2", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#update-1", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#update-2", - // results not isomorphic: the star data LOADS, but Fluree's model - // asserts the base triple of every `<< s p o >>` — RDF 1.2's - // non-asserting reified triples (3) - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#pattern-8-nomatch", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#results-reifiedtriples-1j", + // SRX results: the harness reads no `` result term (1) "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#results-reifiedtriples-1x", // SPARQL lowering: "RDF-star quoted triples in this position lowering is // not yet implemented" (CONSTRUCT templates) (2) "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#construct-1", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#construct-3", - // D-4 CONSTRUCT annotation projection: lowering returns "CONSTRUCT - // projection of edge-annotation metadata is not supported in v1" (1) - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#construct-4", - // results not isomorphic: the expected graph (`r rdf:reifies <<( … )>>`) - // now parses, but CONSTRUCT emits the annotation body without the - // reifier's attachment — D-4 annotation projection (1) + // results not isomorphic: CONSTRUCT WHERE reuses the matched reifier for + // an anonymous `{| |}`, where the template mints a fresh blank node per + // solution (1) "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#construct-5", // SPARQL lowering: triple-term values in VALUES data not implemented (1) "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#expr-2", From 8e53cb07a0704b463f6f717960b3b7f034e663b2 Mon Sep 17 00:00:00 2001 From: bplatz Date: Sat, 3 Oct 2026 09:32:55 -0400 Subject: [PATCH 57/92] feat(transact): deleting a triple leaves its reifiers outside LPG mode RDF 1.2 deletes nothing but the named triples: a reifier's link and body outlive the triple it reifies. Outside LPG mode a transaction no longer retracts links whose base edge it retracts, nor bodies of blank reifiers left without a link, nor links of reifiers whose whole body it retracts. The annotation syntax joins the base edge, so it stops matching such claims; the reified-triple form still finds them. LPG mode (`opts.lpgEdgeLifecycle`, which Cypher DELETE sets) keeps the relationship lifecycle: retracting an edge retracts its links, and a reifier left with no link loses its body. --- docs/concepts/edge-annotations.md | 75 ++-- docs/design/README.md | 2 +- docs/design/edge-annotations.md | 11 +- docs/guides/README.md | 2 +- docs/guides/cookbook-edge-annotations.md | 19 +- docs/transactions/insert.md | 2 +- docs/transactions/retractions.md | 20 +- docs/transactions/turtle.md | 2 +- fluree-db-api/tests/it_edge_annotations.rs | 417 ++++++++---------- .../tests/it_edge_annotations_indexed.rs | 81 +--- fluree-db-api/tests/it_import_turtle_star.rs | 10 +- fluree-db-api/tests/it_tracing_spans.rs | 7 +- fluree-db-api/tests/it_triple_term_links.rs | 5 +- fluree-db-transact/src/stage.rs | 96 +--- 14 files changed, 275 insertions(+), 474 deletions(-) diff --git a/docs/concepts/edge-annotations.md b/docs/concepts/edge-annotations.md index b41ab24782..036f58aca3 100644 --- a/docs/concepts/edge-annotations.md +++ b/docs/concepts/edge-annotations.md @@ -25,7 +25,7 @@ If a fact is naturally about a *node* (Alice's birthdate, Acme's industry), put | Surface | How | Notes | |---|---|---| -| **JSON-LD insert / upsert / update** | `@annotation` (or alias `@edge`) on a value object | Most ergonomic. Covers literal-valued edges (with explicit `@type` / `@language`), parallel annotations, named reifiers, body cascades, and **named-graph edges** (an annotation on an edge inside a named graph is written into that same graph, keeping the edge's graph identity). A node's `@reifies` (`{"@id": s, p: o}`, or an array of them) makes the node a reifier of that triple **without asserting it**, as `r rdf:reifies <<( s p o )>>` does. | +| **JSON-LD insert / upsert / update** | `@annotation` (or alias `@edge`) on a value object | Most ergonomic. Covers literal-valued edges (with explicit `@type` / `@language`), parallel annotations, named reifiers, and **named-graph edges** (an annotation on an edge inside a named graph is written into that same graph, keeping the edge's graph identity). A node's `@reifies` (`{"@id": s, p: o}`, or an array of them) makes the node a reifier of that triple **without asserting it**, as `r rdf:reifies <<( s p o )>>` does. | | **SPARQL 1.2 UPDATE** | `INSERT DATA { :s :p :o {\| ... \|} }`, `~ `, optional `INSERT { } WHERE { }` templates | Use this when integrating with SPARQL pipelines or when porting from RDF 1.2 / SPARQL-star. **Default graph only:** an annotation tail inside an explicit `GRAPH { }` block, or under a `WITH ` template, is rejected — use the JSON-LD surface or TriG-star for named-graph edge annotations. See [SPARQL 1.2 surface](#sparql-12--rdf-12-surface) below for the per-operation rules. | | **Turtle / N-Triples / TriG / N-Quads ingest** (`insert`, `upsert`, bulk `import`, graph sync (`fluree sync`, `/sync`), memory import; TriG via `insert` / `upsert` / `import` / `fluree sync` / `/sync`) | RDF 1.2 forms: the annotation syntax `:s :p :o ~ {\| ... \|}`, `<< :s :p :o ~ :r >>` in subject or object position, and the canonical `:r rdf:reifies <<( :s :p :o )>>` (the only star spelling N-Triples and N-Quads have) — in the default graph and inside TriG `GRAPH { }` blocks alike | Same stored link as `@annotation`. An annotation inside a `GRAPH { }` block, or on an N-Quads statement with a graph label, is written into that graph with the edge's graph identity, exactly as JSON-LD `@graph` + `@annotation` does. As in RDF 1.2, only the annotation syntax asserts the triple: `<< s p o >>` and `rdf:reifies <<( s p o )>>` reify it without asserting it, so N-Triples and N-Quads state an annotated triple as the triple plus its reifier's link. Rejected with a specific error: `<<( ... )>>` anywhere other than the object of `rdf:reifies`, nested triple terms, star constructs inside an annotation body, annotations on collections, and annotations in a TriG `<#txn-meta>` block. Paths that convert to JSON-LD first (`upsert`, `graph sync`, memory import) also reject an annotation on an `rdf:type` edge. See [Turtle ingest](../transactions/turtle.md#edge-annotations-rdf-12--turtle-star). | @@ -282,27 +282,28 @@ Querying without binding the annotation (`?person ex:worksFor ?org`) returns one ## Anonymous vs explicit annotation IDs -The two forms have deliberately different lifecycle behavior. The default is conservative: anonymous annotations behave like LPG edge properties; explicit-IRI annotations behave like ordinary RDF resources. +The two forms differ in visibility: an anonymous annotation is an edge property you reach through its edge, while an explicit-IRI annotation is an ordinary RDF resource. | | Anonymous (no `@id`) | Explicit `@id` | |---|---|---| | Visible in `select: "*"` | No — hidden from wildcard subject expansion | Yes | | Visible in graph crawl | Only via `@annotation` projection | Yes, like any subject | -| Retract base edge → owned facts cascade | Yes (the annotation is intrinsic to the edge) | No, by default — explicit IRIs are not deleted surprisingly | +| Retract base edge → link and body removed | Only in LPG mode | Only in LPG mode | The anonymous-hide rule means a user wildcard query against Alice doesn't suddenly start returning a sea of internal annotation SIDs once you adopt edge metadata. Annotations participate in queries that ask for them and stay out of the way otherwise. -The explicit-ID-doesn't-cascade rule protects user-named resources from accidental deletion when an edge gets retracted. Opt out via *LPG mode* (below) when you actually want property-graph "delete the relationship deletes its properties" semantics. - ## Retraction semantics -### Which spelling does what +A transaction retracts the triples it names and nothing else, as RDF 1.2 defines it: deleting a triple does not delete the statements of a reifier that reifies it. A reifier's link (`r rdf:reifies <<( s p o )>>`) is a triple of its own, so it outlives the edge, and so does its body. Two read surfaces then disagree, on purpose: -Two independent rules meet here, and the combination surprises people. +- The **annotation syntax** (`s p o {| … |}`, `~`, JSON-LD `@annotation`, a Cypher relationship) asserts its triple, so it stops matching a claim once the edge is gone. +- The **reified-triple form** (`<< s p o ~ ?r >>`, `?r rdf:reifies <<( … )>>`, JSON-LD `@reifies`) reads the link alone and still finds it. -**1. The annotation form asserts the base triple, so deleting it retracts the base edge.** Both `s p o ~ :r {| … |}` and the bare `s p o ~ :r` expand to the base triple *plus* the reification: RDF 1.2 Turtle §2.11.1 defines the syntax as one that both reifies **and asserts** a triple, and SPARQL 1.2 Update §3.1.2 admits the same production into `DELETE DATA`. So `DELETE DATA { :alice :knows :bob ~ :claim1 {| … |} }` is a base-edge retraction. This is what the specs require — a store that kept the edge here would be the one diverging. +Opt into the property-graph lifecycle with [LPG mode](#lpg-mode-opt-in-per-transaction) when deleting an edge should delete its claims. + +### Which spelling does what -**2. Retracting a base edge cascades to every reifier attached to it**, including reifiers the delete never named. This rule is Fluree's own. Neither RDF 1.2 nor SPARQL 1.2 entails it, and SPARQL 1.2 Update §3.1.2 Example 6 makes the converse point — deleting a *reifying* triple leaves the asserted triple in place. Fluree cascades because an edge's claims should not outlive the edge it describes. It is stated here so that a reader who checks the spec, finds Fluree retracting more than was named, and concludes there is a second bug, knows it is deliberate. +**The annotation form asserts the base triple, so deleting it retracts the base edge.** Both `s p o ~ :r {| … |}` and the bare `s p o ~ :r` expand to the base triple *plus* the reification: RDF 1.2 Turtle §2.11.1 defines the syntax as one that both reifies **and asserts** a triple, and SPARQL 1.2 Update §3.1.2 admits the same production into `DELETE DATA`. So `DELETE DATA { :alice :knows :bob ~ :claim1 {| … |} }` retracts the edge, `:claim1`'s link and its body. Seeded with one edge and two independent claims about it: @@ -311,51 +312,33 @@ Seeded with one edge and two independent claims about it: :alice :knows :bob ~ :claim2 {| :confidence 0.6 ; :source :sourceB |} . ``` -| you write | edge | `:claim1` body | `:claim2` body | still attached | -| --- | --- | --- | --- | --- | -| `DELETE DATA { :alice :knows :bob ~ :claim1 {\| :confidence 0.8 ; :source :sourceA \|} }` | **gone** | gone | survives | none | -| `DELETE DATA { :alice :knows :bob ~ :claim1 }` | **gone** | **survives** | **survives** | none | -| `DELETE WHERE { :alice :knows :bob ~ ?c {\| :confidence ?f \|} }` | **gone** | `:source` only | `:source` only | none | -| `DELETE DATA { :alice :knows :bob }` | **gone** | survives | survives | none | -| `upsert` restating the edge with a different object | object replaced | survives | survives | none | -| `DELETE DATA { :claim1 :confidence 0.8 ; :source :sourceA }` | survives | gone | survives | `:claim2` | -| JSON-LD `delete` with `"@annotation": {"@id": ":claim1"}` | survives | survives | survives | `:claim2` | -| `DELETE DATA { :alice :knows :bob {\| … \|} }` | *refused* — an anonymous block has no addressable identity to delete | | | | +| you write | edge | `:claim1` body | `:claim2` body | still linked | annotation syntax matches | +| --- | --- | --- | --- | --- | --- | +| `DELETE DATA { :alice :knows :bob ~ :claim1 {\| :confidence 0.8 ; :source :sourceA \|} }` | **gone** | gone | survives | `:claim2` | none | +| `DELETE DATA { :alice :knows :bob ~ :claim1 }` | **gone** | survives | survives | `:claim2` | none | +| `DELETE WHERE { :alice :knows :bob ~ ?c {\| :confidence ?f \|} }` | **gone** | `:source` only | `:source` only | none | none | +| `DELETE DATA { :alice :knows :bob }` | **gone** | survives | survives | both | none | +| `upsert` restating `:claim1` on a different object | object replaced | survives | survives | `:claim2` | none | +| `DELETE DATA { :claim1 :confidence 0.8 ; :source :sourceA }` | survives | gone | survives | both | `:claim2` | +| JSON-LD `delete` with `"@annotation": {"@id": ":claim1"}` | survives | survives | survives | `:claim2` | `:claim2` | +| `DELETE DATA { :alice :knows :bob {\| … \|} }` | *refused* — an anonymous block has no addressable identity to delete | | | | | Three rows deserve calling out: -- **`~ :claim1` with no body block is the sharpest edge in the table.** It reads like "detach claim1" and does close to the opposite: the edge goes, *both* claims are detached, and *both* bodies are left standing — well-formed RDF about reifiers that no longer reify anything. -- **A variable reifier matches every claim on the edge.** `~ ?c {| :confidence ?f |}` strips the body properties the block names from *all* of them. The result is not "claim1 withdrawn" but "every claim partially gutted, and the edge gone". -- **The cascade is not delete-specific.** An `upsert` that changes the object retracts the old edge and fires the identical cascade, with no delete written anywhere. +- **`~ :claim1` with no body block deletes the edge.** It reads like "detach claim1", but the annotation form asserts the triple, so the delete retracts it too, and the annotation syntax stops matching `:claim2` as well. +- **A variable reifier matches every claim on the edge.** `~ ?c {| :confidence ?f |}` retracts the link and the body properties the block names, from *all* of them, and the edge with them. +- **An `upsert` that changes the object retracts the old edge** with no delete written anywhere, and the claims left on it stop matching the annotation syntax. So: -- To **withdraw one claim**, retract its body facts — `DELETE DATA { :claim1 :confidence 0.8 ; :source :sourceA }`. The edge and every other claim stay put, and the now-empty attachment is retired for you. -- To **detach one claim but keep its body** as ordinary RDF, use the JSON-LD `@annotation` delete. It is the only spelling that means exactly that. -- To **remove the edge and everything about it**, delete the base edge and set `opts.lpgEdgeLifecycle: true` (see LPG mode below) so the bodies go too. - -### RDF mode (default) - -Retracting a base edge removes the attachment and any owned facts on **anonymous** annotations. Explicit-IRI annotations keep their non-attachment facts — only the attachment row is retracted. - -```json -{ - "delete": { - "@id": "ex:alice", - "ex:worksFor": { "@id": "ex:acme" } - } -} -``` - -After this: -- Anonymous `_:annN` subjects attached to the edge: gone (attachment + body). -- Explicit `ex:employment/alice-acme-2024`: attachment retracted, but `ex:role`, `ex:since`, etc. are still in the graph as ordinary RDF. +- To **withdraw one claim** and keep the edge, retract its link with the JSON-LD `@annotation` delete; retract its body as well if it should go. +- To **remove the edge and everything about it**, delete the edge in [LPG mode](#lpg-mode-opt-in-per-transaction). -History preserves both events — query at the pre-retract `t` and the annotation comes back, unchanged. +History preserves every event — query at the pre-retract `t` and the annotation comes back, unchanged. ### LPG mode (opt-in per transaction) -For property-graph relationship lifecycle — "deleting the relationship deletes the relationship's properties" — set `lpgEdgeLifecycle: true` in transaction options: +For property-graph relationship lifecycle — "deleting the relationship deletes the relationship's properties" — set `lpgEdgeLifecycle: true` in transaction options. Cypher `DELETE` sets it. ```json { @@ -367,7 +350,7 @@ For property-graph relationship lifecycle — "deleting the relationship deletes } ``` -Now explicit-IRI annotations cascade their owned metadata too. +Retracting the edge now retracts every link naming it, and a reifier left with no link loses its body, whether it is anonymous or has an explicit `@id`. ### Updating annotation properties @@ -583,7 +566,7 @@ The mandated SPARQL 1.2 `VERSION "1.2"` prologue declaration is **accepted** (le - An annotation is its `rdf:reifies` link plus its properties; a ledger with no annotations stores none and pays nothing for them. - Plain triple queries take exactly the same plan they did before annotations existed; the planner only reads links when the query mentions `@annotation`, `@reifies`, `rdf:reifies` or a reified triple. - Annotation properties are stored as ordinary RDF facts. Time travel, policy, history, export, and reasoning all work on them without special cases. -- Retraction cascade has a fast path: when both the index root and current novelty know the ledger has no annotations, base-edge retracts skip the attachment lookup entirely. +- The LPG-mode retraction cascade has a fast path: when both the index root and current novelty know the ledger has no annotations, base-edge retracts skip the link lookup entirely. - An index built by an earlier Fluree version holds annotations in the `f:reifies*` form and no links, so on an annotated ledger annotation queries fail with an error asking for a rebuild until it is reindexed (`fluree reindex `). The next index build after a new annotation is written is a full rebuild, which links them all. For the term dictionary and the transaction-time rules, see the [Edge annotations design doc](../design/edge-annotations.md). diff --git a/docs/design/README.md b/docs/design/README.md index fe87f0b711..e794dcede0 100644 --- a/docs/design/README.md +++ b/docs/design/README.md @@ -42,7 +42,7 @@ Binary columnar index format: branch/leaf/leaflet hierarchy, dictionary artifact ### [Edge annotations (storage internals)](edge-annotations.md) -Storage of RDF 1.2 edge annotations: the `rdf:reifies` link as the record, the term dictionary that indexes it, the transaction-time cascade, and ledgers written before links. (User-facing contract lives in [Edge annotations (concept doc)](../concepts/edge-annotations.md).) +Storage of RDF 1.2 edge annotations: the `rdf:reifies` link as the record, the term dictionary that indexes it, the LPG-mode transaction-time cascade, and ledgers written before links. (User-facing contract lives in [Edge annotations (concept doc)](../concepts/edge-annotations.md).) ### [Spatial Index](spatial-index.md) diff --git a/docs/design/edge-annotations.md b/docs/design/edge-annotations.md index ddc0cfbf8a..c0e42b0ac6 100644 --- a/docs/design/edge-annotations.md +++ b/docs/design/edge-annotations.md @@ -37,13 +37,14 @@ Query lowering of annotation patterns is described in [Annotation patterns read ## Transaction-time rules -`cascade_attachment_retracts` keeps links pointing at live edges: +A transaction retracts what it names. RDF 1.2 deletes nothing else when a triple goes: a reifier's link and body outlive the triple it reifies, and the annotation syntax, which joins the triple, stops matching them. -1. A retracted triple retracts every link naming it (a POST probe on `rdf:reifies` with the term as the object). -2. A transaction that retracts all of a reifier's body retracts its links. -3. A reifier left with no link loses its body when it is a blank node, or in LPG mode (`opts.lpgEdgeLifecycle`, which Cypher `DELETE` sets); an IRI reifier's body otherwise stays as ordinary RDF. +LPG mode (`opts.lpgEdgeLifecycle`, which Cypher `DELETE` sets) adds the property-graph relationship lifecycle in `cascade_attachment_retracts`: -The cascade is a Fluree rule, not an entailment: RDF 1.2 does not delete a reifier's statements when the triple it reifies is deleted. Annotation-syntax reads rely on it (see below). A ledger that has never held an annotation pays nothing: `snapshot.has_annotations` and `Novelty::has_annotations` gate the pass. +1. A retracted triple retracts every link naming it (a POST probe on `rdf:reifies` with the term as the object, in the triple's graph). +2. A reifier left with no link loses its body. + +A ledger that has never held an annotation pays nothing: `snapshot.has_annotations` and `Novelty::has_annotations` gate the pass. A reifier may reify several triples. Re-pointing one is a retract of the old link and an assert of the new; a JSON-LD upsert does that for a reifier it names. In LPG mode, an empty `@annotation: {}` mints a fresh property-less reifier so the relationship keeps an identity; in RDF mode it writes nothing. diff --git a/docs/guides/README.md b/docs/guides/README.md index 197e379b74..9eef5a82c4 100644 --- a/docs/guides/README.md +++ b/docs/guides/README.md @@ -54,4 +54,4 @@ Define data quality constraints: required properties, datatype validation, value ### [Edge Annotations](cookbook-edge-annotations.md) -Attach properties to a relationship: model property-graph edges, record statement-level provenance, represent parallel relationships, query inline or annotation-rooted, and understand the retract cascade — in JSON-LD and SPARQL 1.2. +Attach properties to a relationship: model property-graph edges, record statement-level provenance, represent parallel relationships, query inline or annotation-rooted, and understand what a retraction leaves — in JSON-LD and SPARQL 1.2. diff --git a/docs/guides/cookbook-edge-annotations.md b/docs/guides/cookbook-edge-annotations.md index 6d18a6a515..614a5ac3aa 100644 --- a/docs/guides/cookbook-edge-annotations.md +++ b/docs/guides/cookbook-edge-annotations.md @@ -170,9 +170,9 @@ Once you've bound the annotation — by `@id` or by selector — it's an ordinar } ``` -## Retract an edge — and understand the cascade +## Retract an edge — and understand what stays -Retracting the base edge cascades to the annotation's attachment. What happens to the annotation's *body* depends on the mode. +Retracting the base edge retracts that triple and nothing else, as RDF 1.2 defines it: the annotation's link and body stay, and the annotation syntax (`@annotation`, `{| |}`) stops matching it because it joins the edge. `@reifies` still finds it. ```json { @@ -183,8 +183,7 @@ Retracting the base edge cascades to the annotation's attachment. What happens t } ``` -- **RDF mode (default):** anonymous annotation subjects on the edge are fully removed (attachment + body). Explicit-IRI annotations keep their body facts as ordinary RDF — only the attachment is retracted, so a user-named resource is never deleted by surprise. -- **LPG mode (`opts.lpgEdgeLifecycle: true`):** explicit-IRI annotations cascade their body too — the property-graph "delete the relationship deletes its properties" lifecycle. +To delete the edge's annotations with it — the property-graph "delete the relationship deletes its properties" lifecycle — set LPG mode: ```json { @@ -193,7 +192,7 @@ Retracting the base edge cascades to the annotation's attachment. What happens t } ``` -History preserves both events either way — query at the pre-retract `t` and the annotation comes back. See [Retractions](../transactions/retractions.md#edge-annotation-cascade) for the metadata-only-retract and same-transaction-replacement rules. +History preserves every event either way — query at the pre-retract `t` and the annotation comes back. See [Retractions](../transactions/retractions.md#edge-annotation-cascade). ## The same patterns in SPARQL 1.2 @@ -291,12 +290,12 @@ SPARQL The natural-looking SPARQL form deletes more than the claim: ```sparql -# Retracts the base edge ex:alice ex:knows ex:bob — and with it the -# attachment of EVERY claim on that edge, not only ex:claim1. +# Retracts the base edge ex:alice ex:knows ex:bob — so the annotation +# syntax stops matching EVERY claim on that edge, not only ex:claim1. DELETE DATA { ex:alice ex:knows ex:bob ~ ex:claim1 {| ex:confidence 0.9 |} . } ``` -`DELETE DATA` / `DELETE WHERE` with an annotation tail always retract the base edge, and the edge retract cascades to all of its annotations ([Retractions](../transactions/retractions.md#edge-annotation-cascade)). SPARQL has no form for "retract this one claim, keep the edge". Use the JSON-LD by-id retract, which removes exactly one attachment: +`DELETE DATA` / `DELETE WHERE` with an annotation tail always retract the base edge ([Retractions](../transactions/retractions.md#edge-annotation-cascade)). To retract one claim and keep the edge, retract its link — the JSON-LD by-id retract, which removes exactly one attachment: ```json { @@ -308,12 +307,12 @@ DELETE DATA { ex:alice ex:knows ex:bob ~ ex:claim1 {| ex:confidence 0.9 |} . } } ``` -The edge and every other claim on it stay live. In RDF mode the named claim's body (`ex:confidence`, `ex:source`) survives as ordinary RDF about `ex:claim1` — retract it in the same transaction if it should go too; in LPG mode (`opts.lpgEdgeLifecycle: true`) the body is removed with the attachment. +The edge and every other claim on it stay live. The named claim's body (`ex:confidence`, `ex:source`) survives as ordinary RDF about `ex:claim1` — retract it in the same transaction if it should go too; in LPG mode (`opts.lpgEdgeLifecycle: true`) the body is removed with the attachment. ## Gotchas - **One annotation `@id` may reify several triples**, and a single edge carries many parallel annotations. To re-home an explicit-IRI annotation, retract the old attachment and assert the new one in the same transaction; a JSON-LD upsert of the annotation does that for you. -- **Deleting a claim with `DELETE DATA { … ~ :claim {| … |} }` deletes the edge** and detaches every other claim on it. Retract one claim with the JSON-LD by-id form (see [above](#retract-one-claim-and-keep-the-edge)). +- **Deleting a claim with `DELETE DATA { … ~ :claim {| … |} }` deletes the edge**, so the annotation syntax stops matching every other claim on it. Retract one claim with the JSON-LD by-id form (see [above](#retract-one-claim-and-keep-the-edge)). - **Don't write `f:reifies*` predicates by hand.** They're reserved and rejected on every write surface; they're also hidden from `?p` scans and `select: "*"`. Use `@annotation` / the annotation tail. (See [Vocabulary](../reference/vocabulary.md#edge-annotation-predicates-reserved).) - **Empty `@annotation: {}`** is a no-op in RDF mode (no subject minted); in LPG mode it mints a property-less relationship with identity. - **Not yet supported** (all reject cleanly, no silent partial results): annotations on `@list` elements, triple terms as object values, annotation output in Turtle/CONSTRUCT, and the SPARQL 1.2 triple-term functions (`TRIPLE`, `isTRIPLE`, …). See [Current limits](../concepts/edge-annotations.md#current-limits). diff --git a/docs/transactions/insert.md b/docs/transactions/insert.md index 4777d1ea78..1385a5e235 100644 --- a/docs/transactions/insert.md +++ b/docs/transactions/insert.md @@ -382,7 +382,7 @@ Inline `@annotation` queries return one row per occurrence. - Annotation-of-annotation (nested `@annotation` inside an annotation body). - Hand-authored mention of the [reserved system predicates](../reference/vocabulary.md#edge-annotation-predicates-reserved) that back annotations (compact or full IRI form). -For the full surface — including SPARQL 1.2 / RDF 1.2 annotation tails (`{| |}`), the named reifier (`~`), the cardinality / multiplicity contract, anonymous vs explicit-IRI lifecycle, and named-graph behavior — see the [Edge annotations concept doc](../concepts/edge-annotations.md). For cascade semantics when a base edge or annotation metadata is removed, see [Retractions](retractions.md). +For the full surface — including SPARQL 1.2 / RDF 1.2 annotation tails (`{| |}`), the named reifier (`~`), the cardinality / multiplicity contract, anonymous vs explicit-IRI lifecycle, and named-graph behavior — see the [Edge annotations concept doc](../concepts/edge-annotations.md). For what a retraction leaves, and the LPG-mode cascade, see [Retractions](retractions.md#edge-annotation-cascade). ## Turtle and TriG diff --git a/docs/transactions/retractions.md b/docs/transactions/retractions.md index 9053a38d56..b374ecb79f 100644 --- a/docs/transactions/retractions.md +++ b/docs/transactions/retractions.md @@ -197,25 +197,13 @@ Deletes order and all its items. ## Edge-Annotation Cascade -When a transaction retracts a base edge that has annotations attached (see [Insert: Edge Annotations](insert.md#edge-annotations)), the transactor automatically retracts the link that attaches each annotation to that edge. Without this cascade, retracted edges would still surface their annotations through `@reifies` queries. +A transaction retracts the triples it names and nothing else (see [Insert: Edge Annotations](insert.md#edge-annotations)). As RDF 1.2 defines it, deleting a triple does not delete the statements of a reifier that reifies it: the reifier's `rdf:reifies` link and its body outlive the edge. The annotation syntax (`{| |}`, `~`, JSON-LD `@annotation`, Cypher relationships) joins the edge, so it stops matching those claims; the reified-triple form (`<< s p o ~ ?r >>`, `@reifies`) still finds them. -**Base-edge retract** — fires on every annotated retract: +**The annotation form of a *delete* is a base-edge retract.** `DELETE DATA { :alice :knows :bob ~ :claim1 {| … |} }` — and the bare `~ :claim1` tail with no block — expand to include the base triple, because RDF 1.2 annotation syntax both reifies *and asserts* the triple it annotates. The annotation syntax therefore stops matching **every** claim on that edge, not only the one named. An `upsert` that changes an annotated edge's object does the same thing, with no delete written at all. See [Which spelling does what](../concepts/edge-annotations.md#which-spelling-does-what) for the full table. -- The attachment linking each currently-asserted annotation to the edge is retracted in the same transaction. -- Anonymous (blank-node) annotation subjects also have their body metadata retracted, since the synthetic SID is unaddressable once the attachment is gone. -- Explicit-IRI annotation subjects keep their body metadata as ordinary RDF on the named subject (default RDF mode). To extend cleanup to explicit-IRI annotations as well, set `opts.lpgEdgeLifecycle: true` on the transaction — this matches the property-graph relationship lifecycle. +**LPG mode** (`opts.lpgEdgeLifecycle: true`, which Cypher `DELETE` sets) adds the property-graph relationship lifecycle: retracting an edge retracts every link naming it, and a reifier left with no link loses its body. The cascade is graph-aware: named-graph links are retracted in the same named graph as the edge they reify. -**The annotation form of a *delete* is a base-edge retract.** `DELETE DATA { :alice :knows :bob ~ :claim1 {| … |} }` — and the bare `~ :claim1` tail with no block — expand to include the base triple, because RDF 1.2 annotation syntax both reifies *and asserts* the triple it annotates. They therefore fire the cascade above against **every** claim on that edge, not only the one named, and a sibling claim is left with its body intact but nothing to attach to. An `upsert` that changes an annotated edge's object does the same thing, with no delete written at all. See [Which spelling does what](../concepts/edge-annotations.md#which-spelling-does-what) for the full table, including the two spellings that withdraw or detach a single claim. - -**Metadata-only retract** — fires when the user retracts every body fact of an annotation subject without touching the base edge: - -- The attachment is also retracted, so the annotation is fully disposed of and inline `@annotation` queries no longer surface it. -- Same-transaction replacements (delete one body fact, insert another on the same annotation in a single update) keep the attachment — the post-transaction metadata set is non-empty, so the cascade reads "the user is updating, not removing." -- Partial retracts (some body facts gone, others still asserted) keep the attachment — the annotation is still meaningful. - -The cascade is graph-aware: named-graph annotations are retracted in the same named graph as the edge they reify, never by mismatched-graph retracts. - -Manage annotation lifecycle through `@annotation` and the cascade above — the [reserved system predicates](../reference/vocabulary.md#edge-annotation-predicates-reserved) that back annotations can't be written by hand on any surface. +Manage annotation lifecycle through `@annotation` and LPG mode — the [reserved system predicates](../reference/vocabulary.md#edge-annotation-predicates-reserved) that back older annotations can't be written by hand on any surface. ## Soft Delete vs Hard Retraction diff --git a/docs/transactions/turtle.md b/docs/transactions/turtle.md index e0e6dbed0b..b68905caf5 100644 --- a/docs/transactions/turtle.md +++ b/docs/transactions/turtle.md @@ -457,7 +457,7 @@ ex:dataset-import-2024-01-22 a ex:DatasetImport ; ## Edge annotations (RDF 1.2 / Turtle-star) -The Turtle parser (which also reads N-Triples) accepts the RDF 1.2 *asserting* forms on every Turtle write path — `insert`, `upsert`, bulk `import`, `fluree graph sync`, and the memory importer. All of them store the same `rdf:reifies` link that the JSON-LD `@annotation` and SPARQL 1.2 `{| |}` surfaces write, so cascade retracts and hydration treat every surface as one, and the annotations are queryable from every query surface: +The Turtle parser (which also reads N-Triples) accepts the RDF 1.2 reifying forms on every Turtle write path — `insert`, `upsert`, bulk `import`, `fluree graph sync`, and the memory importer. All of them store the same `rdf:reifies` link that the JSON-LD `@annotation` and SPARQL 1.2 `{| |}` surfaces write, so retractions and hydration treat every surface as one, and the annotations are queryable from every query surface: ```turtle @prefix ex: . diff --git a/fluree-db-api/tests/it_edge_annotations.rs b/fluree-db-api/tests/it_edge_annotations.rs index a0b8aecb1f..aca73d382c 100644 --- a/fluree-db-api/tests/it_edge_annotations.rs +++ b/fluree-db-api/tests/it_edge_annotations.rs @@ -851,125 +851,98 @@ async fn annotation_rooted_query_returns_no_rows_when_metadata_doesnt_match() { ); } +/// Retracting a base edge leaves its reifier and link (RDF 1.2): `@reifies` +/// still finds the claim and the annotation syntax, which joins the edge, +/// stops matching it until the edge is asserted again. In LPG mode the +/// retract deletes the relationship: the link goes with the edge, so +/// re-asserting the edge alone brings no claim back. #[tokio::test] -async fn retracting_base_edge_cascades_f_reifies_bundle() { - // M1b cascade: when a base edge is retracted, the `f:reifies*` - // bundle pointing at it must be retracted in the same - // transaction so the durable encoding doesn't keep orphaned - // attachment pointers. - // - // The naïve "post-delete @reifies returns zero rows" check is - // ambiguous: the base-edge triple emitted by the M1b expansion - // *also* drops the row when the edge isn't currently asserted, - // so zero rows after delete tells us nothing about whether the - // f:reifies* bundle was retracted or merely orphaned. - // - // The discriminating test: after the cascade-eligible delete, - // re-insert *just* the base edge (no `@annotation` block). This - // re-asserts the visibility-check edge but does not re-emit any - // f:reifies* facts. So: - // - if cascade fired, the f:reifies* facts are retracted, - // re-inserting the edge doesn't bring them back, and - // `@reifies` returns zero rows. - // - if cascade didn't fire, the f:reifies* facts are still - // asserted from the original insert, the visibility check - // now passes, and `@reifies` returns the original - // annotation — proving the bundle was orphaned, not cleaned. - let fluree = FlureeBuilder::memory().build_memory(); - let ledger_id = "it/edge-annotations:cascade-base-retract"; - let ledger0 = genesis_ledger(&fluree, ledger_id); +async fn retracting_a_base_edge_keeps_its_reifier_outside_lpg_mode() { + for lpg in [false, true] { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/edge-annotations:base-retract"; + let ledger0 = genesis_ledger(&fluree, ledger_id); - // 1. Insert an annotated edge. - let insert = json!({ - "@context": ctx(), - "@id": "ex:alice", - "ex:worksFor": { - "@id": "ex:acme", - "@annotation": { - "@id": "ex:emp/alice-acme", - "ex:role": "Engineer" + let insert = json!({ + "@context": ctx(), + "@id": "ex:alice", + "ex:worksFor": { + "@id": "ex:acme", + "@annotation": { "@id": "ex:emp/alice-acme", "ex:role": "Engineer" } } - } - }); - let after_insert = fluree - .insert(ledger0, &insert) - .await - .expect("annotated insert"); - - let q = json!({ - "@context": ctx(), - "select": ["?person", "?org"], - "where": { - "ex:role": "Engineer", - "@reifies": { "@id": "?person", "ex:worksFor": { "@id": "?org" } } - } - }); - - // Sanity: the annotation is reachable via @reifies before delete. - let pre = support::query_jsonld_formatted(&fluree, &after_insert.ledger, &q) - .await - .expect("pre-cascade query"); - assert_eq!( - pre.as_array().expect("array").len(), - 1, - "@reifies should find the annotation before cascade: {pre:#?}" - ); - - // 2. Retract the base edge via SPARQL-style update. The - // transactor's cascade pass should retract the corresponding - // `f:reifies*` bundle automatically. - let delete = json!({ - "@context": ctx(), - "where": { "@id": "?s", "ex:worksFor": { "@id": "?o" } }, - "delete": { "@id": "?s", "ex:worksFor": { "@id": "?o" } } - }); - let after_delete = fluree - .update(after_insert.ledger, &delete) - .await - .expect("base-edge delete"); + }); + let after_insert = fluree + .insert(ledger0, &insert) + .await + .expect("annotated insert"); - // 3. Re-insert *only* the base edge. No `@annotation` block, - // so no f:reifies* assertions are emitted by the lowering. - let reinsert = json!({ - "@context": ctx(), - "@id": "ex:alice", - "ex:worksFor": { "@id": "ex:acme" } - }); - let after_reinsert = fluree - .insert(after_delete.ledger, &reinsert) - .await - .expect("plain re-insert"); + let reified = json!({ + "@context": ctx(), + "select": ["?person", "?org"], + "where": { + "ex:role": "Engineer", + "@reifies": { "@id": "?person", "ex:worksFor": { "@id": "?org" } } + } + }); + let annotated = json!({ + "@context": ctx(), + "select": ["?role"], + "where": { + "@id": "ex:alice", + "ex:worksFor": { "@id": "ex:acme", "@annotation": { "ex:role": "?role" } } + } + }); + let rows = |ledger: &MemoryLedger, q: &JsonValue| { + let (fluree, ledger, q) = (&fluree, ledger.clone(), q.clone()); + async move { + support::query_jsonld_formatted(fluree, &ledger, &q) + .await + .expect("query") + .as_array() + .expect("array") + .len() + } + }; + assert_eq!(rows(&after_insert.ledger, &reified).await, 1); - // 4. The base edge is now currently asserted again (visibility - // check passes), so any zero-row result must come from the - // f:reifies* facts being retracted — the cascade contract. - let post = support::query_jsonld_formatted(&fluree, &after_reinsert.ledger, &q) - .await - .expect("post-cascade-and-reinsert query"); - let arr = post.as_array().expect("array"); - assert!( - arr.is_empty(), - "after cascade + plain re-insert, @reifies must return zero rows \ - (proving the f:reifies* bundle was retracted, not just orphaned). got: {arr:#?}" - ); + let mut delete = json!({ + "@context": ctx(), + "where": { "@id": "?s", "ex:worksFor": { "@id": "?o" } }, + "delete": { "@id": "?s", "ex:worksFor": { "@id": "?o" } } + }); + if lpg { + delete["opts"] = json!({ "lpgEdgeLifecycle": true }); + } + let after_delete = fluree + .update(after_insert.ledger, &delete) + .await + .expect("base-edge delete"); + assert_eq!( + rows(&after_delete.ledger, &reified).await, + usize::from(!lpg), + "[lpg={lpg}] the link outlives the edge only outside LPG mode" + ); + assert_eq!( + rows(&after_delete.ledger, &annotated).await, + 0, + "[lpg={lpg}] the annotation syntax needs the edge" + ); - // Cross-check: a bare-triple query for the re-inserted edge - // must return one row, confirming the visibility-check side of - // the proof — the edge IS currently asserted, so zero rows - // above isn't a visibility miss. - let bare = json!({ - "@context": ctx(), - "select": ["?person", "?org"], - "where": { "@id": "?person", "ex:worksFor": { "@id": "?org" } } - }); - let bare_rows = support::query_jsonld_formatted(&fluree, &after_reinsert.ledger, &bare) - .await - .expect("bare triple query after re-insert"); - assert_eq!( - bare_rows.as_array().expect("array").len(), - 1, - "the re-inserted base edge must be currently asserted (cross-check)" - ); + let reinsert = json!({ + "@context": ctx(), + "@id": "ex:alice", + "ex:worksFor": { "@id": "ex:acme" } + }); + let after_reinsert = fluree + .insert(after_delete.ledger, &reinsert) + .await + .expect("plain re-insert"); + assert_eq!( + rows(&after_reinsert.ledger, &annotated).await, + usize::from(!lpg), + "[lpg={lpg}] re-asserting the edge brings back only a claim whose link survived" + ); + } } #[tokio::test] @@ -1151,12 +1124,13 @@ async fn first_annotation_through_incremental_index_flips_has_annotations() { fast-path would skip post-reindex retracts)" ); - // Step 4: retract the base edge. The cascade gate sees - // `snapshot.has_annotations = true` so the scan runs. + // Step 4: retract the base edge in LPG mode. The cascade gate + // sees `snapshot.has_annotations = true` so the scan runs. let delete = json!({ "@context": ctx(), "where": { "@id": "?s", "ex:worksFor": { "@id": "?o" } }, - "delete": { "@id": "?s", "ex:worksFor": { "@id": "?o" } } + "delete": { "@id": "?s", "ex:worksFor": { "@id": "?o" } }, + "opts": { "lpgEdgeLifecycle": true } }); let after_delete = fluree .update(post_reindex, &delete) @@ -1250,7 +1224,8 @@ async fn cascade_fires_for_indexed_annotation_when_edge_is_retracted() { let delete = json!({ "@context": ctx(), "where": { "@id": "?s", "ex:worksFor": { "@id": "?o" } }, - "delete": { "@id": "?s", "ex:worksFor": { "@id": "?o" } } + "delete": { "@id": "?s", "ex:worksFor": { "@id": "?o" } }, + "opts": { "lpgEdgeLifecycle": true } }); let after_delete = fluree.update(reloaded, &delete).await.expect("delete"); @@ -1441,86 +1416,57 @@ async fn subject_expansion_emits_no_annotation_when_edge_has_none() { ); } +/// An anonymous annotation's body is ordinary RDF about its blank-node +/// reifier: it outlives the edge (RDF 1.2), unless the retract runs in LPG +/// mode, where deleting the edge deletes the relationship and its properties. #[tokio::test] -async fn cascade_cleans_up_anonymous_annotation_metadata() { - // RDF-mode cleanup contract: when the cascade retracts the - // `f:reifies*` bundle for an anonymous (blank-node) annotation, - // it must also retract the annotation's body metadata. Without - // this, the body flakes (`_:fluree_ann_0 ex:role "Engineer"`) - // remain in the graph as orphaned RDF — unreachable through - // `@reifies` (the bundle is gone) but still discoverable via - // a `?s ex:role "Engineer"` scan. - // - // Explicit-IRI annotations are deliberately NOT cleaned up in - // RDF mode — they're user-addressable subjects that may have - // independent meaning. The opt-in `lpgEdgeLifecycle` flag - // would extend cleanup to those; not in scope here. - let fluree = FlureeBuilder::memory().build_memory(); - let ledger_id = "it/edge-annotations:cascade-anonymous-metadata"; - let ledger0 = genesis_ledger(&fluree, ledger_id); +async fn anonymous_annotation_body_outlives_its_edge_outside_lpg_mode() { + for lpg in [false, true] { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/edge-annotations:anonymous-body"; + let ledger0 = genesis_ledger(&fluree, ledger_id); + let txn = json!({ + "@context": ctx(), + "@id": "ex:alice", + "ex:worksFor": { "@id": "ex:acme", "@annotation": { "ex:role": "Engineer" } } + }); + let after_insert = fluree.insert(ledger0, &txn).await.expect("insert"); - // Insert with an *anonymous* annotation (no @id on the - // annotation block — the lowering mints a blank-node SID). - let txn = json!({ - "@context": ctx(), - "@id": "ex:alice", - "ex:worksFor": { - "@id": "ex:acme", - "@annotation": { "ex:role": "Engineer" } + let mut delete = json!({ + "@context": ctx(), + "where": { "@id": "?s", "ex:worksFor": { "@id": "?o" } }, + "delete": { "@id": "?s", "ex:worksFor": { "@id": "?o" } } + }); + if lpg { + delete["opts"] = json!({ "lpgEdgeLifecycle": true }); } - }); - let after_insert = fluree.insert(ledger0, &txn).await.expect("insert"); - - // Sanity: the role is queryable before the cascade. - let q_role = json!({ - "@context": ctx(), - "select": ["?role"], - "where": { "ex:role": "?role" } - }); - let pre = support::query_jsonld_formatted(&fluree, &after_insert.ledger, &q_role) - .await - .expect("pre-cascade role query"); - assert_eq!( - pre.as_array().expect("array").len(), - 1, - "ex:role should be present before cascade: {pre:#?}" - ); - - // Retract the base edge — cascade fires. - let delete = json!({ - "@context": ctx(), - "where": { "@id": "?s", "ex:worksFor": { "@id": "?o" } }, - "delete": { "@id": "?s", "ex:worksFor": { "@id": "?o" } } - }); - let after_delete = fluree - .update(after_insert.ledger, &delete) - .await - .expect("delete"); + let after_delete = fluree + .update(after_insert.ledger, &delete) + .await + .expect("delete"); - // After the cascade, no row should match `?s ex:role ?role` — - // the anonymous annotation's body metadata is gone too. - let post = support::query_jsonld_formatted(&fluree, &after_delete.ledger, &q_role) - .await - .expect("post-cascade role query"); - let arr = post.as_array().expect("array"); - assert!( - arr.is_empty(), - "anonymous annotation's metadata must be cleaned up by RDF-mode cascade; \ - got: {arr:#?}" - ); + let q_role = json!({ + "@context": ctx(), + "select": ["?role"], + "where": { "ex:role": "?role" } + }); + let post = support::query_jsonld_formatted(&fluree, &after_delete.ledger, &q_role) + .await + .expect("role query"); + assert_eq!( + post.as_array().expect("array").len(), + usize::from(!lpg), + "[lpg={lpg}] the body outlives the edge only outside LPG mode: {post:#?}" + ); + } } #[tokio::test] -async fn retracting_all_annotation_metadata_cleans_bundle_too() { - // When a user retracts every asserted user-property flake of - // an annotation subject in a single transaction, the cascade - // should also retract the `f:reifies*` bundle pointing at the - // (still-asserted) base edge. Without this auto-cleanup, the - // bundle stays asserted as an orphan: an inline `@annotation` - // query would still surface the annotation subject (because - // `f:reifiesSubject/Predicate/Object` still pin it to the - // base edge), even though the user clearly intended to delete - // the whole annotation. +async fn retracting_an_annotation_body_keeps_its_link() { + // Retracting every property of a reifier retracts just those + // triples: the reifier still reifies the edge (its link is a triple + // of its own), so `@annotation` still finds it. The JSON-LD + // `@annotation` delete retracts the link. let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "it/edge-annotations:metadata-retract-cleans-bundle"; let ledger0 = genesis_ledger(&fluree, ledger_id); @@ -1571,15 +1517,13 @@ async fn retracting_all_annotation_metadata_cleans_bundle_too() { .await .expect("metadata-only retract"); - // The bundle should also be gone — inline `@annotation` no - // longer finds the orphaned annotation. let post = support::query_jsonld_formatted(&fluree, &after_delete.ledger, &q_ann) .await - .expect("post-cleanup ?ann query"); - let arr = post.as_array().expect("array"); - assert!( - arr.is_empty(), - "after metadata retract, the bundle must be cleaned too; got: {arr:#?}" + .expect("post-retract ?ann query"); + assert_eq!( + post.as_array().expect("array").len(), + 1, + "the link outlives the body: {post:#?}" ); // The base edge itself should still be queryable — the cleanup @@ -2351,8 +2295,8 @@ async fn wildcard_subject_hydration_keeps_explicit_iri_annotations_visible() { #[tokio::test] async fn cascade_retracts_named_graph_annotations_in_their_own_graph() { - // Regression: the cascade's link retract must carry the named graph of - // the original assertion. A default-graph retract would not match it in + // Regression: the LPG cascade's link retract must carry the named graph + // of the original assertion. A default-graph retract would not match it in // Fluree's flake identity model, leaving the link live in the named // graph after its edge is gone. let fluree = FlureeBuilder::memory().build_memory(); @@ -2387,7 +2331,8 @@ async fn cascade_retracts_named_graph_annotations_in_their_own_graph() { "@id": "ex:alice", "@graph": "ex:hr-graph", "ex:worksFor": { "@id": "ex:acme" } - } + }, + "opts": { "lpgEdgeLifecycle": true } }); let after_delete = fluree .update(after_insert.ledger, &delete) @@ -6026,17 +5971,16 @@ async fn one_reifier_on_the_same_edge_in_two_graphs_writes_in_one_transaction() // =========================================================================== // Annotation-form DELETE: the full spelling matrix (issue #1861) // -// Two separate rules combine into a result that surprises people, and the -// docs used to describe each one only in isolation: +// Two rules combine into a result that surprises people: // // 1. The annotation form ASSERTS the base triple. RDF 1.2 Turtle §2.11.1 // defines `s p o ~ :r {| … |}` as "both reify and assert" the triple, and // SPARQL 1.2 Update §3.1.2 routes `DELETE DATA`'s QuadData through the // same production. So the annotation form of a delete is a base-edge -// retraction. This is spec-mandated, not a Fluree choice. -// 2. Retracting a base edge cascades to EVERY reifier attached to it, not -// just the one named. This one IS a Fluree choice — neither spec entails -// it — and it is why sibling claims lose their attachment. +// retraction. +// 2. Retracting a base edge leaves every reifier's link (RDF 1.2), but the +// annotation syntax joins the edge, so sibling claims stop matching it +// while the reified-triple form still finds them. // // The rows below are measured, not assumed. `docs/concepts/edge-annotations.md` // quotes this table; if a row changes here, that section is wrong. @@ -6057,8 +6001,11 @@ struct Survivors { claim1_body: usize, /// Body properties still on `:claim2`. claim2_body: usize, - /// Reifiers still attached to `:alice :knows :bob`. + /// Reifiers the annotation syntax still matches on `:alice :knows :bob` + /// (it needs the edge and a `:confidence`). attached: usize, + /// Reifiers still linked to `:alice :knows :bob`, asserted or not. + linked: usize, /// Rows for `:alice :knows :carol` — 1 means the object was rewritten. new_object: usize, } @@ -6095,6 +6042,11 @@ async fn annotation_matrix_survivors(fluree: &MemoryFluree, ledger_id: &str) -> SELECT ?c WHERE { :alice :knows :bob ~ ?c {| :confidence ?f |} }", ) .await, + linked: count( + "PREFIX : \ + SELECT ?c WHERE { << :alice :knows :bob ~ ?c >> }", + ) + .await, new_object: count( "PREFIX : \ SELECT ?o WHERE { :alice :knows ?o . FILTER(?o = :carol) }", @@ -6118,10 +6070,10 @@ enum MatrixOp { /// them change what the documentation has to say: /// /// - **Row 10** (`~ :claim1` with no block) reads as "detach claim1" and is -/// the worst outcome in the table: the edge goes, BOTH claims are detached, -/// and BOTH bodies are left standing as well-formed-looking orphans. -/// - **Row 7** is an `upsert`, with no delete written anywhere, and it fires -/// the identical cascade. +/// the worst outcome in the table: the edge goes too, so claim2 stops +/// matching the annotation syntax. +/// - **Row 7** is an `upsert`, with no delete written anywhere, and it +/// retracts the edge claim2 reifies all the same. #[tokio::test] async fn annotation_form_delete_matrix() { let rows: Vec<(&str, MatrixOp, Option)> = vec![ @@ -6138,6 +6090,7 @@ async fn annotation_form_delete_matrix() { claim1_body: 0, claim2_body: 2, attached: 0, + linked: 1, new_object: 0, }), ), @@ -6155,6 +6108,7 @@ async fn annotation_form_delete_matrix() { claim1_body: 1, claim2_body: 1, attached: 0, + linked: 0, new_object: 0, }), ), @@ -6168,11 +6122,12 @@ async fn annotation_form_delete_matrix() { claim1_body: 2, claim2_body: 2, attached: 1, + linked: 1, new_object: 0, }), ), - // Row 4 — property-level retraction: what a reader usually means by - // "withdraw claim1". Pass 2 then retires the now-empty reifier. + // Row 4 — property-level retraction: the body goes, and claim1 stays + // linked to the edge, a reifier with nothing left to say. ( "property-level retraction of the claim body", MatrixOp::Sparql( @@ -6184,10 +6139,12 @@ async fn annotation_form_delete_matrix() { claim1_body: 0, claim2_body: 2, attached: 1, + linked: 2, new_object: 0, }), ), - // Row 6 — the baseline: deleting the bare edge detaches both claims. + // Row 6 — the baseline: deleting the bare edge leaves both claims + // linked to a triple no longer asserted. ( "bare base edge", MatrixOp::Sparql("PREFIX : DELETE DATA { :alice :knows :bob }"), @@ -6196,11 +6153,12 @@ async fn annotation_form_delete_matrix() { claim1_body: 2, claim2_body: 2, attached: 0, + linked: 2, new_object: 0, }), ), - // Row 7 — an upsert that changes the object. No delete is written and - // the same cascade fires. + // Row 7 — an upsert that changes the object. The upsert re-points + // claim1; claim2 stays linked to the retracted edge. ( "upsert changing the object", MatrixOp::UpsertTurtle( @@ -6212,6 +6170,7 @@ async fn annotation_form_delete_matrix() { claim1_body: 2, claim2_body: 2, attached: 0, + linked: 1, new_object: 1, }), ), @@ -6226,7 +6185,7 @@ async fn annotation_form_delete_matrix() { None, ), // Row 10 — the sharpest footgun, and absent from the issue. Reads as - // "detach claim1"; removes the edge and orphans both bodies. + // "detach claim1"; removes the edge as well. ( "bare reifier, no body block", MatrixOp::Sparql( @@ -6237,6 +6196,7 @@ async fn annotation_form_delete_matrix() { claim1_body: 2, claim2_body: 2, attached: 0, + linked: 1, new_object: 0, }), ), @@ -6327,16 +6287,12 @@ async fn annotation_form_delete_retracts_the_base_edge_in_both_spellings() { } } -/// The half that is Fluree's own semantics: the cascade reaches reifiers the -/// delete never named. -/// -/// Pinned separately and explicitly because nothing in RDF 1.2 or SPARQL 1.2 -/// entails it — SPARQL 1.2 Update §3.1.2 Example 6 makes the converse point, -/// that deleting a reifying triple leaves the asserted triple alone. A reader -/// who checks the spec and finds Fluree deleting more will otherwise conclude -/// there is a second bug. +/// Deleting the base edge reaches reifiers the delete never named only in +/// what the annotation syntax matches: their links and bodies stay (RDF 1.2; +/// SPARQL 1.2 Update §3.1.2 Example 6 makes the converse point), and the +/// annotation syntax, which joins the edge, stops matching them. #[tokio::test] -async fn base_edge_retraction_detaches_sibling_reifiers_the_delete_never_named() { +async fn base_edge_retraction_leaves_sibling_reifiers_the_delete_never_named() { let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "ann-sibling-cascade:main"; fluree @@ -6362,13 +6318,10 @@ async fn base_edge_retraction_detaches_sibling_reifiers_the_delete_never_named() let after = annotation_matrix_survivors(&fluree, ledger_id).await; assert_eq!( after.attached, 0, - ":claim2 is detached though it was never named" - ); - assert_eq!( - after.claim2_body, 2, - ":claim2's body survives its attachment — this is the orphan state the \ - docs must warn about" + "the annotation syntax no longer matches :claim2, though it was never named" ); + assert_eq!(after.linked, 1, ":claim2's link outlives the edge"); + assert_eq!(after.claim2_body, 2, ":claim2's body outlives the edge"); } /// Wildcard reads of a legacy bundle show the link derived from it, never diff --git a/fluree-db-api/tests/it_edge_annotations_indexed.rs b/fluree-db-api/tests/it_edge_annotations_indexed.rs index 0c17961207..52c1dec3e1 100644 --- a/fluree-db-api/tests/it_edge_annotations_indexed.rs +++ b/fluree-db-api/tests/it_edge_annotations_indexed.rs @@ -623,11 +623,11 @@ async fn live_reifies_flakes(fluree: &fluree_db_api::Fluree, ledger_id: &str) -> live } -/// Deleting a base edge must retract the claim that reifies it, even once the -/// link has been indexed in a named graph — the cascade's link lookup and the -/// retract it writes are both scoped to the edge's graph. +/// Deleting a base edge in LPG mode must retract the claim that reifies it, +/// even once the link has been indexed in a named graph — the cascade's link +/// lookup and the retract it writes are both scoped to the edge's graph. #[tokio::test] -async fn deleting_an_indexed_named_graph_edge_cascades_to_its_claim() { +async fn deleting_an_indexed_named_graph_edge_in_lpg_mode_cascades_to_its_claim() { let fluree = FlureeBuilder::memory().build_memory(); let ledger_id = "it/edge-annotations-indexed:cascade-named-graph"; @@ -661,78 +661,23 @@ async fn deleting_an_indexed_named_graph_edge_cascades_to_its_claim() { let deleted = fluree .graph(ledger_id) .transact() - .sparql_update( - "PREFIX ex: \n\ - DELETE DATA { GRAPH \ - { ex:alice ex:knows ex:bob } }", - ) - .commit() - .await - .expect("delete the base edge"); - assert!(deleted.receipt.t > committed.ledger.t()); - - assert_eq!( - live_reifies_flakes(&fluree, ledger_id).await, - 0, - "the claim's link must not outlive the edge it reifies" - ); -} - -/// The *other* cascade pass against an indexed named-graph link. -/// -/// Pass 1 fires when the base edge is deleted. Pass 2 fires when the user -/// deletes an annotation's last piece of metadata without touching the edge, -/// which would leave the link behind with nothing to describe. -#[tokio::test] -async fn deleting_an_indexed_named_graph_claim_body_cascades_its_link() { - let fluree = FlureeBuilder::memory().build_memory(); - let ledger_id = "it/edge-annotations-indexed:cascade-named-graph-orphan"; - - let committed = fluree - .insert( - genesis_ledger(&fluree, ledger_id), - &json!({ - "@context": ctx(), + .update(&json!({ + "@context": ctx(), + "delete": { "@id": "ex:alice", "@graph": "ex:claims-graph", - "ex:knows": { - "@id": "ex:bob", - // A string, deliberately. A bare `0.9` in the SPARQL below - // is an `xsd:decimal` while JSON-LD stores it as an - // `xsd:double`, so `DELETE DATA` would match nothing, the - // transaction would still commit, and this test would pass - // without the cascade ever running. - "@annotation": {"@id": "ex:claim1", "ex:role": "Engineer"} - } - }), - ) - .await - .expect("annotated named-graph insert"); - - support::rebuild_and_publish_index(&fluree, ledger_id).await; - - assert!( - live_reifies_flakes(&fluree, ledger_id).await > 0, - "the indexed link must be visible to this read before the delete" - ); - - // Delete the claim's only metadata fact, leaving the edge itself alone. - let deleted = fluree - .graph(ledger_id) - .transact() - .sparql_update( - "PREFIX ex: \n\ - DELETE DATA { GRAPH \ - { ex:claim1 ex:role \"Engineer\" } }", - ) + "ex:knows": {"@id": "ex:bob"} + }, + "opts": {"lpgEdgeLifecycle": true} + })) .commit() .await - .expect("delete the claim body"); + .expect("delete the base edge"); assert!(deleted.receipt.t > committed.ledger.t()); assert_eq!( live_reifies_flakes(&fluree, ledger_id).await, 0, - "a link whose claim has no body left must not survive as an orphan" + "the claim's link must not outlive the edge it reifies" ); } diff --git a/fluree-db-api/tests/it_import_turtle_star.rs b/fluree-db-api/tests/it_import_turtle_star.rs index 2139923e38..aab42c811d 100644 --- a/fluree-db-api/tests/it_import_turtle_star.rs +++ b/fluree-db-api/tests/it_import_turtle_star.rs @@ -245,9 +245,9 @@ async fn links_in( } /// TriG import writes a GRAPH block's link into that graph, naming the -/// block's edge, and deleting that edge must cascade to it. +/// block's edge; deleting the edge leaves the link, as RDF 1.2 does. #[tokio::test] -async fn imported_trig_star_link_lands_in_its_graph_and_cascades() { +async fn imported_trig_star_link_lands_in_its_graph_and_outlives_its_edge() { let alias = "it/import-trig-star:graph-anchored"; let trig = format!( "@prefix ex: .\n\ @@ -277,11 +277,9 @@ async fn imported_trig_star_link_lands_in_its_graph_and_cascades() { .await .expect("delete the imported base edge"); + // RDF 1.2: deleting a triple leaves its reifier's link. let remaining = links_in(&fluree, alias, CLAIMS_GRAPH).await; - assert!( - remaining.is_empty(), - "the claim's link must not outlive the edge it reifies: {remaining:#?}" - ); + assert_eq!(remaining.len(), 1, "{remaining:#?}"); } /// A multi-chunk import: a fixture large enough to be cut up, with star diff --git a/fluree-db-api/tests/it_tracing_spans.rs b/fluree-db-api/tests/it_tracing_spans.rs index a4b6c8de12..fb1411027e 100644 --- a/fluree-db-api/tests/it_tracing_spans.rs +++ b/fluree-db-api/tests/it_tracing_spans.rs @@ -909,8 +909,8 @@ async fn annotation_hydration_emits_inject_annotations_span() { #[tokio::test(flavor = "current_thread")] async fn annotation_cascade_emits_cascade_reifies_bundle_span() { - // Retracting a base edge that has annotations should emit a - // `cascade_reifies_bundle` span tagged with the cascade row + // Retracting a base edge that has annotations in LPG mode should + // emit a `cascade_reifies_bundle` span tagged with the cascade row // count. On non-annotation ledgers the gate skips it. let fluree = FlureeBuilder::memory().build_memory(); let ledger0 = support::genesis_ledger(&fluree, "tracing-cascade:main"); @@ -943,7 +943,8 @@ async fn annotation_cascade_emits_cascade_reifies_bundle_span() { "delete": { "@id": "ex:alice", "ex:worksFor": { "@id": "ex:acme" } - } + }, + "opts": { "lpgEdgeLifecycle": true } }), ) .await diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index b5134f87fb..8dafada8b4 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -1069,11 +1069,12 @@ async fn change_claims_without_indexing( ledger, &json!({ "@context": { "ex": "http://example.org/" }, - "delete": { "@id": "ex:alice", "ex:knows": { "@id": "ex:carol" } } + "delete": { "@id": "ex:alice", "ex:knows": { "@id": "ex:carol" } }, + "opts": { "lpgEdgeLifecycle": true } }), ) .await - .expect("retract an annotated edge") + .expect("retract an annotated edge and its relationship") .ledger } diff --git a/fluree-db-transact/src/stage.rs b/fluree-db-transact/src/stage.rs index ca24e70672..bd1aee45af 100644 --- a/fluree-db-transact/src/stage.rs +++ b/fluree-db-transact/src/stage.rs @@ -51,16 +51,15 @@ use fluree_db_shacl::{ShaclCache, ShaclEngine, ValidationReport}; /// Given `graph_sids` (ledger GraphId → Sid), returns the /// inverse mapping. Used by SHACL/policy to determine which graph a flake /// belongs to based on its `Flake.g` field. -/// Retract the annotations a transaction's retracts leave dangling, so a link -/// never names a triple that is no longer asserted (annotation-syntax reads -/// rely on it and skip the base edge). +/// In LPG mode (`opts.lpgEdgeLifecycle`, which Cypher `DELETE` sets), deleting +/// a relationship's edge deletes the relationship: /// /// 1. A retracted triple retracts every link naming it. -/// 2. Retracting all of a reifier's body retracts its links. -/// 3. A reifier left with no link loses its body when it is a blank node, or -/// in LPG mode (`opts.lpgEdgeLifecycle`), where deleting a relationship -/// deletes its properties. An IRI reifier's body otherwise stays, as -/// ordinary RDF about a named resource. +/// 2. A reifier left with no link loses its body. +/// +/// RDF 1.2 entails neither, so outside LPG mode a transaction retracts only +/// what it names: a reifier and its link outlive the triple they reify, and +/// the annotation syntax, which joins the triple, stops matching them. /// /// Links derived from legacy `f:reifies*` bundles read like any other link; /// retracting one leaves the bundle, whose derived assert the retract cancels @@ -78,7 +77,9 @@ async fn cascade_attachment_retracts( use std::collections::BTreeMap; let mut cascade = Vec::new(); - if !ledger.snapshot.has_annotations && !ledger.novelty.has_annotations() { + if !lpg_edge_lifecycle + || (!ledger.snapshot.has_annotations && !ledger.novelty.has_annotations()) + { return Ok(cascade); } @@ -166,67 +167,12 @@ async fn cascade_attachment_retracts( } } - // 2. Reifiers whose whole body this transaction retracts. Same-txn - // asserts count toward what survives, so replacing a body value - // keeps the reifier. - type FlakeIdentity = (Sid, FlakeValue, Sid, Option); - let identity = - |f: &Flake| -> FlakeIdentity { (f.p.clone(), f.o.clone(), f.dt.clone(), f.m.clone()) }; - let mut body_retracts: BTreeMap<(GraphId, Sid), (Option, HashSet)> = - BTreeMap::new(); - let mut body_asserts: HashSet<(GraphId, Sid)> = HashSet::new(); - for f in flakes.iter().filter(|f| !is_annotation_predicate(&f.p)) { - let key = (resolve_flake_graph_id(f, reverse_graph)?, f.s.clone()); - if f.op { - body_asserts.insert(key); - } else { - body_retracts - .entry(key) - .or_insert_with(|| (f.g.clone(), HashSet::new())) - .1 - .insert(identity(f)); - } - } - for ((g_id, ann), (g_sid, retracted)) in body_retracts { - if body_asserts.contains(&(g_id, ann.clone())) { - continue; - } - let current = scan( - g_id, - IndexType::Spot, - RangeMatch::new().with_subject(ann.clone()), - ) - .await?; - let (links, body): (Vec, Vec) = current - .into_iter() - .filter(|f| !fluree_db_core::is_reserved_reifies_predicate(&f.p)) - .partition(|f| is_rdf_reifies(&f.p)); - if links.is_empty() || body.iter().any(|f| !retracted.contains(&identity(f))) { - continue; - } - for link in links { - let already = unlinked - .get(&(g_id, ann.clone())) - .is_some_and(|(_, terms)| terms.contains(&link.o)); - if already || txn_link_retracts.contains(&(g_id, ann.clone(), link.o.clone())) { - continue; - } - cascade.push(retract(&link, g_sid.as_ref())); - unlinked - .entry((g_id, ann.clone())) - .or_insert_with(|| (g_sid.clone(), Vec::new())) - .1 - .push(link.o); - } - } - - // 3. Bodies of reifiers left with no link. + // 2. Bodies of reifiers left with no link. for (key, g_sid) in explicit { unlinked.entry(key).or_insert((g_sid, Vec::new())); } for ((g_id, ann), (g_sid, terms)) in unlinked { - let anonymous = ann.namespace_code == fluree_vocab::namespaces::BLANK_NODE; - if !(anonymous || lpg_edge_lifecycle) || txn_linked.contains(&(g_id, ann.clone())) { + if txn_linked.contains(&(g_id, ann.clone())) { continue; } let current = scan( @@ -704,22 +650,8 @@ pub async fn stage_with_graph_delta( f }; - // Cascade-retract `f:reifies*` bundles for any base edge that - // is being retracted in this transaction. Without this, a - // DELETE of the base edge would leave the attachment pointers - // orphaned in the durable encoding, and `@reifies` queries - // would still surface annotations for retracted edges. - // - // **M2 scan-based path:** the cascade looks up annotations - // via `range_with_overlay` over the merged snapshot+novelty - // view. This catches annotations whether they're still in - // the novelty overlay or have rolled into indexed base - // storage post-reindex. - // - // M1b minimum: retracts the `f:reifies*` bundle only. The - // anonymous-annotation metadata cascade (RDF default) and the - // explicit-IRI metadata cascade (LPG mode opt-in) are tracked - // as follow-ups in the plan. + // LPG relationship lifecycle: deleting an edge deletes its + // relationships (see `cascade_attachment_retracts`). let lpg_edge_lifecycle = txn.opts.lpg_edge_lifecycle.unwrap_or(false); let cascade = cascade_attachment_retracts( &flakes, From bc25ceb0504f137e7a9939e869fc831d7e6bc9dc Mon Sep 17 00:00:00 2001 From: bplatz Date: Sat, 3 Oct 2026 09:41:00 -0400 Subject: [PATCH 58/92] feat(sparql): UPDATE takes reified triples and named-graph annotations SPARQL UPDATE desugars `<< s p o ~ r >>` in subject or object position, nested ones included, to its reifier plus `r rdf:reifies <<( s p o )>>`, which does not assert `s p o` (RDF 1.2). Annotation tails inside `GRAPH { }` blocks and under `WITH ` are no longer refused: the link and body land in the graph the triple is written to. --- docs/concepts/edge-annotations.md | 9 +- docs/design/edge-annotations.md | 2 +- docs/guides/cookbook-edge-annotations.md | 6 +- docs/guides/cookbook-sparql.md | 3 - docs/query/sparql.md | 5 +- .../tests/it_query_sparql_annotations.rs | 176 +++++++++++------- fluree-db-transact/src/lower_sparql_update.rs | 143 +++++++------- 7 files changed, 195 insertions(+), 149 deletions(-) diff --git a/docs/concepts/edge-annotations.md b/docs/concepts/edge-annotations.md index 036f58aca3..2346e9765e 100644 --- a/docs/concepts/edge-annotations.md +++ b/docs/concepts/edge-annotations.md @@ -26,7 +26,7 @@ If a fact is naturally about a *node* (Alice's birthdate, Acme's industry), put | Surface | How | Notes | |---|---|---| | **JSON-LD insert / upsert / update** | `@annotation` (or alias `@edge`) on a value object | Most ergonomic. Covers literal-valued edges (with explicit `@type` / `@language`), parallel annotations, named reifiers, and **named-graph edges** (an annotation on an edge inside a named graph is written into that same graph, keeping the edge's graph identity). A node's `@reifies` (`{"@id": s, p: o}`, or an array of them) makes the node a reifier of that triple **without asserting it**, as `r rdf:reifies <<( s p o )>>` does. | -| **SPARQL 1.2 UPDATE** | `INSERT DATA { :s :p :o {\| ... \|} }`, `~ `, optional `INSERT { } WHERE { }` templates | Use this when integrating with SPARQL pipelines or when porting from RDF 1.2 / SPARQL-star. **Default graph only:** an annotation tail inside an explicit `GRAPH { }` block, or under a `WITH ` template, is rejected — use the JSON-LD surface or TriG-star for named-graph edge annotations. See [SPARQL 1.2 surface](#sparql-12--rdf-12-surface) below for the per-operation rules. | +| **SPARQL 1.2 UPDATE** | `INSERT DATA { :s :p :o {\| ... \|} }`, `~ `, optional `INSERT { } WHERE { }` templates | Use this when integrating with SPARQL pipelines or when porting from RDF 1.2 / SPARQL-star. Works inside `GRAPH { }` blocks and under `WITH `; `<< :s :p :o ~ :r >>` reifies without asserting. See [SPARQL 1.2 surface](#sparql-12--rdf-12-surface) below for the per-operation rules. | | **Turtle / N-Triples / TriG / N-Quads ingest** (`insert`, `upsert`, bulk `import`, graph sync (`fluree sync`, `/sync`), memory import; TriG via `insert` / `upsert` / `import` / `fluree sync` / `/sync`) | RDF 1.2 forms: the annotation syntax `:s :p :o ~ {\| ... \|}`, `<< :s :p :o ~ :r >>` in subject or object position, and the canonical `:r rdf:reifies <<( :s :p :o )>>` (the only star spelling N-Triples and N-Quads have) — in the default graph and inside TriG `GRAPH { }` blocks alike | Same stored link as `@annotation`. An annotation inside a `GRAPH { }` block, or on an N-Quads statement with a graph label, is written into that graph with the edge's graph identity, exactly as JSON-LD `@graph` + `@annotation` does. As in RDF 1.2, only the annotation syntax asserts the triple: `<< s p o >>` and `rdf:reifies <<( s p o )>>` reify it without asserting it, so N-Triples and N-Quads state an annotated triple as the triple plus its reifier's link. Rejected with a specific error: `<<( ... )>>` anywhere other than the object of `rdf:reifies`, nested triple terms, star constructs inside an annotation body, annotations on collections, and annotations in a TriG `<#txn-meta>` block. Paths that convert to JSON-LD first (`upsert`, `graph sync`, memory import) also reject an annotation on an `rdf:type` edge. See [Turtle ingest](../transactions/turtle.md#edge-annotations-rdf-12--turtle-star). | Mint annotations through `@annotation` / `@edge` (JSON-LD) or the RDF 1.2 forms (`~`, `{| |}`, `<< >>`, `rdf:reifies <<( )>>`) in SPARQL UPDATE and Turtle. The [`f:reifies*` predicates](../reference/vocabulary.md#edge-annotation-predicates-reserved) earlier releases stored annotations under are reserved, and the write surfaces reject them; bulk import reads them, as an export written before links carries them, and stores each annotation's link instead. @@ -527,8 +527,6 @@ These produce a clear error with a span pointing at the offending construct: - **Triple terms outside `rdf:reifies`.** `ex:doc ex:mentions <<( :s :p :o )>>` is rejected. Use a separate annotation subject with `rdf:reifies` if you need to refer to a triple. - **Nested triple terms.** `<<( :s :p <<( :a :b :c )>> )>>` is rejected. - **Annotation on a property-path triple.** `?s ex:p1/ex:p2 ?o {| ... |}` is rejected — the grammar only attaches annotations to simple-predicate triples. -- **Annotation tail inside an explicit `GRAPH { }` block.** `INSERT DATA { GRAPH { :s :p :o {| ... |} } }` is rejected. SPARQL UPDATE annotations are **default-graph only** in v1 — the SPARQL surface doesn't carry the enclosing graph's identity into the stored annotation. Use the JSON-LD `@annotation` surface or a TriG `GRAPH { }` block to annotate an edge inside a named graph. -- **Annotation tail under a `WITH ` template.** `WITH INSERT { :s :p :o {| ... |} } WHERE { ... }` is rejected for the same reason: the annotation would land in `` without recording that graph as the edge's identity, yielding a default-graph edge identity in a named graph. Again, use the JSON-LD surface for named-graph edge annotations. - **Property paths and nested triple terms in a `CONSTRUCT` template's annotation.** A template annotation block (`{| ... |}`) takes simple predicates only, and a template triple term (`?r rdf:reifies <<( ... )>>`) cannot nest. Annotations on literal-valued objects (plain, typed, and language-tagged) are supported on **both** the JSON-LD and SPARQL UPDATE write surfaces — the SPARQL path records the language tag for language-tagged objects so the stored annotation matches the base edge. @@ -547,7 +545,7 @@ SELECT ?age ?t ?op WHERE { This binds `?t` to the transaction time and `?op` to the assert/retract flag of the matched flake. It is **not** edge annotations and is unrelated to the RDF 1.2 reifier surface above. -The legacy reading is selected only by the predicate: a reifier-less `<< s p o >>` whose predicate is `f:t` or `f:op`. Any other predicate, or a `~ reifier`, gives the bare form its RDF 1.2 *reified triple* reading — `<< :s :p :o ~ ?r >> .` and `<< :s :p :o >> ?q ?z` denote the reifier node, exactly as `?r rdf:reifies <<( :s :p :o )>>` does. This applies to query `WHERE` patterns and Turtle ingest; SPARQL UPDATE templates still take only the annotation-tail spelling (`:s :p :o ~ :r {| ... |}`). +The legacy reading is selected only by the predicate: a reifier-less `<< s p o >>` whose predicate is `f:t` or `f:op`. Any other predicate, or a `~ reifier`, gives the bare form its RDF 1.2 *reified triple* reading — `<< :s :p :o ~ ?r >> .` and `<< :s :p :o >> ?q ?z` denote the reifier node, exactly as `?r rdf:reifies <<( :s :p :o )>>` does. This applies to query `WHERE` patterns, Turtle ingest and SPARQL UPDATE alike. The bare-quoted-triple form combined with an annotation tail (`<< :s :p :o >> :pred :obj {| ... |}`) is rejected at parse time — the two surfaces don't compose. @@ -557,7 +555,6 @@ Today's surface covers the common LPG / RDF-star use cases. The following are no - **Annotations on list-occurrence triples.** `@list` membership is in scope as a future extension; the on-disk format already reserves space for it. Today, annotating a list element is rejected at parse time. - **Triple terms as stored values.** `ex:doc ex:mentions <<( ex:s ex:p ex:o )>>` is not a representable value on any write surface (JSON-LD, SPARQL, Turtle): the only accepted position for `<<( ... )>>` is the object of `rdf:reifies`, where it names an edge rather than storing a triple. Use a separate annotation subject. A query can still build and return one: `TRIPLE`, `SUBJECT`, `PREDICATE`, `OBJECT`, `isTRIPLE` and `BIND(<<( ?s ?p ?o )>> AS ?t)` work on terms, and a bound term renders in every result format (see [Output formats](../query/output-formats.md#triple-terms)). JSON-LD queries name them `triple`, `subject`, `predicate`, `object` and `isTriple` (see [JSON-LD query](../query/jsonld-query.md#triple-term-functions)). -- **SPARQL UPDATE annotations inside named graphs.** Annotation tails under `GRAPH { }` / `WITH ` in SPARQL UPDATE are rejected; write named-graph annotations with JSON-LD `@annotation` or TriG-star. The mandated SPARQL 1.2 `VERSION "1.2"` prologue declaration is **accepted** (lex-and-skipped): the RDF 1.2 surface runs ungated, so a conformant 1.2 client that emits the declaration parses normally. @@ -586,6 +583,6 @@ Edge annotations are the storage primitive for the labeled-property-graph shape: - [Cypher](../query/cypher.md) — the property-graph front-end; Cypher relationships map onto edge annotations. - [Edge annotations design](../design/edge-annotations.md) — storage internals (the link, the term dictionary, cascade, ledgers written before links). -- [Datasets and named graphs](datasets-and-named-graphs.md) — annotations work in named graphs (via the JSON-LD `@annotation` surface) as well as the default graph; the SPARQL UPDATE surface is default-graph only. +- [Datasets and named graphs](datasets-and-named-graphs.md) — annotations work in named graphs as well as the default graph, on every write surface. - [Time travel](time-travel.md) — annotation events live in history like every other fact. - [Policy enforcement](policy-enforcement.md) — annotation properties pass through normal policy checks. diff --git a/docs/design/edge-annotations.md b/docs/design/edge-annotations.md index c0e42b0ac6..7064588050 100644 --- a/docs/design/edge-annotations.md +++ b/docs/design/edge-annotations.md @@ -21,7 +21,7 @@ A triple term carries its object's datatype and language tag, so `"chat"@fr` and Every write surface produces the link directly. As in RDF 1.2, only the annotation syntax (`s p o ~ r`, `s p o {| … |}`) asserts the triple it reifies; a reified triple (`<< s p o >>`, `r rdf:reifies <<( s p o )>>`, JSON-LD `@reifies`) does not. - **Turtle / TriG / N-Quads** (`FlakeSink`, `ImportSink`, TriG import): `reified_triple_link` builds the link flake for each `~ r`, `{| … |}`, `<< s p o ~ r >>` and `r rdf:reifies <<( s p o )>>`; the parser emits the base triple itself for the annotation syntax. The Turtle→JSON-LD conversion behind upsert and graph sync writes a reification whose triple the document does not assert as a `@reifies` node. -- **SPARQL UPDATE**: `expand_annotated_triples` desugars an annotation tail as the spec does, to the base triple plus `r rdf:reifies <<( s p o )>>`. Template lowering turns the triple term into `TemplateTerm::TripleTerm`, whose positions resolve per solution; WHERE lowering turns it into the reifier pattern queries use. +- **SPARQL UPDATE**: `expand_annotated_triples` desugars as the spec does: an annotation tail to the base triple plus `r rdf:reifies <<( s p o )>>`, a reified triple `<< s p o ~ r >>` to its reifier plus the link alone, inside `GRAPH { }` blocks and under `WITH` too. Template lowering turns the triple term into `TemplateTerm::TripleTerm`, whose positions resolve per solution; WHERE lowering turns it into the reifier pattern queries use. - **Cypher**: `CREATE (a)-[r:T]->(b)` writes `a T b` and `r rdf:reifies <<( a T b )>>` with `r` a fresh reifier. - **JSON-LD**: JSON-LD has no triple-term syntax, so `@annotation` / `@edge` lower (before expansion) to `f:reifies*` slot keys on a sibling node, and a node's `@reifies` (`{"@id": s, p: o}`, or an array of them) to slot keys on the node itself. After parsing, `fold_slots_into_links` turns each reifier's slots into its link template, and bulk import's `ImportSink` does the same with the slot triples it receives. The slots are an intermediate form and are never stored. A delete-by-selector matches the existing link with the query-side `@reifies` form. diff --git a/docs/guides/cookbook-edge-annotations.md b/docs/guides/cookbook-edge-annotations.md index 614a5ac3aa..5ac1815505 100644 --- a/docs/guides/cookbook-edge-annotations.md +++ b/docs/guides/cookbook-edge-annotations.md @@ -11,7 +11,7 @@ Throughout, the running example is employment: a `worksFor` edge that needs a `r | You have… | Use | Why | |---|---|---| | JSON-LD writes, or you need named-graph edges, or literal-valued edges | **JSON-LD `@annotation`** | Most complete surface — covers everything below. | -| A SPARQL 1.1/1.2 pipeline, or you're porting RDF-star data | **SPARQL 1.2 annotation tail** (`{\| \|}`, `~`, `rdf:reifies`) | Standards syntax. Default-graph only today. | +| A SPARQL 1.1/1.2 pipeline, or you're porting RDF-star data | **SPARQL 1.2 annotation tail** (`{\| \|}`, `~`, `rdf:reifies`) | Standards syntax, in the default graph and named graphs. | | A Turtle / N-Triples / TriG / N-Quads file with RDF 1.2 annotations | **Ingest it as-is** — `insert`, `upsert`, `import` and `graph sync` all accept `{\| \|}`, `~`, `<< >>` and `rdf:reifies <<( )>>`, and TriG `GRAPH { }` blocks accept them too (TriG via `insert` / `upsert` / `import` / `/sync`) | Same on-disk shape as `@annotation`; the annotation syntax asserts its triple, `<< >>` and `rdf:reifies` do not; re-`upsert` the file to update claim bodies (see [Turtle ingest](../transactions/turtle.md#edge-annotations-rdf-12--turtle-star)). | ## Attach metadata to a relationship @@ -234,7 +234,7 @@ SELECT ?person ?org WHERE { } ``` -The triple term `<<( s p o )>>` is accepted **only** as the object of `rdf:reifies`. The bare, parenthesis-free `<< s p o >>` form is the separate Fluree `f:t`/`f:op` flake-metadata construct — the two don't compose. Per-operation reifier rules (variables are template-only; blank/anonymous reifiers are rejected in `DELETE DATA`) are tabulated in the [concept doc](../concepts/edge-annotations.md#sparql-update-rules-by-operation). +The triple term `<<( s p o )>>` is accepted **only** as the object of `rdf:reifies`. The parenthesis-free `<< s p o ~ :r >>` is a reified triple: it stands for its reifier and does not assert `s p o` (a reifier-less `<< s p o >>` under `f:t` / `f:op` is the separate flake-metadata construct). Per-operation reifier rules (variables are template-only; blank/anonymous reifiers are rejected in `DELETE DATA`) are tabulated in the [concept doc](../concepts/edge-annotations.md#sparql-update-rules-by-operation). ## Annotate an edge inside a named graph @@ -254,7 +254,7 @@ Edge annotations live in the same graph as the edge they reify. On the JSON-LD s } ``` -> **SPARQL UPDATE is default-graph only today.** An annotation tail inside an explicit `GRAPH { }` block or under a `WITH ` template is rejected — use the JSON-LD surface above for named-graph edge annotations. +> **Named graphs.** An annotation tail inside a `GRAPH { }` block or under a `WITH ` template writes the link and body into that graph, beside the triple. ## Keep a Turtle claims file in sync diff --git a/docs/guides/cookbook-sparql.md b/docs/guides/cookbook-sparql.md index 02f42af4ab..7d089e426d 100644 --- a/docs/guides/cookbook-sparql.md +++ b/docs/guides/cookbook-sparql.md @@ -201,9 +201,6 @@ returns — see [Edge annotations](../concepts/edge-annotations.md). ## Gotchas -- **Annotations are default-graph only.** A tail inside an explicit `GRAPH { }` - block or under `WITH ` is rejected — use the JSON-LD `@annotation` surface - to annotate an edge inside a named graph. - **Simple-predicate edges only.** A tail on a property-path edge (`?s ex:p1/ex:p2 ?o {| … |}`) is rejected. - **No annotations in `CONSTRUCT` templates** (output form deferred); a diff --git a/docs/query/sparql.md b/docs/query/sparql.md index 529f393ebc..9dbd00e48e 100644 --- a/docs/query/sparql.md +++ b/docs/query/sparql.md @@ -1090,11 +1090,10 @@ INSERT DATA { } ``` -Annotation tails are supported in `INSERT DATA`, `DELETE DATA`, and `INSERT { } WHERE { }` / `DELETE { } WHERE { }` templates, and in `CONSTRUCT` templates, where they carry reifiers into every result format (see [CONSTRUCT](construct.md#edge-annotations-in-the-template)). Per-operation reifier rules (e.g. variables are template-only; blank/anonymous reifiers are rejected in `DELETE DATA`) are tabulated in the [concept doc](../concepts/edge-annotations.md#sparql-update-rules-by-operation). +Annotation tails are supported in `INSERT DATA`, `DELETE DATA`, and `INSERT { } WHERE { }` / `DELETE { } WHERE { }` templates, inside `GRAPH { }` blocks and under `WITH ` (the link and body land in the triple's graph), and in `CONSTRUCT` templates, where they carry reifiers into every result format (see [CONSTRUCT](construct.md#edge-annotations-in-the-template)). Per-operation reifier rules (e.g. variables are template-only; blank/anonymous reifiers are rejected in `DELETE DATA`) are tabulated in the [concept doc](../concepts/edge-annotations.md#sparql-update-rules-by-operation). ### Boundaries (rejected at parse / lowering time) -- **Default graph only.** An annotation tail inside an explicit `GRAPH { }` block, or under a `WITH ` template, is rejected — SPARQL UPDATE annotations target the default graph. Use the JSON-LD `@annotation` surface to annotate an edge inside a named graph. - **Simple-predicate triples only.** `?s ex:p1/ex:p2 ?o {| ... |}` (property-path) is rejected. - **Triple terms only as `rdf:reifies` objects**; any other use errors at parse time. - **No reserved predicates by hand.** The [system predicates](../reference/vocabulary.md#edge-annotation-predicates-reserved) that back annotations are rejected on every UPDATE clause; mint annotations only through the `~` / `{| |}` surface. @@ -1294,7 +1293,7 @@ Current restrictions / boundaries: - **Graph management operations**: `CREATE`, `CLEAR`, `DROP`, `ADD`, `MOVE` and `COPY` are supported, and `CLEAR`/`DROP` accept `GRAPH `, `DEFAULT`, `NAMED` and `ALL`. `DROP` behaves like `CLEAR`: the graph registry is additive, so a dropped graph stays registered but empty. These operations refuse the reserved `#config` and `#txn-meta` graphs. Remote `LOAD` is not supported; `LOAD SILENT` is accepted as a no-op. - **SERVICE**: Only local-ledger endpoints of the form `fluree:ledger:[:]` are supported; arbitrary remote HTTP `SERVICE` endpoints are not supported. - **Property paths**: Supported in `WHERE` (subject to Fluree capability settings). -- **Edge annotations are default-graph only**: an annotation tail (`{| ... |}`) inside an explicit `GRAPH { }` block or under a `WITH ` template is rejected; a blank or anonymous reifier is rejected in `DELETE DATA`. See [Edge annotations](#edge-annotations-sparql-12--rdf-12) for the full boundary list. +- **Edge annotations**: a blank or anonymous reifier is rejected in `DELETE DATA`. See [Edge annotations](#edge-annotations-sparql-12--rdf-12) for the full boundary list. ### Endpoint Usage diff --git a/fluree-db-api/tests/it_query_sparql_annotations.rs b/fluree-db-api/tests/it_query_sparql_annotations.rs index 18fdf36581..01cc46a47d 100644 --- a/fluree-db-api/tests/it_query_sparql_annotations.rs +++ b/fluree-db-api/tests/it_query_sparql_annotations.rs @@ -546,40 +546,66 @@ async fn sparql_reifies_hidden_in_annotation_block_body_is_rejected() { ); } +/// The rows `select` returns over `ledger_id`'s current state. +async fn select_rows(fluree: &MemoryFluree, ledger_id: &str, select: &str) -> Vec { + let ledger = fluree.ledger(ledger_id).await.expect("load"); + support::query_sparql(fluree, &ledger, select) + .await + .expect("query") + .to_sparql_json(&ledger.snapshot) + .expect("sparql json")["results"]["bindings"] + .as_array() + .expect("bindings") + .clone() +} + +/// Annotation tails write into the graph their triple is written to, inside +/// a `GRAPH` block and under `WITH` alike. #[tokio::test] -async fn sparql_with_scoped_annotation_template_is_rejected() { - // `WITH ` re-homes default-position template triples into after - // annotation expansion, but v1 expansion omits f:reifiesGraph (the edge - // identity is default-graph). Allowing it would mint graph-tagged - // reifications carrying a default-graph edge identity — a broken edge - // that never hydrates or cascades. Reject until graph-aware expansion - // lands. Annotation tails inside explicit GRAPH blocks are rejected - // separately by the quad-pattern expansion. - let ledger0 = { - let fluree = FlureeBuilder::memory().build_memory(); - genesis_ledger(&fluree, "it/sparql-ann-update/with-scoped-rej") - }; - let update = r#" - PREFIX ex: - WITH - INSERT { ?person ex:worksFor ex:acme {| ex:role "Engineer" |} } - WHERE { ?person a ex:Person } - "#; - let parsed = fluree_db_sparql::parse_sparql(update); - assert!( - !parsed.has_errors(), - "parse should succeed: {:?}", - parsed.diagnostics - ); - let ast = parsed.ast.unwrap(); - let mut ns = NamespaceRegistry::from_db(&ledger0.snapshot); - let err = fluree_db_transact::lower_sparql_update_ast(&ast, &mut ns, TxnOpts::default()) - .expect_err("annotation tail on a WITH-scoped template must be rejected"); - let msg = format!("{err:?} {err}"); - assert!( - msg.contains("WITH-scoped"), - "expected WITH-scoped annotation rejection, got: {msg}" - ); +async fn sparql_update_annotations_land_in_named_graphs() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/sparql-ann-update/named-graphs"; + fluree + .create_ledger(ledger_id) + .await + .expect("create ledger"); + for update in [ + r"PREFIX ex: + INSERT DATA { + GRAPH { ex:a ex:knows ex:b {| ex:src ex:x |} } + GRAPH { ex:alice a ex:Person } + }", + r#"PREFIX ex: + WITH + INSERT { ?person ex:worksFor ex:acme {| ex:role "Engineer" |} } + WHERE { ?person a ex:Person }"#, + ] { + fluree + .graph(ledger_id) + .transact() + .sparql_update(update) + .commit() + .await + .expect("update"); + } + let src = select_rows( + &fluree, + ledger_id, + "PREFIX ex: + SELECT ?src WHERE { GRAPH { ex:a ex:knows ex:b {| ex:src ?src |} } }", + ) + .await; + assert_eq!(src.len(), 1, "{src:?}"); + let role = select_rows( + &fluree, + ledger_id, + "PREFIX ex: + SELECT ?role WHERE { GRAPH \ + { ex:alice ex:worksFor ex:acme {| ex:role ?role |} } }", + ) + .await; + assert_eq!(role.len(), 1, "{role:?}"); + assert_eq!(role[0]["role"]["value"], "Engineer"); } #[tokio::test] @@ -976,38 +1002,62 @@ async fn sparql_quoted_triple_with_annotation_tail_is_rejected() { ); } +/// A reified triple in SPARQL UPDATE stands for its reifier and does not +/// assert its triple (RDF 1.2), in subject and object position alike, and an +/// annotation tail on the triple it is part of still applies. #[tokio::test] -async fn sparql_update_quoted_triple_with_annotation_tail_does_not_panic() { - // Mirrors the read-side test but exercises the UPDATE path. - // `<<:s :p :o>> ~ {| :ann :v |}` used to hit - // `unreachable!()` inside `expand_annotated_triples` because - // a QuotedTriple subject reached `subject_to_object`. The - // expansion path must reject this explicitly with an - // `UnsupportedFeature` error before that helper is called. - let ledger0 = { - let fluree = FlureeBuilder::memory().build_memory(); - genesis_ledger(&fluree, "it/sparql-ann-update/quoted-triple-tail") - }; - let update = r" - PREFIX ex: - INSERT DATA { - << ex:alice ex:worksFor ex:acme >> ex:ann ex:v {| ex:role ex:eng |} . - } - "; - let parsed = fluree_db_sparql::parse_sparql(update); - if parsed.has_errors() { - // Parser may already reject this shape; that's also acceptable - // — the point of the test is "no panic in the lowering path". - return; - } - let ast = parsed.ast.unwrap(); - let mut ns = NamespaceRegistry::from_db(&ledger0.snapshot); - let err = fluree_db_transact::lower_sparql_update_ast(&ast, &mut ns, TxnOpts::default()) - .expect_err("quoted-triple subject + annotation tail must be rejected"); - let msg = format!("{err:?} {err}"); +async fn sparql_update_reified_triples_do_not_assert_their_triple() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/sparql-ann-update/reified-triples"; + fluree + .create_ledger(ledger_id) + .await + .expect("create ledger"); + fluree + .graph(ledger_id) + .transact() + .sparql_update( + r"PREFIX ex: + INSERT DATA { + << ex:alice ex:worksFor ex:acme ~ ex:claim >> ex:source ex:hr {| ex:seenBy ex:bob |} . + ex:doc ex:cites << ex:x ex:y ex:z >> . + }", + ) + .commit() + .await + .expect("INSERT DATA with reified triples"); + + let rows = |select: &'static str| select_rows(&fluree, ledger_id, select); + assert!( + rows("PREFIX ex: SELECT ?o WHERE { ex:alice ex:worksFor ?o }") + .await + .is_empty(), + "the reified triple is not asserted" + ); + let claim = rows( + "PREFIX ex: + SELECT ?r ?s WHERE { << ex:alice ex:worksFor ex:acme ~ ?r >> ex:source ?s }", + ) + .await; + assert_eq!(claim.len(), 1, "{claim:?}"); + assert_eq!(claim[0]["r"]["value"], "http://example.org/claim"); + let seen = rows( + "PREFIX ex: + SELECT ?w WHERE { ex:claim ex:source ex:hr {| ex:seenBy ?w |} }", + ) + .await; + assert_eq!(seen.len(), 1, "{seen:?}"); + let cited = rows( + "PREFIX ex: + PREFIX rdf: + SELECT ?r WHERE { ex:doc ex:cites ?r . ?r rdf:reifies <<( ex:x ex:y ex:z )>> }", + ) + .await; + assert_eq!(cited.len(), 1, "{cited:?}"); assert!( - msg.contains("quoted-triple") || msg.contains("annotation tail"), - "expected UnsupportedFeature on quoted-triple + tail, got: {msg}" + rows("PREFIX ex: SELECT ?z WHERE { ex:x ex:y ?z }") + .await + .is_empty() ); } diff --git a/fluree-db-transact/src/lower_sparql_update.rs b/fluree-db-transact/src/lower_sparql_update.rs index 20947970ff..6e3b0c5a3b 100644 --- a/fluree-db-transact/src/lower_sparql_update.rs +++ b/fluree-db-transact/src/lower_sparql_update.rs @@ -50,8 +50,8 @@ use fluree_db_sparql::ast::{ AnnotationUnit, AnnotationVerb, BlankNode, BlankNodeValue, GraphMgmtRef, GraphOrDefault, GraphPattern, GraphRefAll, GraphTransfer, Iri, IriValue, Literal, LiteralValue as SparqlLiteralValue, Load, Modify, PredicateTerm, Prologue, PropertyPath, - QuadData, QuadPattern, QuadPatternElement, QueryBody, ReifierId, SparqlAst, SubjectTerm, Term, - TriplePattern, TripleTerm as SparqlTripleTerm, UpdateOperation, + QuadData, QuadPattern, QuadPatternElement, QueryBody, QuotedTriple, ReifierId, SparqlAst, + SubjectTerm, Term, TriplePattern, TripleTerm as SparqlTripleTerm, UpdateOperation, }; use fluree_db_sparql::SourceSpan; use thiserror::Error; @@ -272,13 +272,11 @@ fn anon_in_mode_msg(op: &'static str) -> &'static str { } } -/// Expand any annotated triples in a Vec into the equivalent set of -/// unannotated triples, as RDF 1.2 defines the annotation syntax: the base -/// triple, `reifier rdf:reifies <<( s p o )>>` per annotation, and the body's -/// predicate-object pairs. -/// -/// Default-graph only in v1; an annotation tail inside a `GRAPH` block -/// is rejected by the caller before this is invoked. +/// Expand any reified or annotated triples in a Vec into the equivalent set +/// of plain triples, as RDF 1.2 defines them. A reified triple (`<< s p o ~ +/// r >>`) becomes its reifier plus `r rdf:reifies <<( s p o )>>`, without +/// `s p o`; an annotated triple becomes the base triple, `reifier rdf:reifies +/// <<( s p o )>>` per annotation, and the body's predicate-object pairs. fn expand_annotated_triples( triples: &mut Vec, mode: AnnotationExpansionMode, @@ -287,21 +285,18 @@ fn expand_annotated_triples( let original = std::mem::take(triples); let mut out: Vec = Vec::with_capacity(original.len()); - for tp in original { + for mut tp in original { + if let SubjectTerm::QuotedTriple(qt) = &tp.subject { + tp.subject = reify(qt, mode, bnodes, &mut out)?; + } + if let Term::QuotedTriple(qt) = &tp.object { + tp.object = reify(qt, mode, bnodes, &mut out)?.into(); + } let Some(annotation) = tp.annotation.clone() else { out.push(tp); continue; }; - // A reified triple as the annotated triple's subject is deferred. - if let SubjectTerm::QuotedTriple(qt) = &tp.subject { - return Err(LowerError::UnsupportedFeature { - feature: "RDF-star quoted-triple subject combined with an RDF 1.2 \ - annotation tail (`{| ... |}`) in SPARQL UPDATE", - span: qt.span, - }); - } - if let SubjectTerm::TripleTerm(tt) = &tp.subject { return Err(LowerError::UnsupportedFeature { feature: "SPARQL 1.2 triple-term subject combined with an RDF 1.2 \ @@ -368,10 +363,45 @@ fn expand_annotated_triples( Ok(()) } -/// Walk the QuadPatternElement list and expand every annotated triple -/// in-place. Annotation tails inside a GRAPH block are rejected with a -/// "deferred to a follow-up" message so the v1 default-graph contract -/// stays unambiguous. +/// The reifier `qt` denotes, after pushing its `rdf:reifies` link (and any +/// nested reified triple's) onto `out`. +fn reify( + qt: &QuotedTriple, + mode: AnnotationExpansionMode, + bnodes: &mut BlankNodeCounter, + out: &mut Vec, +) -> Result { + let subject = match qt.subject.as_ref() { + SubjectTerm::QuotedTriple(inner) => reify(inner, mode, bnodes, out)?, + subject => subject.clone(), + }; + let object = match qt.object.as_ref() { + Term::QuotedTriple(inner) => reify(inner, mode, bnodes, out)?.into(), + object => object.clone(), + }; + let unit = AnnotationUnit { + reifier: qt.reifier.as_ref().and_then(|r| r.id.clone()), + block: None, + span: qt.span, + }; + let reifier = resolve_reifier(&unit, mode, bnodes)?; + out.push(TriplePattern::new( + reifier.clone(), + PredicateTerm::Iri(Iri::full(fluree_vocab::rdf::REIFIES, qt.span)), + Term::TripleTerm(Box::new(SparqlTripleTerm { + subject, + predicate: qt.predicate.clone(), + object, + span: qt.span, + })), + qt.span, + )); + Ok(reifier) +} + +/// Walk the QuadPatternElement list and expand every reified or annotated +/// triple in-place, inside `GRAPH` blocks too: the link and body land in the +/// block's graph, beside the triple. fn expand_annotated_triples_in_quad_pattern( pattern: &mut QuadPattern, mode: AnnotationExpansionMode, @@ -386,16 +416,10 @@ fn expand_annotated_triples_in_quad_pattern( QuadPatternElement::Triple(t) => default_triples.push(*t), QuadPatternElement::Graph { name, - triples, + mut triples, span, } => { - if triples.iter().any(|t| t.annotation.is_some()) { - return Err(LowerError::UnsupportedFeature { - feature: "annotation tail inside a GRAPH block in SPARQL UPDATE \ - (default-graph only in v1)", - span, - }); - } + expand_annotated_triples(&mut triples, mode, bnodes)?; graph_blocks.push(QuadPatternElement::Graph { name, triples, @@ -587,28 +611,6 @@ fn reject_user_authored_reifies_in_quad_pattern( Ok(()) } -/// Reject RDF 1.2 annotation tails on `WITH `-scoped template triples: -/// SPARQL UPDATE annotations are default-graph only for now. Annotation tails -/// inside explicit `GRAPH { ... }` blocks are rejected by -/// [`expand_annotated_triples_in_quad_pattern`]; this covers the top-level -/// (WITH-scoped) triples it would otherwise expand as default-graph. -fn reject_with_scoped_annotations(pattern: &QuadPattern) -> Result<(), LowerError> { - for el in &pattern.patterns { - if let QuadPatternElement::Triple(tp) = el { - if tp.annotation.is_some() { - return Err(LowerError::UnsupportedFeature { - feature: "RDF 1.2 annotation tail (`{| ... |}`) on a WITH-scoped \ - SPARQL UPDATE template (SPARQL UPDATE annotations are \ - default-graph only; use the JSON-LD @annotation surface to \ - annotate an edge in a named graph)", - span: tp.span, - }); - } - } - } - Ok(()) -} - /// Assign stable variable names for SPARQL blank nodes when lowering /// DELETE WHERE forms — the triple-only fast path directly, and the /// GRAPH-bearing path via [`rewrite_blank_nodes_to_vars`]. @@ -1451,9 +1453,6 @@ fn lower_modify( let delete_templates = if let Some(delete_clause) = &modify.delete_clause { reject_blank_nodes_in_delete_quad_pattern(delete_clause, "DELETE templates")?; reject_user_authored_reifies_in_quad_pattern(delete_clause, prologue)?; - if default_template_graph.is_some() { - reject_with_scoped_annotations(delete_clause)?; - } let mut expanded = delete_clause.clone(); expand_annotated_triples_in_quad_pattern( &mut expanded, @@ -1475,9 +1474,6 @@ fn lower_modify( let insert_templates = if let Some(insert_clause) = &modify.insert_clause { reject_user_authored_reifies_in_quad_pattern(insert_clause, prologue)?; - if default_template_graph.is_some() { - reject_with_scoped_annotations(insert_clause)?; - } let mut expanded = insert_clause.clone(); expand_annotated_triples_in_quad_pattern( &mut expanded, @@ -2823,21 +2819,28 @@ mod tests { ); } - /// The pre-existing QuotedTriple twin of the guard: `<< s p o >>` subject - /// + annotation tail is likewise a clean error, not a panic. + /// A reified triple stands for its reifier and is not asserted: `<< s p + /// o >> q o2 {| a b |}` writes the reifier's link, `r q o2` and that + /// triple's annotation, and no `s p o`. #[test] - fn test_quoted_triple_subject_with_annotation_tail_is_rejected() { - let result = parse_and_lower( + fn test_reified_triple_subject_with_annotation_tail_does_not_assert_it() { + let txn = parse_and_lower( r"PREFIX ex: INSERT DATA { << ex:s ex:p ex:o >> ex:q ex:o2 {| ex:a ex:b |} }", - ); + ) + .expect("lower"); + let links = txn + .insert_templates + .iter() + .filter(|t| matches!(&t.object, TemplateTerm::TripleTerm(_))) + .count(); + assert_eq!(links, 2, "the reified triple's link and the annotation's"); assert!( - matches!( - &result, - Err(LowerError::UnsupportedFeature { feature, .. }) - if feature.contains("quoted-triple subject") - ), - "expected a clean quoted-triple-subject UnsupportedFeature, got {result:?}" + !txn.insert_templates + .iter() + .any(|t| matches!(&t.subject, TemplateTerm::Sid(sid) if &*sid.name == "s")), + "the reified triple is not asserted: {:?}", + txn.insert_templates ); } } From 549a5ed6c7acbab8558de39b582b401fa9ef6f01 Mon Sep 17 00:00:00 2001 From: bplatz Date: Sat, 3 Oct 2026 10:41:39 -0400 Subject: [PATCH 59/92] feat: triple terms are values under any predicate (RDF 1.2) `ex:doc ex:mentions <<( ex:s ex:p ex:o )>>` stores the term itself: it neither asserts nor reifies its triple, and `?r rdf:reifies ?t` does not find it. Every write surface takes one in object position: - Turtle, TriG, N-Triples and N-Quads (insert, upsert, bulk import, named-graph blocks included) - SPARQL UPDATE: INSERT DATA, DELETE DATA, DELETE WHERE and templates - JSON-LD: the triple's node as the value's `@id`, `{"@id": {"@id": s, p: o}}` (insert, update, import) Queries match one as a constant or by its components, in SPARQL and JSON-LD (`{"@id": {"@id": "ex:s", "ex:p": "?o"}}`). Results, CONSTRUCT output and exports write it back in the same forms, so it round-trips; RDF/XML has no syntax for one and refuses it. - graph IR: `Term::TripleTerm` and `GraphSink::term_triple`, implemented by the transact and import sinks - JSON-LD expansion kept no `@id` that was not a string, so an embedded node became a fresh blank node; it now expands as the triple's node - JSON-LD export skipped an untranslated triple-term row - CONSTRUCT wrote a bound triple term as an `f:tripleTerm` string literal - DELETE WHERE with a triple-term object takes the graph-pattern lane; the triple-only lane's reifier arm is gone - W3C: order-1 and RDF 1.2 Turtle syntax turtle12-3/7/8 now pass; the remaining triple-term failures are nested triple terms and triple-term VALUES data --- docs/concepts/edge-annotations.md | 21 +- docs/guides/cookbook-edge-annotations.md | 4 +- docs/guides/cookbook-time-travel.md | 2 +- docs/query/construct.md | 5 +- docs/query/jsonld-query.md | 19 +- docs/query/output-formats.md | 7 +- docs/query/sparql.md | 5 +- docs/reference/compatibility.md | 21 +- docs/transactions/insert.md | 10 + docs/transactions/turtle.md | 2 +- fluree-db-api/src/export.rs | 81 +++-- fluree-db-api/src/format/construct.rs | 3 + fluree-db-api/src/format/rdf_xml.rs | 11 +- fluree-db-api/src/tx.rs | 14 + fluree-db-api/tests/it_triple_term_links.rs | 313 +++++++++++++++++- fluree-db-query/src/ir.rs | 2 +- fluree-db-query/src/ir/term_components.rs | 21 ++ fluree-db-query/src/parse/ast.rs | 10 +- fluree-db-query/src/parse/lower.rs | 17 + fluree-db-query/src/parse/node_map.rs | 68 +++- fluree-db-r2rml/src/loader/extractor.rs | 1 + fluree-db-sparql/src/lower/rdf_star.rs | 20 ++ fluree-db-transact/src/flake_sink.rs | 49 ++- fluree-db-transact/src/generate/flakes.rs | 14 + fluree-db-transact/src/import.rs | 26 +- fluree-db-transact/src/import_sink.rs | 62 +++- fluree-db-transact/src/lower_sparql_update.rs | 153 +++------ fluree-db-transact/src/parse/jsonld.rs | 52 ++- fluree-db-transact/src/parse/trig_meta.rs | 100 ++++-- fluree-graph-format/src/jsonld.rs | 18 +- fluree-graph-format/src/rdf_text.rs | 16 + fluree-graph-ir/src/sink.rs | 49 ++- fluree-graph-ir/src/term.rs | 16 +- fluree-graph-json-ld/src/adapter.rs | 49 +++ fluree-graph-json-ld/src/expand.rs | 21 +- fluree-graph-turtle/src/adapter.rs | 17 +- fluree-graph-turtle/src/parser.rs | 84 +++-- testsuite-sparql/src/manifest.rs | 2 +- testsuite-sparql/src/result_comparison.rs | 11 +- testsuite-sparql/src/result_format.rs | 22 ++ testsuite-sparql/tests/registers/mod.rs | 29 +- 41 files changed, 1145 insertions(+), 302 deletions(-) diff --git a/docs/concepts/edge-annotations.md b/docs/concepts/edge-annotations.md index 2346e9765e..1ae8b9c13a 100644 --- a/docs/concepts/edge-annotations.md +++ b/docs/concepts/edge-annotations.md @@ -27,7 +27,7 @@ If a fact is naturally about a *node* (Alice's birthdate, Acme's industry), put |---|---|---| | **JSON-LD insert / upsert / update** | `@annotation` (or alias `@edge`) on a value object | Most ergonomic. Covers literal-valued edges (with explicit `@type` / `@language`), parallel annotations, named reifiers, and **named-graph edges** (an annotation on an edge inside a named graph is written into that same graph, keeping the edge's graph identity). A node's `@reifies` (`{"@id": s, p: o}`, or an array of them) makes the node a reifier of that triple **without asserting it**, as `r rdf:reifies <<( s p o )>>` does. | | **SPARQL 1.2 UPDATE** | `INSERT DATA { :s :p :o {\| ... \|} }`, `~ `, optional `INSERT { } WHERE { }` templates | Use this when integrating with SPARQL pipelines or when porting from RDF 1.2 / SPARQL-star. Works inside `GRAPH { }` blocks and under `WITH `; `<< :s :p :o ~ :r >>` reifies without asserting. See [SPARQL 1.2 surface](#sparql-12--rdf-12-surface) below for the per-operation rules. | -| **Turtle / N-Triples / TriG / N-Quads ingest** (`insert`, `upsert`, bulk `import`, graph sync (`fluree sync`, `/sync`), memory import; TriG via `insert` / `upsert` / `import` / `fluree sync` / `/sync`) | RDF 1.2 forms: the annotation syntax `:s :p :o ~ {\| ... \|}`, `<< :s :p :o ~ :r >>` in subject or object position, and the canonical `:r rdf:reifies <<( :s :p :o )>>` (the only star spelling N-Triples and N-Quads have) — in the default graph and inside TriG `GRAPH { }` blocks alike | Same stored link as `@annotation`. An annotation inside a `GRAPH { }` block, or on an N-Quads statement with a graph label, is written into that graph with the edge's graph identity, exactly as JSON-LD `@graph` + `@annotation` does. As in RDF 1.2, only the annotation syntax asserts the triple: `<< s p o >>` and `rdf:reifies <<( s p o )>>` reify it without asserting it, so N-Triples and N-Quads state an annotated triple as the triple plus its reifier's link. Rejected with a specific error: `<<( ... )>>` anywhere other than the object of `rdf:reifies`, nested triple terms, star constructs inside an annotation body, annotations on collections, and annotations in a TriG `<#txn-meta>` block. Paths that convert to JSON-LD first (`upsert`, `graph sync`, memory import) also reject an annotation on an `rdf:type` edge. See [Turtle ingest](../transactions/turtle.md#edge-annotations-rdf-12--turtle-star). | +| **Turtle / N-Triples / TriG / N-Quads ingest** (`insert`, `upsert`, bulk `import`, graph sync (`fluree sync`, `/sync`), memory import; TriG via `insert` / `upsert` / `import` / `fluree sync` / `/sync`) | RDF 1.2 forms: the annotation syntax `:s :p :o ~ {\| ... \|}`, `<< :s :p :o ~ :r >>` in subject or object position, and the canonical `:r rdf:reifies <<( :s :p :o )>>` (the only star spelling N-Triples and N-Quads have) — in the default graph and inside TriG `GRAPH { }` blocks alike | Same stored link as `@annotation`. An annotation inside a `GRAPH { }` block, or on an N-Quads statement with a graph label, is written into that graph with the edge's graph identity, exactly as JSON-LD `@graph` + `@annotation` does. As in RDF 1.2, only the annotation syntax asserts the triple: `<< s p o >>` and `rdf:reifies <<( s p o )>>` reify it without asserting it, so N-Triples and N-Quads state an annotated triple as the triple plus its reifier's link. Rejected with a specific error: `<<( ... )>>` as a subject, nested triple terms, star constructs inside an annotation body, annotations on collections, and annotations in a TriG `<#txn-meta>` block. Paths that convert to JSON-LD first (`upsert`, `graph sync`, memory import) also reject an annotation on an `rdf:type` edge. See [Turtle ingest](../transactions/turtle.md#edge-annotations-rdf-12--turtle-star). | Mint annotations through `@annotation` / `@edge` (JSON-LD) or the RDF 1.2 forms (`~`, `{| |}`, `<< >>`, `rdf:reifies <<( )>>`) in SPARQL UPDATE and Turtle. The [`f:reifies*` predicates](../reference/vocabulary.md#edge-annotation-predicates-reserved) earlier releases stored annotations under are reserved, and the write surfaces reject them; bulk import reads them, as an export written before links carries them, and stores each annotation's link instead. @@ -141,7 +141,7 @@ A few rules that keep the annotation's identity in sync with the base flake: - **Language-tagged literals are language-pinned.** Two annotations on `"chat"@fr` and `"chat"@en` are independent; selector-form retracts and hydration both match on language. - **Hydration promotes annotated literals to value-object form.** A subject expansion (`select: {"?s": ["*"]}`) renders unannotated `ex:name "Alice"` as the scalar `"Alice"`, but renders the annotated form as `{"@value": "Alice", "@annotation": {...}}` so the annotation has somewhere to attach. -The deferred shapes from "Current limits" below (list occurrences, multi-triple reifiers, triple terms as object values) still apply on the literal path. +The deferred shapes from "Current limits" below (list occurrences, nested triple terms) still apply on the literal path. ### Querying inline: edge first, metadata second @@ -427,7 +427,7 @@ Notes: - An `annotationBlock` without a preceding `~` mints a fresh anonymous reifier. - A bare `~` (no identifier) is equivalent to `~` + a fresh blank node — useful when you want a reifier variable bound in WHERE but don't care about its IRI. -- `tripleTerm` (the parenthesized `<<( s p o )>>` form) is accepted **only** as the object of `rdf:reifies`. Other uses error at parse time. +- `tripleTerm` (the parenthesized `<<( s p o )>>` form) is a value in object position: a reifier's triple under `rdf:reifies`, a stored value under any other predicate (see [Triple terms as values](#triple-terms-as-values)). As a subject, or nested in another triple term, it errors at parse time. - Property-path triples cannot carry an annotation tail. `?s ex:p1/ex:p2 ?o {| ... |}` is rejected — write a simple-predicate triple instead. ### SPARQL UPDATE rules by operation @@ -520,11 +520,20 @@ A reifier may describe a triple that is not in the graph ("X claims Alice works Reified-triple patterns (`<< :alice :worksFor ?o ~ ?r >>`, `?r rdf:reifies <<( … )>>`, JSON-LD `@reifies`) find such a reifier; the annotation syntax (`:alice :worksFor ?o {| … |}`, JSON-LD `@annotation`, a Cypher relationship) matches only asserted triples, as RDF 1.2 defines it. Export writes the reifier as its link (`@reifies` in JSON-LD), so it round-trips. +#### Triple terms as values + +A triple term is also an ordinary value under any predicate. `ex:doc ex:mentions <<( ex:s ex:p ex:o )>>` stores the term itself: it neither asserts `ex:s ex:p ex:o` nor reifies it, so `?r rdf:reifies ?t` does not find `ex:doc`. Every write surface takes one in object position: + +- Turtle, TriG, N-Triples and N-Quads: `<<( s p o )>>`. +- SPARQL UPDATE: `INSERT DATA`, `DELETE DATA`, `DELETE WHERE` and templates. +- JSON-LD: the triple's node as the value's `@id`, `"ex:mentions": {"@id": {"@id": "ex:s", "ex:p": {"@id": "ex:o"}}}`. + +A query matches one as a constant (`?d ex:mentions <<( ex:s ex:p ex:o )>>`) or by its components (`?d ex:mentions <<( ex:s ?p ?o )>>`; in JSON-LD, `{"@id": {"@id": "ex:s", "ex:p": "?o"}}`), and `SUBJECT`, `PREDICATE` and `OBJECT` take a bound one apart (JSON-LD names them `subject`, `predicate` and `object`; see [JSON-LD query](../query/jsonld-query.md#triple-term-functions)). Results, CONSTRUCT output and exports write it back in the same forms (see [Output formats](../query/output-formats.md#triple-terms)), so it round-trips. + ### Deferred SPARQL shapes (rejected at parse time) These produce a clear error with a span pointing at the offending construct: -- **Triple terms outside `rdf:reifies`.** `ex:doc ex:mentions <<( :s :p :o )>>` is rejected. Use a separate annotation subject with `rdf:reifies` if you need to refer to a triple. - **Nested triple terms.** `<<( :s :p <<( :a :b :c )>> )>>` is rejected. - **Annotation on a property-path triple.** `?s ex:p1/ex:p2 ?o {| ... |}` is rejected — the grammar only attaches annotations to simple-predicate triples. - **Property paths and nested triple terms in a `CONSTRUCT` template's annotation.** A template annotation block (`{| ... |}`) takes simple predicates only, and a template triple term (`?r rdf:reifies <<( ... )>>`) cannot nest. @@ -554,7 +563,9 @@ The bare-quoted-triple form combined with an annotation tail (`<< :s :p :o >> :p Today's surface covers the common LPG / RDF-star use cases. The following are not yet supported and produce a clear validation error rather than silent partial behavior: - **Annotations on list-occurrence triples.** `@list` membership is in scope as a future extension; the on-disk format already reserves space for it. Today, annotating a list element is rejected at parse time. -- **Triple terms as stored values.** `ex:doc ex:mentions <<( ex:s ex:p ex:o )>>` is not a representable value on any write surface (JSON-LD, SPARQL, Turtle): the only accepted position for `<<( ... )>>` is the object of `rdf:reifies`, where it names an edge rather than storing a triple. Use a separate annotation subject. A query can still build and return one: `TRIPLE`, `SUBJECT`, `PREDICATE`, `OBJECT`, `isTRIPLE` and `BIND(<<( ?s ?p ?o )>> AS ?t)` work on terms, and a bound term renders in every result format (see [Output formats](../query/output-formats.md#triple-terms)). JSON-LD queries name them `triple`, `subject`, `predicate`, `object` and `isTriple` (see [JSON-LD query](../query/jsonld-query.md#triple-term-functions)). +- **Nested triple terms.** A triple term whose object is another triple term (`<<( :s :p <<( :a :b :c )>> )>>`), and a reifier of a triple whose object is one, are rejected on every write surface. +- **Triple-term constants in `VALUES`.** `VALUES ?t { <<( :s :p :o )>> }` is rejected; bind one with `BIND(<<( :s :p :o )>> AS ?t)` instead. +- **Triple-term constants in a `CONSTRUCT` template** other than the object of `rdf:reifies`. A template variable bound to a term (`CONSTRUCT { ?d :mentions ?t }`) writes it. The mandated SPARQL 1.2 `VERSION "1.2"` prologue declaration is **accepted** (lex-and-skipped): the RDF 1.2 surface runs ungated, so a conformant 1.2 client that emits the declaration parses normally. diff --git a/docs/guides/cookbook-edge-annotations.md b/docs/guides/cookbook-edge-annotations.md index 5ac1815505..80b8699ab5 100644 --- a/docs/guides/cookbook-edge-annotations.md +++ b/docs/guides/cookbook-edge-annotations.md @@ -234,7 +234,7 @@ SELECT ?person ?org WHERE { } ``` -The triple term `<<( s p o )>>` is accepted **only** as the object of `rdf:reifies`. The parenthesis-free `<< s p o ~ :r >>` is a reified triple: it stands for its reifier and does not assert `s p o` (a reifier-less `<< s p o >>` under `f:t` / `f:op` is the separate flake-metadata construct). Per-operation reifier rules (variables are template-only; blank/anonymous reifiers are rejected in `DELETE DATA`) are tabulated in the [concept doc](../concepts/edge-annotations.md#sparql-update-rules-by-operation). +The triple term `<<( s p o )>>` names a reifier's triple as the object of `rdf:reifies`; under any other predicate it is an ordinary stored value (see [Triple terms as values](../concepts/edge-annotations.md#triple-terms-as-values)). The parenthesis-free `<< s p o ~ :r >>` is a reified triple: it stands for its reifier and does not assert `s p o` (a reifier-less `<< s p o >>` under `f:t` / `f:op` is the separate flake-metadata construct). Per-operation reifier rules (variables are template-only; blank/anonymous reifiers are rejected in `DELETE DATA`) are tabulated in the [concept doc](../concepts/edge-annotations.md#sparql-update-rules-by-operation). ## Annotate an edge inside a named graph @@ -315,7 +315,7 @@ The edge and every other claim on it stay live. The named claim's body (`ex:conf - **Deleting a claim with `DELETE DATA { … ~ :claim {| … |} }` deletes the edge**, so the annotation syntax stops matching every other claim on it. Retract one claim with the JSON-LD by-id form (see [above](#retract-one-claim-and-keep-the-edge)). - **Don't write `f:reifies*` predicates by hand.** They're reserved and rejected on every write surface; they're also hidden from `?p` scans and `select: "*"`. Use `@annotation` / the annotation tail. (See [Vocabulary](../reference/vocabulary.md#edge-annotation-predicates-reserved).) - **Empty `@annotation: {}`** is a no-op in RDF mode (no subject minted); in LPG mode it mints a property-less relationship with identity. -- **Not yet supported** (all reject cleanly, no silent partial results): annotations on `@list` elements, triple terms as object values, annotation output in Turtle/CONSTRUCT, and the SPARQL 1.2 triple-term functions (`TRIPLE`, `isTRIPLE`, …). See [Current limits](../concepts/edge-annotations.md#current-limits). +- **Not yet supported** (all reject cleanly, no silent partial results): annotations on `@list` elements and nested triple terms. See [Current limits](../concepts/edge-annotations.md#current-limits). ## See also diff --git a/docs/guides/cookbook-time-travel.md b/docs/guides/cookbook-time-travel.md index edf3c75698..a67b7f0a4c 100644 --- a/docs/guides/cookbook-time-travel.md +++ b/docs/guides/cookbook-time-travel.md @@ -456,7 +456,7 @@ It is rejected on `/v1/fluree/query/{ledger}` and on the streaming endpoint `/v1 `<< ex:alice ex:name ?name >> f:t ?t` is valid; `<< ex:alice ex:name "Alice" >> f:t ?t` is rejected — there is no object binding to attach metadata to. **`<< s p o >>` and `<<( s p o )>>` are different things.** -The bare form is this Fluree-specific flake-metadata construct. The parenthesized triple term is RDF 1.2 reification, valid only as the object of `rdf:reifies`. They do not compose — see [Edge annotations](../concepts/edge-annotations.md). +The bare form is this Fluree-specific flake-metadata construct. The parenthesized triple term is an RDF 1.2 value: the object of `rdf:reifies` in a reification, or a stored value under any other predicate. They do not compose — see [Edge annotations](../concepts/edge-annotations.md). **Reasoning is rejected in history mode.** A history-range query that requests reasoning returns an error rather than silently dropping derived facts. diff --git a/docs/query/construct.md b/docs/query/construct.md index e72ed5c3ef..d6f4ec64eb 100644 --- a/docs/query/construct.md +++ b/docs/query/construct.md @@ -360,8 +360,9 @@ details of each. - `GRAPH` blocks cannot nest, and the `CONSTRUCT WHERE` shorthand has no `GRAPH` form (its template is a basic graph pattern, per SPARQL 1.1). -- A triple term in a template is accepted only as the object of `rdf:reifies`; nested triple - terms and property paths inside a template annotation block are rejected. +- A triple-term constant in a template is accepted only as the object of `rdf:reifies`; nested + triple terms and property paths inside a template annotation block are rejected. A template + variable bound to a stored triple term (`CONSTRUCT { ?d ex:mentions ?t }`) writes the term. - `?r rdf:reifies <<( s p o )>>` in a template writes the reification without `s p o`, as RDF 1.2 defines it; the annotation tail `s p o ~ ?r` writes both. `?r rdf:reifies ?t`, with `?t` bound to a triple term, writes the same as the first. A JSON-LD result writes a reification diff --git a/docs/query/jsonld-query.md b/docs/query/jsonld-query.md index d44d33863e..23a5a07916 100644 --- a/docs/query/jsonld-query.md +++ b/docs/query/jsonld-query.md @@ -1104,7 +1104,7 @@ Edge annotations attach metadata to a specific `(subject, predicate, object)` ed - **Inline form** with `@annotation` — match an edge and pull metadata about it. - **Annotation-rooted form** with `@reifies` — match metadata first, find the edges it reifies. -`@edge` is an alias for `@annotation`; the two are interchangeable. For how to *write* annotations (`@annotation` on insert), the storage model, the cardinality contract, and worked output, see the [Edge annotations](../concepts/edge-annotations.md) concept doc. Note `@reifies` is a **query-side** construct only — user-authored `@reifies` on an insert/update is rejected; write with `@annotation` instead. +`@edge` is an alias for `@annotation`; the two are interchangeable. For how to *write* annotations (`@annotation` on insert), the storage model, the cardinality contract, and worked output, see the [Edge annotations](../concepts/edge-annotations.md) concept doc. **Inline form (`@annotation`):** @@ -1154,7 +1154,22 @@ Filter by annotation metadata first, then surface the reified edge. } ``` -The base edge identified by `@reifies` is also matched as an ordinary triple, so the visibility check is automatic — if the edge is currently retracted or hidden by policy, the row drops. +As in RDF 1.2, `@reifies` matches the reifier's link only, so it also finds reifiers of triples that are not asserted. To require the edge, add it as an ordinary pattern, or query with `@annotation`. + +**Triple terms as values:** a value whose `@id` is a node describing one triple matches a stored triple term, by its components when they are variables: + +```json +{ + "@context": { "ex": "http://example.org/" }, + "select": ["?doc", "?o"], + "where": { + "@id": "?doc", + "ex:mentions": { "@id": { "@id": "ex:alice", "ex:knows": "?o" } } + } +} +``` + +See [Triple terms as values](../concepts/edge-annotations.md#triple-terms-as-values). **Subject expansion output:** diff --git a/docs/query/output-formats.md b/docs/query/output-formats.md index 492fe13f06..2d6aee695a 100644 --- a/docs/query/output-formats.md +++ b/docs/query/output-formats.md @@ -353,7 +353,7 @@ All formats use the same representation: ### Triple Terms -A triple term (an `rdf:reifies` object, or the result of `TRIPLE(...)`) is written in SPARQL +A triple term (a stored value, an `rdf:reifies` object, or the result of `TRIPLE(...)`) is written in SPARQL JSON as a `triple` term ([SPARQL 1.2 Query Results JSON Format](https://www.w3.org/TR/sparql12-results-json/)): @@ -372,7 +372,10 @@ node, the form of the JSON-LD-star community group report: {"@id": {"@id": "ex:alice", "ex:knows": {"@id": "ex:bob"}}} ``` -TSV and CSV write `<<( s p o )>>`. +TSV and CSV write `<<( s p o )>>`. In a CONSTRUCT result, Turtle, TriG, N-Triples and N-Quads +write `<<( s p o )>>` and JSON-LD the embedded node; RDF/XML has no syntax for a triple term and +refuses such a result. JSON-LD writes take the same embedded node as a value (see +[Edge annotations](../concepts/edge-annotations.md#triple-terms-as-values)). ## Rust API diff --git a/docs/query/sparql.md b/docs/query/sparql.md index 9dbd00e48e..c30fce98b6 100644 --- a/docs/query/sparql.md +++ b/docs/query/sparql.md @@ -1077,7 +1077,7 @@ SELECT ?ann ?role WHERE { } ``` -A triple term `<<( s p o )>>` is accepted **only** as the object of `rdf:reifies`. The bare (parenthesis-free) `<< s p o >>` form is a separate, Fluree-specific construct for `f:t` / `f:op` flake-metadata extraction (see [Time Travel](#history-queries) above) — the two do not compose. +Under any other predicate a triple term is an ordinary value: `?d ex:mentions <<( ex:s ?p ?o )>>` matches stored terms by their components, and `INSERT DATA { ex:doc ex:mentions <<( ex:s ex:p ex:o )>> }` stores one without asserting or reifying its triple (see [Triple terms as values](../concepts/edge-annotations.md#triple-terms-as-values)). The bare (parenthesis-free) `<< s p o >>` form is a separate, Fluree-specific construct for `f:t` / `f:op` flake-metadata extraction (see [Time Travel](#history-queries) above) — the two do not compose. ### Updating with annotations @@ -1095,10 +1095,9 @@ Annotation tails are supported in `INSERT DATA`, `DELETE DATA`, and `INSERT { } ### Boundaries (rejected at parse / lowering time) - **Simple-predicate triples only.** `?s ex:p1/ex:p2 ?o {| ... |}` (property-path) is rejected. -- **Triple terms only as `rdf:reifies` objects**; any other use errors at parse time. +- **Triple terms in object position only**, not nested, and not as `VALUES` data. - **No reserved predicates by hand.** The [system predicates](../reference/vocabulary.md#edge-annotation-predicates-reserved) that back annotations are rejected on every UPDATE clause; mint annotations only through the `~` / `{| |}` surface. - **`CONSTRUCT` template annotation blocks take simple predicates only**, and a template triple term cannot nest. -- **SPARQL 1.2 triple-term functions** (`TRIPLE`, `SUBJECT`, `PREDICATE`, `OBJECT`, `isTRIPLE`, and the `BIND(<<( ?s ?p ?o )>> AS ?t)` constructor) are deferred. ## SPARQL UPDATE diff --git a/docs/reference/compatibility.md b/docs/reference/compatibility.md index b2fff3a570..089fbc46d0 100644 --- a/docs/reference/compatibility.md +++ b/docs/reference/compatibility.md @@ -25,6 +25,7 @@ Fluree implements the W3C RDF 1.1 specification: Fluree implements the RDF 1.2 reification model used for edge annotations: - `rdf:reifies` with triple terms (`<<( s p o )>>`) as the reified object - Reifiers identified by IRI, blank node, or variable +- Triple terms as values under any predicate, on every write surface Fluree also exposes a non-standard extension that reads commit metadata off a quoted triple (`<< s p o >> f:t ?t`, `f:op ?op`) for transaction-time and @@ -42,10 +43,7 @@ The vendored W3C RDF 1.1 and RDF 1.2 Turtle suites run in CI (`testsuite-sparql/tests/w3c_rdf.rs`), with known gaps in the skip register. Not yet supported: -- Triple terms as arbitrary object values: `<<( ... )>>` is accepted on - ingest only as the object of `rdf:reifies` -- Triple terms in subject position and nested triple terms -- Multiple triples reified by a single annotation +- Nested triple terms (`<<( s p <<( ... )>> )>>`) See [Edge annotations](../concepts/edge-annotations.md). @@ -192,13 +190,14 @@ Supported query and update annotation syntax: - Annotations in `CONSTRUCT` templates (`~ ?r`, `{| ... |}`, and `?r rdf:reifies <<( s p o )>>`), written by every result format +Also supported: triple terms as values in patterns, `INSERT DATA` / `DELETE DATA` / +`DELETE WHERE` and templates, and the triple-term functions `TRIPLE()`, `SUBJECT()`, +`PREDICATE()`, `OBJECT()` and `isTRIPLE()`. + Not yet supported: -- Triple-term accessor functions: `TRIPLE()`, `SUBJECT()`, `PREDICATE()`, - `OBJECT()`, `isTRIPLE()` -- Triple terms as arbitrary values (a `CONSTRUCT` template accepts one only as the object of - `rdf:reifies`) or in subject position; multi-triple and nested annotations -- Named-graph edge annotations in SPARQL UPDATE (default graph only) -- W3C SPARQL 1.2 test-suite execution (manifests present but not yet run) +- Nested triple terms, and triple-term constants in `VALUES` data or (other than the object + of `rdf:reifies`) in a `CONSTRUCT` template +- Nested annotations (annotation-of-annotation) **Specification:** https://www.w3.org/TR/sparql12-query/ @@ -485,7 +484,7 @@ Export Fluree data to: - SPARQL 1.1 Federation: remote `SERVICE` endpoints (local-ledger `SERVICE` is supported) - Remote `LOAD` in SPARQL UPDATE - GeoSPARQL: remaining OGC functions (only `geof:distance` is implemented today) -- RDF 1.2 / SPARQL 1.2: triple terms as values and the triple-term accessor functions; the RDF 1.2 Turtle evaluation suite (blocked on triple terms) +- RDF 1.2 / SPARQL 1.2: nested triple terms **Storage:** - Additional cloud providers (GCP, Azure) diff --git a/docs/transactions/insert.md b/docs/transactions/insert.md index 1385a5e235..90612dde83 100644 --- a/docs/transactions/insert.md +++ b/docs/transactions/insert.md @@ -374,9 +374,19 @@ Inline `@annotation` queries return one row per occurrence. } ``` +**Triple terms as values:** a value whose `@id` is the node of one triple stores that triple term, without asserting or reifying the triple (see [Triple terms as values](../concepts/edge-annotations.md#triple-terms-as-values)): + +```json +{ + "@id": "ex:doc", + "ex:mentions": { "@id": { "@id": "ex:alice", "ex:knows": { "@id": "ex:bob" } } } +} +``` + **Deferred shapes** error with explicit messages: - Annotations on list-occurrence triples (`@list` membership). +- Nested triple terms (a triple term's object that is itself `{"@id": {...}}`). - More than one predicate-object pair in one `@reifies` block (use an array of blocks to reify several triples). - Annotation-of-annotation (nested `@annotation` inside an annotation body). diff --git a/docs/transactions/turtle.md b/docs/transactions/turtle.md index b68905caf5..5a784d8d4a 100644 --- a/docs/transactions/turtle.md +++ b/docs/transactions/turtle.md @@ -483,7 +483,7 @@ ex:emp1 rdf:reifies <<( ex:alice ex:worksFor ex:acme )>> . Two rules to know: - **Only the annotation syntax asserts the triple.** As RDF 1.2 defines them, `s p o ~ r` and `s p o {| … |}` put `s p o` in the graph and attach the reifier to it, while `<< s p o >>` and `r rdf:reifies <<( s p o )>>` attach the reifier without asserting `s p o`. The reifier's own triples (the annotation body) are ordinary RDF about the reifier. Each anonymous `<< s p o >>` / `{| |}` occurrence mints a fresh reifier — two textual occurrences are two annotations. -- **`<<( ... )>>` is accepted only as the object of `rdf:reifies`.** As a plain value (`ex:doc ex:mentions <<( ... )>>`), nested inside another triple term, or inside an annotation body, it is rejected with a specific "deferred" error rather than silently dropped. +- **`<<( ... )>>` is a value.** Under `rdf:reifies` it is a reifier's triple; under any other predicate (`ex:doc ex:mentions <<( ... )>>`) it is stored as a value, without asserting or reifying its triple (see [Triple terms as values](../concepts/edge-annotations.md#triple-terms-as-values)). As a subject, nested inside another triple term, or inside an annotation body, it is rejected with a specific error rather than silently dropped. TriG and N-Quads accept the same forms inside `GRAPH { }` blocks (and on N-Quads statements with a graph label). The annotation is written into that graph and carries the edge's graph identity, exactly as JSON-LD `@graph` + `@annotation` does: diff --git a/fluree-db-api/src/export.rs b/fluree-db-api/src/export.rs index 2f29774bf0..2a83ff4741 100644 --- a/fluree-db-api/src/export.rs +++ b/fluree-db-api/src/export.rs @@ -395,6 +395,9 @@ struct AnnotationContext<'a> { /// space, persisted and ephemeral. Hoisted out of the row loop: suppression /// is then a scan of at most fourteen `u32`s, not an IRI comparison. reifies_p_ids: Vec, + /// `rdf:reifies`' `p_id`s, persisted and ephemeral: a triple-term row + /// under any other predicate is a value, not a link. + link_p_ids: Vec, graph_sid: Option, } @@ -414,13 +417,33 @@ impl<'a> AnnotationContext<'a> { .filter(|(_, sid)| fluree_db_core::namespaces::is_reserved_reifies_predicate(sid)) .map(|(p_id, _)| *p_id), ); + let mut link_p_ids: Vec = resolver + .store + .find_predicate_id(fluree_vocab::rdf::REIFIES) + .into_iter() + .collect(); + link_p_ids.extend( + resolver + .ephemeral_preds_reverse + .iter() + .filter(|(_, sid)| fluree_db_core::is_rdf_reifies(sid)) + .map(|(p_id, _)| *p_id), + ); Some(Self { probe, reifies_p_ids, + link_p_ids, graph_sid: config.graph_sid.clone(), }) } + /// An `rdf:reifies` link row: replaced by annotation syntax, and written + /// as the triple it is under `--raw-reifies`. + #[inline] + fn is_link_row(&self, p_id: u32, o_type: u16) -> bool { + o_type == OType::TRIPLE_TERM.as_u16() && self.link_p_ids.contains(&p_id) + } + #[inline] fn is_reifies_row(&self, p_id: u32) -> bool { self.reifies_p_ids.contains(&p_id) @@ -447,13 +470,6 @@ impl<'a> AnnotationContext<'a> { } } -/// An `rdf:reifies` link row: replaced by annotation syntax, and written as -/// the triple it is under `--raw-reifies`. -#[inline] -fn is_link_row(o_type: u16) -> bool { - o_type == OType::TRIPLE_TERM.as_u16() -} - /// Live reifiers for every row of `batch`, row-aligned. /// /// Returns an empty vec when the export is not emitting annotation syntax; @@ -479,7 +495,7 @@ async fn batch_reifiers( for row in 0..batch.row_count { let p_id = batch.p_id.get_or(row, 0); let o_type = batch.o_type.get_or(row, 0); - if ann.is_reifies_row(p_id) || is_link_row(o_type) { + if ann.is_reifies_row(p_id) || ann.is_link_row(p_id, o_type) { continue; } let o_key = batch.o_key.get(row); @@ -705,7 +721,7 @@ fn write_turtle_batch( // Annotation syntax replaces each link with the `~ ` marker // emitted below; a legacy `f:reifies*` bundle is read as its link. if let Some(ann) = ann { - if is_link_row(o_type) + if ann.is_link_row(p_id, o_type) && ann.link_becomes_marker(resolver, s_id, (o_type, o_key, p_id), g_id)? { ann.probe.note_link_in_scope(resolver, s_id); @@ -900,7 +916,7 @@ pub async fn export_graph_jsonld( let o_type = batch.o_type.get_or(row, 0); let o_key = batch.o_key.get(row); if let Some(ann) = ann.as_ref() { - if is_link_row(o_type) + if ann.is_link_row(p_id, o_type) && ann.link_becomes_marker( &resolver, s_id, @@ -1089,21 +1105,30 @@ fn merge_untranslated_jsonld( const REIFIES_KEY: &str = "@reifies"; -/// An `rdf:reifies` link's triple as the JSON-LD `@reifies` block, -/// `{"@id": s, p: o}` — the form JSON-LD inserts take for a reification. -/// `None` for any other row, and for a nested term, which has no block. +/// An `rdf:reifies` link's triple as the JSON-LD `@reifies` block — the +/// form JSON-LD inserts take for a reification. `None` for any other row. fn reifies_jsonld( p_iri: &str, value: &FlakeValue, store: &BinaryIndexStore, prefixes: &PrefixMap, ) -> Option { - let FlakeValue::TripleTerm(term) = value else { - return None; - }; - if p_iri != fluree_vocab::rdf::REIFIES { - return None; + match value { + FlakeValue::TripleTerm(term) if p_iri == fluree_vocab::rdf::REIFIES => { + triple_term_jsonld(term, store, prefixes) + } + _ => None, } +} + +/// A triple term as the JSON-LD node naming its triple, `{"@id": s, p: o}`; +/// as a value it is wrapped as `{"@id": {...}}`. `None` for a nested term, +/// which has no such node. +fn triple_term_jsonld( + term: &fluree_db_core::TripleTermValue, + store: &BinaryIndexStore, + prefixes: &PrefixMap, +) -> Option { let meta = term.lang.clone().map(|lang| fluree_db_core::FlakeMeta { lang: Some(lang), i: None, @@ -1191,6 +1216,9 @@ fn flake_to_jsonld_raw( "@value": v, "@type": compact_iri(&dt_iri().unwrap_or_else(|| "https://ns.flur.ee/db#vector".to_string()), prefixes) })), + FlakeValue::TripleTerm(term) => { + Some(serde_json::json!({ "@id": triple_term_jsonld(term, store, prefixes)? })) + } // Temporal / other types always encode into V3 ops. _ => None, } @@ -1503,12 +1531,15 @@ fn flake_to_jsonld( } FlakeValue::Null => serde_json::Value::Null, - FlakeValue::TripleTerm(_) => { - let dt = resolve_datatype_iri(store, o_type) - .unwrap_or_else(|| format!("{}tripleTerm", fluree_vocab::fluree::DB)); - let compact_dt = compact_iri(&dt, prefixes); - serde_json::json!({ "@value": value.to_string(), "@type": compact_dt }) - } + FlakeValue::TripleTerm(term) => match triple_term_jsonld(term, store, prefixes) { + Some(node) => serde_json::json!({ "@id": node }), + None => { + let dt = resolve_datatype_iri(store, o_type) + .unwrap_or_else(|| format!("{}tripleTerm", fluree_vocab::fluree::DB)); + let compact_dt = compact_iri(&dt, prefixes); + serde_json::json!({ "@value": value.to_string(), "@type": compact_dt }) + } + }, } } @@ -1646,7 +1677,7 @@ fn write_batch( let o_type = batch.o_type.get_or(row, 0); let o_key = batch.o_key.get(row); if let Some(ann) = ann { - if is_link_row(o_type) + if ann.is_link_row(p_id, o_type) && ann.link_becomes_marker(resolver, s_id, (o_type, o_key, p_id), g_id)? { ann.probe.note_link_in_scope(resolver, s_id); diff --git a/fluree-db-api/src/format/construct.rs b/fluree-db-api/src/format/construct.rs index edc9b9f567..007f00297f 100644 --- a/fluree-db-api/src/format/construct.rs +++ b/fluree-db-api/src/format/construct.rs @@ -482,6 +482,9 @@ impl TermResolver<'_> { fn literal(&mut self, val: &FlakeValue, dtc: &DatatypeConstraint) -> Result> { if let FlakeValue::TripleTerm(term) = val { + if let Some([s, p, o]) = self.term_components(term)? { + return Ok(Some(IrTerm::triple(s, p, o))); + } return Ok(Some(IrTerm::Literal { value: LiteralValue::String(Arc::from(self.triple_term_text(term)?)), datatype: self.datatype(dtc.datatype())?, diff --git a/fluree-db-api/src/format/rdf_xml.rs b/fluree-db-api/src/format/rdf_xml.rs index 33d9668491..9b496c6b05 100644 --- a/fluree-db-api/src/format/rdf_xml.rs +++ b/fluree-db-api/src/format/rdf_xml.rs @@ -144,8 +144,8 @@ fn write_subject_attr(subject: &Term, out: &mut String) -> Result<()> { out.push('"'); Ok(()) } - Term::Literal { .. } => Err(FormatError::InvalidBinding( - "RDF/XML subjects cannot be literals".to_string(), + Term::Literal { .. } | Term::TripleTerm(_) => Err(FormatError::InvalidBinding( + "RDF/XML subjects cannot be literals or triple terms".to_string(), )), } } @@ -185,15 +185,18 @@ fn write_predicate_object( push_node_id(id.as_str(), out); out.push('"'); } - Some(Term::Literal { .. }) => { + Some(Term::Literal { .. } | Term::TripleTerm(_)) => { return Err(FormatError::InvalidBinding( - "a reifier cannot be a literal".to_string(), + "a reifier cannot be a literal or a triple term".to_string(), )) } None => {} } match object { + Term::TripleTerm(_) => Err(FormatError::InvalidBinding( + "RDF/XML output does not write triple-term values".to_string(), + )), Term::Iri(iri) if iri.starts_with("_:") => { out.push_str(r#" rdf:nodeID=""#); push_node_id(&iri[2..], out); diff --git a/fluree-db-api/src/tx.rs b/fluree-db-api/src/tx.rs index 1e1a047490..45f401ccf2 100644 --- a/fluree-db-api/src/tx.rs +++ b/fluree-db-api/src/tx.rs @@ -2455,6 +2455,20 @@ fn convert_named_graphs_to_templates( Some(DatatypeConstraint::Explicit(dt_sid)), )) } + RawObject::TripleTerm { + subject, + predicate, + object, + } => { + let (o, dtc) = convert_object(object, prefixes, ns_registry)?; + let term = fluree_db_transact::TemplateTripleTerm { + s: convert_term(subject, prefixes, ns_registry)?, + p: convert_term(predicate, prefixes, ns_registry)?, + o, + dtc, + }; + Ok((TemplateTerm::TripleTerm(Box::new(term)), None)) + } } } diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 8dafada8b4..48b16392ab 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2096,8 +2096,8 @@ async fn triple_terms_render_in_every_result_format() { /// `?r rdf:reifies ?t` in a CONSTRUCT template, with a triple term bound to /// ?t, writes what the explicit `<<( s p o )>>` template writes: the -/// reification, without asserting its triple. Under any other predicate the term (which the graph model cannot hold as an object) -/// is written as its N-Triples text. +/// reification, without asserting its triple. Under any other predicate it +/// is written as a triple-term value. #[tokio::test] async fn construct_writes_triple_terms_as_reifications() { const REIFIES: &str = ""; @@ -2185,7 +2185,7 @@ async fn construct_writes_triple_terms_as_reifications() { let nt = render(&about, FormatterConfig::ntriples()); assert!( nt.contains( - r#" "<<( \"chat\"@fr )>>"^^"# + r#" <<( "chat"@fr )>> ."# ), "{nt}" ); @@ -2546,3 +2546,310 @@ async fn links_written_into_commits_read_back() { "incremental" ); } + +/// `ex:doc ex:mentions <<( ex:s ex:p ex:o )>>` and +/// `ex:doc ex:quotes <<( ex:s ex:says "chat"@fr )>>`, as Turtle. +const TERM_VALUES: &str = "@prefix ex: .\n\ + ex:doc ex:mentions <<( ex:s ex:p ex:o )>> .\n\ + ex:doc ex:quotes <<( ex:s ex:says \"chat\"@fr )>> .\n"; + +fn term_values_jsonld() -> JsonValue { + json!({ + "@context": {"ex": "http://example.org/"}, + "@id": "ex:doc", + "ex:mentions": {"@id": {"@id": "ex:s", "ex:p": {"@id": "ex:o"}}}, + "ex:quotes": {"@id": {"@id": "ex:s", "ex:says": {"@value": "chat", "@language": "fr"}}} + }) +} + +/// [`TERM_VALUES`] reads back from `ledger`, with `docs` subjects +/// mentioning the first term, through SPARQL and JSON-LD. +async fn assert_term_values( + fluree: &fluree_db_api::Fluree, + ledger: &LedgerState, + label: &str, + docs: usize, +) { + let run = |q: &str| run_link_query(fluree, ledger, q.to_string()); + let got = run("SELECT ?s ?p ?o WHERE { ex:doc ex:mentions ?t \ + BIND(SUBJECT(?t) AS ?s) BIND(PREDICATE(?t) AS ?p) BIND(OBJECT(?t) AS ?o) }") + .await; + assert_eq!(got, strings(&[&["ex:s", "ex:p", "ex:o"]]), "[{label}]"); + let got = run( + "SELECT ?o WHERE { ex:doc ex:quotes ?t BIND(OBJECT(?t) AS ?o) \ + FILTER(LANG(?o) = \"fr\") }", + ) + .await; + assert_eq!(got.len(), 1, "[{label}] the term keeps its tag: {got:?}"); + let got = run("SELECT ?d WHERE { ?d ex:mentions <<( ex:s ex:p ex:o )>> }").await; + assert_eq!( + got.len(), + docs, + "[{label}] a constant term matches: {got:?}" + ); + let got = run("SELECT ?d WHERE { BIND(<<( ex:s ex:p ex:o )>> AS ?t) ?d ex:mentions ?t }").await; + assert_eq!(got.len(), docs, "[{label}] a built term joins: {got:?}"); + let got = run("SELECT ?r WHERE { ?r rdf:reifies ?t }").await; + assert!(got.is_empty(), "[{label}] a value is not a link: {got:?}"); + + let jsonld = |term: JsonValue| { + json!({ + "@context": {"ex": "http://example.org/"}, + "select": ["?d", "?o"], + "where": {"@id": "?d", "ex:mentions": {"@id": term}} + }) + }; + let got = support::query_jsonld_formatted( + fluree, + ledger, + &jsonld(json!({"@id": "ex:s", "ex:p": "?o"})), + ) + .await + .unwrap_or_else(|e| panic!("[{label}] JSON-LD triple-term pattern: {e}")); + assert_eq!( + got.as_array().map(Vec::len), + Some(docs), + "[{label}] JSON-LD matches the term by its components: {got}" + ); + let got = support::query_jsonld_formatted( + fluree, + ledger, + &jsonld(json!({"@id": "ex:s", "ex:p": {"@id": "ex:other"}})), + ) + .await + .expect("JSON-LD constant term"); + assert_eq!(got, json!([]), "[{label}] another term does not match"); +} + +/// A triple term is a value under any predicate, not only as a link's +/// object: it reads back, decomposes and matches as a constant from novelty, +/// a full rebuild and an incremental build, and is never read as a link. +#[tokio::test] +async fn triple_terms_are_values_under_any_predicate() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/triple-term-links:values"; + let ledger = fluree + .insert_turtle(support::genesis_ledger(&fluree, ledger_id), TERM_VALUES) + .await + .expect("insert triple-term values") + .ledger; + assert_term_values(&fluree, &ledger, "novelty", 1).await; + let constructed = support::query_sparql( + &fluree, + &ledger, + "PREFIX ex: + CONSTRUCT { ?d ex:mentions ?t } WHERE { ?d ex:mentions ?t }", + ) + .await + .expect("CONSTRUCT a term value") + .to_construct(&ledger.snapshot) + .expect("format"); + let mentions = constructed.to_string(); + assert!( + mentions.contains(r#""@id":{"@id":"ex:s","ex:p":{"@id":"ex:o"}}"#), + "CONSTRUCT writes the term, not its text: {mentions}" + ); + support::rebuild_and_publish_index(&fluree, ledger_id).await; + let ledger = fluree.ledger(ledger_id).await.expect("load"); + assert_term_values(&fluree, &ledger, "rebuild", 1).await; + + let ledger = fluree + .insert_turtle( + ledger, + "@prefix ex: .\n\ + ex:doc2 ex:mentions <<( ex:s ex:p ex:o )>> .\n\ + ex:doc2 ex:cites <<( ex:s ex:fresh ex:o )>> .", + ) + .await + .expect("insert over an index") + .ledger; + assert_term_values(&fluree, &ledger, "novelty over an index", 2).await; + + // Indexed rows and novelty rows export as terms and re-import, including + // a term whose predicate the index has never seen. + for (format, file) in [ + (fluree_db_api::export::ExportFormat::Turtle, "values.ttl"), + (fluree_db_api::export::ExportFormat::JsonLd, "values.jsonld"), + ] { + let mut buf = Vec::new(); + fluree + .export(ledger_id) + .format(format) + .write_to(&mut buf) + .await + .unwrap_or_else(|e| panic!("export {file}: {e}")); + let exported = String::from_utf8(buf).expect("utf8"); + let (reimported, ledger) = import( + &[(file, &exported)], + &format!("it/triple-term-links:{file}"), + ) + .await; + assert_term_values(&reimported, &ledger, file, 2).await; + let cited = run_link_query( + &reimported, + &ledger, + "SELECT ?p WHERE { ex:doc2 ex:cites ?t BIND(PREDICATE(?t) AS ?p) }".to_string(), + ) + .await; + assert_eq!(cited, strings(&[&["ex:fresh"]]), "[{file}] {exported}"); + } + + support::build_and_publish_index(&fluree, ledger_id).await; + let ledger = fluree.ledger(ledger_id).await.expect("load"); + assert_term_values(&fluree, &ledger, "incremental", 2).await; +} + +/// Every write surface takes a triple-term value, and the delete forms +/// remove one. +#[tokio::test] +async fn every_write_path_takes_triple_term_values() { + let fluree = FlureeBuilder::memory().build_memory(); + let fresh = |id: &str| support::genesis_ledger(&fluree, id); + + let ledger = fluree + .insert(fresh("it/tt-values:jsonld"), &term_values_jsonld()) + .await + .expect("JSON-LD insert") + .ledger; + assert_term_values(&fluree, &ledger, "JSON-LD insert", 1).await; + + let ledger = fluree + .upsert_turtle(fresh("it/tt-values:upsert"), TERM_VALUES) + .await + .expect("Turtle upsert") + .ledger; + assert_term_values(&fluree, &ledger, "Turtle upsert", 1).await; + + let ledger_id = "it/tt-values:sparql"; + fluree.create_ledger(ledger_id).await.expect("create"); + let update = |body: &str| { + let body = format!("PREFIX ex: \n{body}"); + let fluree = &fluree; + async move { + fluree + .graph(ledger_id) + .transact() + .sparql_update(&body) + .commit() + .await + .unwrap_or_else(|e| panic!("{body}: {e}")); + fluree.ledger(ledger_id).await.expect("load") + } + }; + let ledger = update( + "INSERT DATA { ex:doc ex:mentions <<( ex:s ex:p ex:o )>> ; \ + ex:quotes <<( ex:s ex:says \"chat\"@fr )>> }", + ) + .await; + assert_term_values(&fluree, &ledger, "SPARQL INSERT DATA", 1).await; + let mentions = |ledger: LedgerState| { + let fluree = &fluree; + async move { + run_link_query( + fluree, + &ledger, + "SELECT ?p ?t WHERE { ex:doc ?p ?t }".to_string(), + ) + .await + .len() + } + }; + let ledger = update("DELETE DATA { ex:doc ex:mentions <<( ex:s ex:p ex:o )>> }").await; + assert_eq!(mentions(ledger).await, 1, "DELETE DATA removes the value"); + let ledger = update("DELETE WHERE { ?d ex:quotes <<( ex:s ex:says ?o )>> }").await; + assert_eq!( + mentions(ledger).await, + 0, + "DELETE WHERE matches it by components" + ); + + let ledger = fluree + .update( + fluree.ledger("it/tt-values:jsonld").await.expect("load"), + &json!({ + "@context": {"ex": "http://example.org/"}, + "delete": {"@id": "ex:doc", "ex:mentions": {"@id": {"@id": "ex:s", "ex:p": {"@id": "ex:o"}}}} + }), + ) + .await + .expect("JSON-LD delete") + .ledger; + assert_eq!( + mentions(ledger).await, + 1, + "a JSON-LD delete removes the value" + ); + + let (imported, ledger) = + import(&[("values.ttl", TERM_VALUES)], "it/tt-values:import-ttl").await; + assert_term_values(&imported, &ledger, "Turtle import", 1).await; + let (imported, ledger) = import( + &[("values.jsonld", &term_values_jsonld().to_string())], + "it/tt-values:import-jsonld", + ) + .await; + assert_term_values(&imported, &ledger, "JSON-LD import", 1).await; + + // Graph sync stores them, and a re-sync of the same text finds no delta. + let ledger_id = "it/tt-values:sync"; + fluree.create_ledger(ledger_id).await.expect("create"); + let sync = || { + fluree.sync_named_graph_rdf_with( + ledger_id, + "http://example.org/g", + TERM_VALUES, + fluree_db_api::SyncGraphOpts::default(), + fluree_db_api::TxnOpts::default(), + None, + ) + }; + let first = sync().await.expect("sync"); + assert_eq!(first.asserted, 2, "{first:?}"); + let again = sync().await.expect("re-sync"); + assert!( + !again.committed, + "an unchanged graph has no delta: {again:?}" + ); + let synced = run_link_query( + &fluree, + &fluree.ledger(ledger_id).await.expect("load"), + "SELECT ?o WHERE { GRAPH ex:g { ex:doc ex:mentions ?t } BIND(OBJECT(?t) AS ?o) }" + .to_string(), + ) + .await; + assert_eq!(synced, strings(&[&["ex:o"]]), "graph sync"); + + // N-Quads and TriG, in the default graph and a named one. + let nquads = " \ + <<( )>> .\n\ + \ + <<( \"chat\"@fr )>> .\n\ + \ + <<( )>> \ + .\n"; + let (imported, ledger) = import(&[("values.nq", nquads)], "it/tt-values:import-nq").await; + assert_term_values(&imported, &ledger, "N-Quads import", 1).await; + let trig = format!("{TERM_VALUES}ex:g {{ ex:doc ex:mentions <<( ex:s ex:p ex:o )>> . }}\n"); + let (imported, imported_ledger) = + import(&[("values.trig", &trig)], "it/tt-values:import-trig").await; + assert_term_values(&imported, &imported_ledger, "TriG import", 1).await; + let inserted = fluree + .insert_turtle(fresh("it/tt-values:trig-insert"), &trig) + .await + .expect("TriG insert") + .ledger; + assert_term_values(&fluree, &inserted, "TriG insert", 1).await; + for (fluree, ledger, label) in [ + (&imported, &imported_ledger, "TriG import"), + (&fluree, &inserted, "TriG insert"), + ] { + let named = run_link_query( + fluree, + ledger, + "SELECT ?o WHERE { GRAPH ex:g { ex:doc ex:mentions ?t } BIND(OBJECT(?t) AS ?o) }" + .to_string(), + ) + .await; + assert_eq!(named, strings(&[&["ex:o"]]), "[{label}] in the named graph"); + } +} diff --git a/fluree-db-query/src/ir.rs b/fluree-db-query/src/ir.rs index 2c451ab186..499eaf1bc2 100644 --- a/fluree-db-query/src/ir.rs +++ b/fluree-db-query/src/ir.rs @@ -62,5 +62,5 @@ pub use projection::{ }; pub use query::{ConstructTemplate, Query, QueryOutput, Restriction, TemplateReification}; pub use reasoning::{ReasoningConfig, ReasoningModes}; -pub use term_components::{lower_reified_link, Component, TermComponentsPattern}; +pub use term_components::{lower_reified_link, lower_term_value, Component, TermComponentsPattern}; pub use triple::{Ref, Term, TriplePattern}; diff --git a/fluree-db-query/src/ir/term_components.rs b/fluree-db-query/src/ir/term_components.rs index 8243166589..c1f359228c 100644 --- a/fluree-db-query/src/ir/term_components.rs +++ b/fluree-db-query/src/ir/term_components.rs @@ -90,6 +90,27 @@ pub fn lower_reified_link( ); } +/// Lower `subject predicate <<( s p o )>>`, a triple-term value under any +/// predicate: the triple with a term variable (or a composed constant term) +/// and the term's components, as [`lower_reified_link`] lowers a link. +pub fn lower_term_value( + subject: Ref, + predicate: Ref, + term: TriplePattern, + encoder: &E, + vars: &mut VarRegistry, + out: &mut Vec, +) { + link_patterns( + subject, + term, + predicate, + || fresh_term_var(vars), + &|iri| encoder.encode_iri(iri), + out, + ); +} + /// The patterns of [`lower_reified_link`], given the `rdf:reifies` ref, the /// term variable (asked for only when the edge is not constant), and how to /// encode an IRI. diff --git a/fluree-db-query/src/parse/ast.rs b/fluree-db-query/src/parse/ast.rs index ce24b319a1..2443a80319 100644 --- a/fluree-db-query/src/parse/ast.rs +++ b/fluree-db-query/src/parse/ast.rs @@ -1164,6 +1164,13 @@ pub enum UnresolvedPattern { /// non-`@`-keyword properties of the enclosing node). body: Vec, }, + + /// `subject predicate <<( s p o )>>`: a triple term as a value. + TripleTermValue { + subject: UnresolvedTerm, + predicate: UnresolvedTerm, + term: UnresolvedTriplePattern, + }, } impl UnresolvedPattern { @@ -1331,7 +1338,8 @@ impl UnresolvedQuery { | UnresolvedPattern::Path { .. } | UnresolvedPattern::Subquery(_) | UnresolvedPattern::IndexSearch(_) - | UnresolvedPattern::VectorSearch(_) => {} + | UnresolvedPattern::VectorSearch(_) + | UnresolvedPattern::TripleTermValue { .. } => {} } } } diff --git a/fluree-db-query/src/parse/lower.rs b/fluree-db-query/src/parse/lower.rs index e44ade62c3..3251d3ce27 100644 --- a/fluree-db-query/src/parse/lower.rs +++ b/fluree-db-query/src/parse/lower.rs @@ -471,6 +471,18 @@ pub fn lower_unresolved_pattern( out.extend(lower_unresolved_patterns(body, encoder, vars, pp_counter)?); Ok(out) } + UnresolvedPattern::TripleTermValue { + subject, + predicate, + term, + } => { + let subject = lower_ref_term(subject, encoder, vars)?; + let predicate = lower_ref_term(predicate, encoder, vars)?; + let term = lower_triple_pattern(term, encoder, vars)?; + let mut out = Vec::new(); + crate::ir::lower_term_value(subject, predicate, term, encoder, vars, &mut out); + Ok(out) + } } } @@ -1476,6 +1488,11 @@ fn lower_construct_patterns( }; lower_construct_patterns(patterns, Some(&name), encoder, vars, out)?; } + UnresolvedPattern::TripleTermValue { .. } => { + return Err(ParseError::InvalidConstruct( + "a triple-term value in a CONSTRUCT template is not supported".to_string(), + )) + } // Filters, optionals and binds have no meaning in a template. _ => {} } diff --git a/fluree-db-query/src/parse/node_map.rs b/fluree-db-query/src/parse/node_map.rs index 43a81f888a..bf28e03f57 100644 --- a/fluree-db-query/src/parse/node_map.rs +++ b/fluree-db-query/src/parse/node_map.rs @@ -899,6 +899,44 @@ fn parse_property( query, ctx, ); + } else if let Some(term) = nested_map + .get("@id") + .or_else(|| nested_map.get(ctx.ctx.context.id_key.as_str())) + .filter(|id| id.is_object()) + { + // A triple term as the value: `{"@id": {"@id": s, p: o}}`. + if is_reverse || nested_map.len() > 1 { + return Err(ParseError::InvalidWhere( + "a triple term ({\"@id\": {...}}) is a value: it cannot be a \ + subject or carry properties" + .to_string(), + )); + } + let id_key = ctx.ctx.context.id_key.as_str(); + if !term + .get("@id") + .or_else(|| term.get(id_key)) + .is_some_and(JsonValue::is_string) + { + return Err(ParseError::InvalidWhere( + "a triple term must name its subject with @id".to_string(), + )); + } + let term = parse_reifies_edge( + term, + "a triple term", + ctx.ctx, + // The term names its subject, so nothing mints one. + &mut 0, + ctx.nested_counter, + ctx.object_var_parsing, + )?; + query.patterns.push(UnresolvedPattern::TripleTermValue { + subject: subject.clone(), + predicate, + term, + }); + return Ok(()); } else { // Determine the nested subject: // - If nested object has an explicit @id, use it (var or IRI). @@ -998,6 +1036,7 @@ fn parse_annotation_target( let reifies_val = map.get(REIFIES_KEY).expect("caller checked @reifies"); let edge = parse_reifies_edge( reifies_val, + REIFIES_KEY, ctx, subject_counter, nested_counter, @@ -1265,30 +1304,26 @@ fn parse_literal_edge_annotation( Ok(()) } -/// Lower the @reifies value (a node-map describing the base triple) to -/// a single `UnresolvedTriplePattern`. +/// Lower a node-map describing one triple (an `@reifies` value, or a +/// triple-term value's `@id`) to a single `UnresolvedTriplePattern`. /// /// Reuses the regular `parse_node_map` machinery via a buffer, then -/// asserts the result is exactly one triple. Multi-triple shapes and -/// non-triple patterns (paths, value objects) are deferred to v2 with -/// explicit error messages. +/// asserts the result is exactly one triple. `what` names the form in +/// errors. fn parse_reifies_edge( value: &JsonValue, + what: &str, ctx: &JsonLdParseCtx, subject_counter: &mut u32, nested_counter: &mut u32, object_var_parsing: bool, ) -> Result { let JsonValue::Object(rmap) = value else { - return Err(ParseError::InvalidWhere( - "@reifies must be a node-map describing the base triple".to_string(), - )); + return Err(ParseError::InvalidWhere(format!( + "{what} must be a node-map describing the base triple" + ))); }; - // Re-using a fresh UnresolvedQuery as a parsing buffer avoids - // duplicating node-map traversal. The result must lower to exactly - // one Triple; anything else is the deferred multi-triple-reifier - // shape. let mut buffer = UnresolvedQuery::new(ctx.context.clone()); parse_node_map( rmap, @@ -1301,7 +1336,7 @@ fn parse_reifies_edge( if buffer.patterns.len() != 1 { return Err(ParseError::InvalidWhere(format!( - "@reifies must describe exactly one base triple (got {} patterns); \ + "{what} must describe exactly one base triple (got {} patterns); \ multi-triple reifiers are deferred to v2", buffer.patterns.len() ))); @@ -1309,11 +1344,10 @@ fn parse_reifies_edge( match buffer.patterns.into_iter().next().unwrap() { UnresolvedPattern::Triple(tp) => Ok(tp), - _ => Err(ParseError::InvalidWhere( - "@reifies must describe a basic triple pattern; \ + _ => Err(ParseError::InvalidWhere(format!( + "{what} must describe a basic triple pattern; \ property paths, lists, and other shapes are deferred to v2" - .to_string(), - )), + ))), } } diff --git a/fluree-db-r2rml/src/loader/extractor.rs b/fluree-db-r2rml/src/loader/extractor.rs index 0a41de1155..1f1727ef4e 100644 --- a/fluree-db-r2rml/src/loader/extractor.rs +++ b/fluree-db-r2rml/src/loader/extractor.rs @@ -675,6 +675,7 @@ fn describe_term(term: &Term) -> String { Term::Iri(iri) => format!("IRI <{iri}>"), Term::BlankNode(_) => "a blank node".to_string(), Term::Literal { value, .. } => format!("literal \"{}\"", value.lexical()), + Term::TripleTerm(_) => format!("triple term {term}"), } } diff --git a/fluree-db-sparql/src/lower/rdf_star.rs b/fluree-db-sparql/src/lower/rdf_star.rs index 6ff484b5ab..748f9873d7 100644 --- a/fluree-db-sparql/src/lower/rdf_star.rs +++ b/fluree-db-sparql/src/lower/rdf_star.rs @@ -93,6 +93,26 @@ impl LoweringContext<'_, E> { other => self.lower_subject(other)?, }; let p = self.lower_predicate(&tp.predicate)?; + // A triple-term value: the triple and the term's components. + if let SparqlTerm::TripleTerm(tt) = &tp.object { + if tp.annotation.is_some() { + return Err(LowerError::not_implemented( + "an annotation on a triple whose object is a triple term \ + (a nested triple term)", + tt.span, + )); + } + let term = self.lower_triple_term(tt, &mut result)?; + fluree_db_query::ir::lower_term_value( + s, + p, + term, + self.encoder, + self.vars, + &mut result, + ); + continue; + } // Full constraint-preserving object lowering on the annotation // path (see `lower_object_with_constraint`); plain triples keep // only language-tag / explicit-datatype constraints (see diff --git a/fluree-db-transact/src/flake_sink.rs b/fluree-db-transact/src/flake_sink.rs index c58ea2023e..97a79359af 100644 --- a/fluree-db-transact/src/flake_sink.rs +++ b/fluree-db-transact/src/flake_sink.rs @@ -9,7 +9,7 @@ use crate::namespace::{NamespaceRegistry, NsAllocator}; use crate::value_convert::{convert_native_literal, convert_string_literal}; use fluree_db_core::DatatypeConstraint; use fluree_db_core::{Flake, FlakeMeta, FlakeValue, Sid}; -use fluree_graph_ir::{Datatype, GraphSink, LiteralValue, SinkResult, TermId}; +use fluree_graph_ir::{Datatype, GraphSink, LiteralValue, SinkError, SinkResult, TermId}; use std::collections::HashMap; use std::sync::Arc; @@ -217,6 +217,32 @@ impl<'a> FlakeSink<'a> { } } +/// The triple-term value of a resolved `<<( s p o )>>`. A nested triple term +/// as its object is refused. +pub(crate) fn triple_term_value( + s: Option, + p: Option, + o: Option<(FlakeValue, DatatypeConstraint)>, +) -> Result { + let (Some(s), Some(p), Some((o, dtc))) = (s, p, o) else { + return Err(SinkError::rejected( + "a triple term's subject and predicate must be IRIs or blank nodes", + )); + }; + if matches!(o, FlakeValue::TripleTerm(_)) { + return Err(SinkError::rejected("nested triple terms are not supported")); + } + Ok(FlakeValue::TripleTerm(Box::new( + fluree_db_core::TripleTermValue { + s, + p, + o, + dt: dtc.datatype().clone(), + lang: dtc.lang_tag().map(str::to_string), + }, + ))) +} + // --------------------------------------------------------------------------- // GraphSink implementation // --------------------------------------------------------------------------- @@ -323,6 +349,27 @@ impl GraphSink for FlakeSink<'_> { true } + fn supports_triple_terms(&self) -> bool { + true + } + + fn term_triple( + &mut self, + subject: TermId, + predicate: TermId, + object: TermId, + ) -> Result { + let term = triple_term_value( + self.resolve_sid(subject), + self.resolve_sid(predicate), + self.resolve_object(object), + )?; + Ok(self.add_term(ResolvedTerm::Literal { + value: term, + dtc: DatatypeConstraint::Explicit(fluree_db_core::triple_term_datatype_sid().clone()), + })) + } + /// The reified triple's link; the parser has already emitted the base /// triple through `emit_triple`. fn emit_reified_triple( diff --git a/fluree-db-transact/src/generate/flakes.rs b/fluree-db-transact/src/generate/flakes.rs index fd846ebf26..380ba9b5ef 100644 --- a/fluree-db-transact/src/generate/flakes.rs +++ b/fluree-db-transact/src/generate/flakes.rs @@ -606,6 +606,13 @@ pub(crate) fn validate_value_dt_pair(val: &FlakeValue, dt: &Sid) -> Result<()> { "rdf:JSON must pair only with FlakeValue::Json (got {val:?})" ))); } + if (dt == fluree_db_core::triple_term_datatype_sid()) + != matches!(val, FlakeValue::TripleTerm(_)) + { + return Err(TransactError::FlakeGeneration(format!( + "f:tripleTerm must pair only with a triple term (got {val:?})" + ))); + } Ok(()) } @@ -622,6 +629,13 @@ pub(crate) fn reified_triple_link( ann: &Sid, t: i64, ) -> Result { + if matches!(o, FlakeValue::TripleTerm(_)) { + return Err(TransactError::UnsupportedFeature( + "reifying a triple whose object is a triple term needs a nested triple \ + term, which is not supported" + .to_string(), + )); + } let dt = dtc.datatype().clone(); validate_value_dt_pair(&o, &dt)?; let term = fluree_db_core::TripleTermValue { diff --git a/fluree-db-transact/src/import.rs b/fluree-db-transact/src/import.rs index c1ed7b0132..e43af4eb22 100644 --- a/fluree-db-transact/src/import.rs +++ b/fluree-db-transact/src/import.rs @@ -650,7 +650,9 @@ mod inner { // Spool the named-graph flake under its g_id (so it enters // the index), then encode it into the commit blob. - if let Some(sc) = spool_ctx.as_mut() { + if let (Some(sc), FlakeValue::TripleTerm(term)) = (spool_ctx.as_mut(), &o) { + sc.push_named_graph_term(g_id, &s, Some(&p), term, new_t)?; + } else if let Some(sc) = spool_ctx.as_mut() { sc.push_named_graph_record( g_id, crate::import_sink::FlakeRecord { @@ -710,7 +712,7 @@ mod inner { new_t, )?; if let (Some(sc), FlakeValue::TripleTerm(term)) = (spool_ctx.as_mut(), &link.o) { - sc.push_named_graph_link(g_id, &ann, term, new_t)?; + sc.push_named_graph_term(g_id, &ann, None, term, new_t)?; } writer.push_flake(&link).map_err(|e| { TransactError::Parse(format!("failed to encode link flake: {e}")) @@ -909,6 +911,26 @@ mod inner { convert_string_literal(value, datatype, &mut NsAllocator::Cached(ns)); Ok((fv, dt, None)) } + RawObject::TripleTerm { + subject, + predicate, + object, + } => { + let s = expand_term(subject, prefixes, ns, skolem_base)?; + let p = expand_term(predicate, prefixes, ns, skolem_base)?; + let (o, dt, lang) = expand_object(object, prefixes, ns, skolem_base)?; + let dtc = match lang { + Some(lang) => fluree_db_core::DatatypeConstraint::LangTag(Arc::from(lang)), + None => fluree_db_core::DatatypeConstraint::Explicit(dt), + }; + let term = crate::flake_sink::triple_term_value(Some(s), Some(p), Some((o, dtc))) + .map_err(|e| TransactError::Parse(e.to_string()))?; + Ok(( + term, + fluree_db_core::triple_term_datatype_sid().clone(), + None, + )) + } } } diff --git a/fluree-db-transact/src/import_sink.rs b/fluree-db-transact/src/import_sink.rs index 96c2f942fa..9b89a4034b 100644 --- a/fluree-db-transact/src/import_sink.rs +++ b/fluree-db-transact/src/import_sink.rs @@ -40,7 +40,7 @@ mod inner { use fluree_db_indexer::run_index::global_dict::{DictWorkerCache, SharedDictAllocator}; use fluree_db_indexer::run_index::shared_pool::{SharedNumBigPool, SharedVectorArenaPool}; use fluree_db_indexer::run_index::spool::{SpoolFileInfo, SpoolWriter}; - use fluree_graph_ir::{Datatype, GraphSink, LiteralValue, SinkResult, TermId}; + use fluree_graph_ir::{Datatype, GraphSink, LiteralValue, SinkError, SinkResult, TermId}; use rustc_hash::FxHashMap; use std::collections::HashMap; use std::sync::Arc; @@ -586,6 +586,19 @@ mod inner { ann: &Sid, term: &fluree_db_core::TripleTermValue, t: i64, + ) -> Result<(), CommitCodecError> { + self.write_term_record(ann, None, term, t) + } + + /// Spool `s p <<( … )>>`: the term's components as a term-table + /// pseudo-record, and the statement with the term's ordinal as its + /// object. `p` is `rdf:reifies` when `None`. + fn write_term_record( + &mut self, + s: &Sid, + p: Option<&Sid>, + term: &fluree_db_core::TripleTermValue, + t: i64, ) -> Result<(), CommitCodecError> { let s_id = self.assign_subject_id(&term.s); let p_id = self.assign_predicate_id(&term.p); @@ -626,9 +639,10 @@ mod inner { i: LIST_INDEX_NONE, }); - let link_p = match self.rdf_reifies_pid { - Some(p) => p, - None => { + let link_p = match (p, self.rdf_reifies_pid) { + (Some(p), _) => self.assign_predicate_id(p), + (None, Some(p)) => p, + (None, None) => { let p = self.assign_predicate_id(fluree_db_core::rdf_reifies_sid()); self.rdf_reifies_pid = Some(p); p @@ -642,10 +656,10 @@ mod inner { d } }; - let ann_id = self.assign_subject_id(ann); + let stmt_s_id = self.assign_subject_id(s); self.records.push(RunRecord { g_id: self.g_id, - s_id: SubjectId::from_u64(ann_id), + s_id: SubjectId::from_u64(stmt_s_id), p_id: link_p, dt: link_dt, o_kind: ObjKind::TRIPLE_TERM.as_u8(), @@ -690,18 +704,19 @@ mod inner { result } - /// Spool an `rdf:reifies` link in an explicit named graph (`g_id`); - /// see [`Self::write_link_record`]. - pub fn push_named_graph_link( + /// Spool `s p <<( … )>>` in an explicit named graph (`g_id`); see + /// [`Self::write_term_record`]. + pub fn push_named_graph_term( &mut self, g_id: GraphId, - ann: &Sid, + s: &Sid, + p: Option<&Sid>, term: &fluree_db_core::TripleTermValue, t: i64, ) -> Result<(), CommitCodecError> { let saved = self.g_id; self.g_id = g_id; - let result = self.write_link_record(ann, term, t); + let result = self.write_term_record(s, p, term, t); self.g_id = saved; result } @@ -967,7 +982,7 @@ mod inner { // Write spool record only after commit encoding succeeded if let Some(ctx) = &mut self.spool_ctx { if let FlakeValue::TripleTerm(term) = &o { - if let Err(e) = ctx.write_link_record(&s, term, self.t) { + if let Err(e) = ctx.write_term_record(&s, Some(&p), term, self.t) { self.encode_error.get_or_insert(e); } return; @@ -1171,6 +1186,29 @@ mod inner { true } + fn supports_triple_terms(&self) -> bool { + true + } + + fn term_triple( + &mut self, + subject: TermId, + predicate: TermId, + object: TermId, + ) -> Result { + let term = crate::flake_sink::triple_term_value( + self.resolve_sid(subject), + self.resolve_sid(predicate), + self.resolve_object(object), + )?; + Ok(self.add_term(ResolvedTerm::Literal { + value: term, + dtc: DatatypeConstraint::Explicit( + fluree_db_core::triple_term_datatype_sid().clone(), + ), + })) + } + /// The reified triple's link, written to the commit and the spool /// like any triple; the parser has already emitted the base triple /// through `emit_triple`. diff --git a/fluree-db-transact/src/lower_sparql_update.rs b/fluree-db-transact/src/lower_sparql_update.rs index 6e3b0c5a3b..51b044c42a 100644 --- a/fluree-db-transact/src/lower_sparql_update.rs +++ b/fluree-db-transact/src/lower_sparql_update.rs @@ -1074,15 +1074,14 @@ fn lower_delete_where( &mut local_bnodes, )?; - // `GRAPH { ... }` blocks route through the same Modify machinery - // that DELETE/INSERT ... WHERE uses (staging-time SPARQL WHERE lowering + - // graph-scoped delete templates). The triple-only fast path below stays - // byte-identical for patterns without GRAPH blocks. - if expanded_pattern - .patterns - .iter() - .any(|el| matches!(el, QuadPatternElement::Graph { .. })) - { + // `GRAPH { ... }` blocks and triple-term objects route through the + // same Modify machinery that DELETE/INSERT ... WHERE uses (staging-time + // SPARQL WHERE lowering + graph-scoped delete templates). The triple-only + // fast path below stays byte-identical for patterns without either. + if expanded_pattern.patterns.iter().any(|el| match el { + QuadPatternElement::Graph { .. } => true, + QuadPatternElement::Triple(t) => matches!(t.object, Term::TripleTerm(_)), + }) { return lower_delete_where_with_graphs(&expanded_pattern, prologue, ns, vars, opts); } @@ -1104,35 +1103,14 @@ fn lower_delete_where( for tp in triples { // WHERE side: lower to UnresolvedPattern::Triple with bnodes rewritten as vars let s = subject_to_unresolved_delete_where(&tp.subject, prologue, &mut bnode_vars)?; - match &tp.object { - Term::TripleTerm(tt) if is_rdf_reifies(&tp.predicate, prologue)? => { - let edge_s = - subject_to_unresolved_delete_where(&tt.subject, prologue, &mut bnode_vars)?; - let edge_p = predicate_to_unresolved(&tt.predicate, prologue)?; - let edge_o = - object_to_unresolved_delete_where(&tt.object, prologue, &mut bnode_vars)?; - where_patterns.push(UnresolvedPattern::AnnotationTarget { - annotation: s, - edge: UnresolvedTriplePattern { - s: edge_s, - p: edge_p, - o: edge_o.term, - dtc: edge_o.dtc, - }, - body: Vec::new(), - }); - } - object => { - let p = predicate_to_unresolved(&tp.predicate, prologue)?; - let obj = object_to_unresolved_delete_where(object, prologue, &mut bnode_vars)?; - where_patterns.push(UnresolvedPattern::Triple(UnresolvedTriplePattern { - s, - p, - o: obj.term, - dtc: obj.dtc, - })); - } - } + let p = predicate_to_unresolved(&tp.predicate, prologue)?; + let obj = object_to_unresolved_delete_where(&tp.object, prologue, &mut bnode_vars)?; + where_patterns.push(UnresolvedPattern::Triple(UnresolvedTriplePattern { + s, + p, + o: obj.term, + dtc: obj.dtc, + })); // DELETE side: lower to TripleTemplate with the same bnode->var mapping delete_templates.push(lower_triple_to_delete_template_delete_where( @@ -1192,7 +1170,7 @@ fn lower_delete_where_with_graphs( with_graph_iri: None, using_default_graph_iris: Vec::new(), using_named_graph_iris: Vec::new(), - pattern: quad_pattern_to_graph_pattern(&rewritten, prologue)?, + pattern: quad_pattern_to_graph_pattern(&rewritten), }; let mut write_graphs = BTreeSet::new(); @@ -1306,10 +1284,7 @@ fn rewrite_blank_nodes_to_vars(pattern: &QuadPattern) -> QuadPattern { /// Runs of default-graph triples become one BGP; each `GRAPH |?g { ... }` /// block becomes a `GraphPattern::Graph` wrapping its own BGP. Source order is /// preserved so bindings join exactly as the user wrote them. -fn quad_pattern_to_graph_pattern( - pattern: &QuadPattern, - prologue: &Prologue, -) -> Result { +fn quad_pattern_to_graph_pattern(pattern: &QuadPattern) -> GraphPattern { let span = pattern.span; let mut parts: Vec = Vec::new(); let mut bgp: Vec = Vec::new(); @@ -1322,63 +1297,21 @@ fn quad_pattern_to_graph_pattern( triples, span: g_span, } => { - parts.extend(triples_to_graph_patterns( - std::mem::take(&mut bgp), - prologue, - span, - )?); - let inner = triples_to_graph_patterns(triples.clone(), prologue, *g_span)?; - parts.push(GraphPattern::Graph { - name: name.clone(), - pattern: Box::new(group(inner, *g_span)), - span: *g_span, - }); - } - } - } - parts.extend(triples_to_graph_patterns(bgp, prologue, span)?); - Ok(group(parts, span)) -} - -fn group(mut parts: Vec, span: SourceSpan) -> GraphPattern { - if parts.len() == 1 { - parts.pop().expect("len checked") - } else { - GraphPattern::Group { - patterns: parts, - span, - } - } -} - -/// Triples as WHERE patterns: runs of ordinary triples as BGPs, and each -/// `r rdf:reifies <<( s p o )>>` as the reifier pattern the query parser -/// builds for it. -fn triples_to_graph_patterns( - triples: Vec, - prologue: &Prologue, - span: SourceSpan, -) -> Result, LowerError> { - let mut parts = Vec::new(); - let mut bgp = Vec::new(); - for tp in triples { - let reifies = is_rdf_reifies(&tp.predicate, prologue)?; - match tp.object { - Term::TripleTerm(triple_term) if reifies => { if !bgp.is_empty() { parts.push(GraphPattern::Bgp { patterns: std::mem::take(&mut bgp), span, }); } - parts.push(GraphPattern::AnnotationTarget { - reifier: tp.subject, - predicate: tp.predicate, - triple_term, - span: tp.span, + parts.push(GraphPattern::Graph { + name: name.clone(), + pattern: Box::new(GraphPattern::Bgp { + patterns: triples.clone(), + span: *g_span, + }), + span: *g_span, }); } - object => bgp.push(TriplePattern { object, ..tp }), } } if !bgp.is_empty() { @@ -1387,7 +1320,15 @@ fn triples_to_graph_patterns( span, }); } - Ok(parts) + + if parts.len() == 1 { + parts.pop().expect("len checked") + } else { + GraphPattern::Group { + patterns: parts, + span, + } + } } /// Lower Modify operation (DELETE/INSERT with WHERE). @@ -1579,7 +1520,7 @@ fn lower_triple_to_template( let result = literal_to_template(lit, prologue, ns)?; (result.term, result.dtc) } - Term::TripleTerm(tt) if is_rdf_reifies(&triple.predicate, prologue)? => { + Term::TripleTerm(tt) => { let s = subject_to_template(&tt.subject, prologue, ns, vars, bnodes)?; let p = predicate_to_template(&tt.predicate, prologue, ns, vars)?; let (o, dtc) = match &tt.object { @@ -1606,14 +1547,6 @@ fn lower_triple_to_template( }) } -/// True when `p` is `rdf:reifies`, whose object is a triple term. -fn is_rdf_reifies(p: &PredicateTerm, prologue: &Prologue) -> Result { - Ok(match p { - PredicateTerm::Iri(iri) => expand_iri(iri, prologue)? == fluree_vocab::rdf::REIFIES, - PredicateTerm::Var(_) => false, - }) -} - // ============================================================================= // Term conversion for WHERE patterns (UnresolvedTerm) // ============================================================================= @@ -1641,7 +1574,7 @@ fn subject_to_unresolved_delete_where( span: qt.span, }), SubjectTerm::TripleTerm(tt) => Err(LowerError::UnsupportedFeature { - feature: "SPARQL 1.2 triple-term value (`<<( s p o )>>`) in SPARQL UPDATE (deferred)", + feature: "a triple term (`<<( s p o )>>`) as a subject", span: tt.span, }), } @@ -1694,7 +1627,7 @@ fn object_to_unresolved_delete_where( span: qt.span, }), Term::TripleTerm(tt) => Err(LowerError::UnsupportedFeature { - feature: "SPARQL 1.2 triple-term value (`<<( s p o )>>`) in SPARQL UPDATE (deferred)", + feature: "a triple term nested in a triple term", span: tt.span, }), } @@ -1710,7 +1643,7 @@ fn lower_triple_to_delete_template_delete_where( let subject = delete_where_subject_template(&triple.subject, prologue, ns, vars, bnodes)?; let predicate = delete_where_predicate_template(&triple.predicate, prologue, ns, vars)?; let (object, dtc) = match &triple.object { - Term::TripleTerm(tt) if is_rdf_reifies(&triple.predicate, prologue)? => { + Term::TripleTerm(tt) => { let s = delete_where_subject_template(&tt.subject, prologue, ns, vars, bnodes)?; let p = delete_where_predicate_template(&tt.predicate, prologue, ns, vars)?; let (o, dtc) = delete_where_object_template(&tt.object, prologue, ns, vars, bnodes)?; @@ -1768,8 +1701,7 @@ fn delete_where_subject_template( } SubjectTerm::TripleTerm(tt) => { return Err(LowerError::UnsupportedFeature { - feature: - "SPARQL 1.2 triple-term value (`<<( s p o )>>`) in SPARQL UPDATE (deferred)", + feature: "a triple term (`<<( s p o )>>`) as a subject", span: tt.span, }); } @@ -1833,8 +1765,7 @@ fn delete_where_object_template( } Term::TripleTerm(tt) => { return Err(LowerError::UnsupportedFeature { - feature: - "SPARQL 1.2 triple-term value (`<<( s p o )>>`) in SPARQL UPDATE (deferred)", + feature: "a triple term nested in a triple term", span: tt.span, }); } @@ -1952,7 +1883,7 @@ fn subject_to_template( span: qt.span, }), SubjectTerm::TripleTerm(tt) => Err(LowerError::UnsupportedFeature { - feature: "SPARQL 1.2 triple-term value (`<<( s p o )>>`) in SPARQL UPDATE (deferred)", + feature: "a triple term (`<<( s p o )>>`) as a subject", span: tt.span, }), } @@ -2018,7 +1949,7 @@ fn object_to_template( span: qt.span, }), Term::TripleTerm(tt) => Err(LowerError::UnsupportedFeature { - feature: "SPARQL 1.2 triple-term value (`<<( s p o )>>`) in SPARQL UPDATE (deferred)", + feature: "a triple term nested in a triple term", span: tt.span, }), } diff --git a/fluree-db-transact/src/parse/jsonld.rs b/fluree-db-transact/src/parse/jsonld.rs index 98c2782f72..3352f7e0e8 100644 --- a/fluree-db-transact/src/parse/jsonld.rs +++ b/fluree-db-transact/src/parse/jsonld.rs @@ -14,7 +14,8 @@ use super::txn_meta::extract_txn_meta; use crate::error::{Result, TransactError}; use crate::ir::{ - GraphSel, InlineValues, TemplateGraph, TemplateTerm, TripleTemplate, Txn, TxnOpts, TxnType, + GraphSel, InlineValues, TemplateGraph, TemplateTerm, TemplateTripleTerm, TripleTemplate, Txn, + TxnOpts, TxnType, }; use crate::namespace::NamespaceRegistry; use fluree_db_core::DatatypeConstraint; @@ -1612,6 +1613,15 @@ fn parse_expanded_value_with_ctx( ) -> Result { match value { Value::Object(obj) => { + if let Some(Value::Object(term)) = obj.get("@id") { + if obj.len() > 1 { + return Err(TransactError::Parse( + "a triple term ({\"@id\": {...}}) is a value and cannot carry properties" + .to_string(), + )); + } + return parse_expanded_triple_term_with_ctx(term, ctx); + } // Check for @id (reference) if let Some(id) = obj.get("@id") { // If the object has additional keys, materialize it as a nested node. @@ -1718,6 +1728,46 @@ fn parse_expanded_value_with_ctx( } } +/// `{"@id": {"@id": s, p: o}}`: the triple term `<<( s p o )>>`. +fn parse_expanded_triple_term_with_ctx( + term: &serde_json::Map, + ctx: &mut TemplateParseCtx<'_>, +) -> Result { + let invalid = |msg: &str| TransactError::Parse(format!("triple term: {msg}")); + let s = match term.get("@id") { + Some(id) => parse_expanded_id_with_ctx(id, ctx)?, + None => return Err(invalid("@id must name the subject")), + }; + let mut pairs = term.iter().filter(|(k, _)| !k.starts_with('@')); + let (Some((key, values)), None) = (pairs.next(), pairs.next()) else { + return Err(invalid("it must describe exactly one triple")); + }; + let p = if key.starts_with('?') { + TemplateTerm::Var(ctx.vars.get_or_insert(key)) + } else { + TemplateTerm::Sid(ctx.ns_registry.sid_for_iri(key)) + }; + // A node with properties would assert them; a list or another term is + // not one object. + let mut asserted = Vec::new(); + let mut objects = parse_expanded_objects_with_ctx(values, ctx, &mut asserted)?; + let o = match objects.pop() { + Some(o) if objects.is_empty() && asserted.is_empty() && o.list_index.is_none() => o, + _ => return Err(invalid("it must describe exactly one triple")), + }; + if matches!(o.term, TemplateTerm::TripleTerm(_)) { + return Err(invalid("a triple term cannot nest another")); + } + Ok(ParsedValue::new(TemplateTerm::TripleTerm(Box::new( + TemplateTripleTerm { + s, + p, + o: o.term, + dtc: o.dtc, + }, + )))) +} + // Compatibility wrapper used by unit tests. #[cfg(test)] #[allow(clippy::too_many_arguments)] diff --git a/fluree-db-transact/src/parse/trig_meta.rs b/fluree-db-transact/src/parse/trig_meta.rs index 39c9829cc6..979aee9f9c 100644 --- a/fluree-db-transact/src/parse/trig_meta.rs +++ b/fluree-db-transact/src/parse/trig_meta.rs @@ -187,6 +187,12 @@ pub enum RawObject { TypedLiteral { value: String, datatype: String }, /// Language-tagged string. LangString { value: String, lang: String }, + /// A triple term, `<<( subject predicate object )>>`. + TripleTerm { + subject: RawTerm, + predicate: RawTerm, + object: Box, + }, } /// Phase 1: Parse TriG input and extract GRAPH blocks (no namespace resolution). @@ -381,6 +387,9 @@ fn raw_object_to_txn_meta_value( dt_name: dt_sid.name.to_string(), }) } + RawObject::TripleTerm { .. } => Err(TransactError::Parse( + "txn-meta does not support triple-term values".to_string(), + )), } } @@ -571,6 +580,7 @@ enum ObjectValue { value: String, lang: String, }, + TripleTerm(Box<(TermValue, TermValue, ObjectValue)>), } impl<'a> TrigMetaParser<'a> { @@ -947,15 +957,14 @@ impl<'a> TrigMetaParser<'a> { // RDF 1.2 star constructs inside GRAPH blocks (TriG-star) // --------------------------------------------------------------------- // - // Mirrors the streaming Turtle parser's asserting forms: the reified - // base triple is asserted, each anonymous occurrence mints a fresh - // reifier, `<<( … )>>` is a value only as the object of `rdf:reifies`, - // and star constructs inside an annotation body are deferred. + // Mirrors the streaming Turtle parser: each anonymous occurrence mints a + // fresh reifier, `<<( … )>>` is a value only in object position, and star + // constructs inside an annotation body are deferred. fn triple_term_value_error(&self) -> TransactError { TransactError::Parse( - "RDF 1.2 triple terms as values ('<<( … )>>') are deferred; inside a TriG \ - GRAPH block a triple term is accepted only as the object of rdf:reifies" + "a triple term ('<<( … )>>') is a value: it cannot be a subject, and \ + nested triple terms are not supported" .to_string(), ) } @@ -1046,6 +1055,24 @@ impl<'a> TrigMetaParser<'a> { if self.annotation_depth > 0 { return Err(self.annotation_of_annotation_error()); } + let (subject, predicate, object) = self.parse_triple_term_parts()?; + if matches!( + self.current().kind, + TokenKind::Tilde | TokenKind::AnnotationOpen + ) { + return Err(TransactError::Parse( + "an annotation tail on an 'rdf:reifies <<( … )>>' statement would reify \ + the reification itself (annotation-of-annotation), which is deferred; \ + annotate the base triple instead" + .to_string(), + )); + } + self.attach_reifier(&subject, &predicate, &object, reifier); + Ok(()) + } + + /// `<<( ttSubject predicate ttObject )>>`, the `<<(` token current. + fn parse_triple_term_parts(&mut self) -> Result<(TermValue, TermValue, ObjectValue)> { self.advance(); // `<<(` let subject = match self.current().kind { TokenKind::ReifiedTripleStart | TokenKind::TripleTermStart => { @@ -1072,19 +1099,7 @@ impl<'a> TrigMetaParser<'a> { ))); } self.advance(); - if matches!( - self.current().kind, - TokenKind::Tilde | TokenKind::AnnotationOpen - ) { - return Err(TransactError::Parse( - "an annotation tail on an 'rdf:reifies <<( … )>>' statement would reify \ - the reification itself (annotation-of-annotation), which is deferred; \ - annotate the base triple instead" - .to_string(), - )); - } - self.attach_reifier(&subject, &predicate, &object, reifier); - Ok(()) + Ok((subject, predicate, object)) } /// `reifier ::= '~' (iri | BlankNode)?` — the `~` is already consumed; @@ -1283,11 +1298,20 @@ impl<'a> TrigMetaParser<'a> { ) -> Result> { let mut objects = Vec::with_capacity(1); loop { - if self.check(&TokenKind::TripleTermStart) { - if !self.predicate_is_reifies(predicate)? { + if self.check(&TokenKind::TripleTermStart) && self.predicate_is_reifies(predicate)? { + self.parse_reifies_triple_term(subject)?; + } else if self.check(&TokenKind::TripleTermStart) { + if self.annotation_depth > 0 { + return Err(self.annotation_of_annotation_error()); + } + let (s, p, o) = self.parse_triple_term_parts()?; + if matches!( + self.current().kind, + TokenKind::Tilde | TokenKind::AnnotationOpen + ) { return Err(self.triple_term_value_error()); } - self.parse_reifies_triple_term(subject)?; + objects.push(ObjectValue::TripleTerm(Box::new((s, p, o)))); } else { let object = self.parse_object()?; if matches!( @@ -1655,6 +1679,19 @@ impl<'a> TrigMetaParser<'a> { /// Convert an ObjectValue to RawObject. fn convert_object_to_raw(&self, obj: &ObjectValue) -> Result { match obj { + ObjectValue::TripleTerm(term) => { + let (subject, predicate, object) = &**term; + if matches!(predicate, TermValue::BlankNode(_)) { + return Err(TransactError::Parse( + "blank nodes not allowed as predicate".to_string(), + )); + } + Ok(RawObject::TripleTerm { + subject: Self::convert_node_to_raw(subject), + predicate: Self::convert_node_to_raw(predicate), + object: Box::new(self.convert_object_to_raw(object)?), + }) + } ObjectValue::String(s) => Ok(RawObject::String(s.clone())), ObjectValue::Integer(n) => Ok(RawObject::Integer(*n)), ObjectValue::Double(n) => { @@ -1721,6 +1758,9 @@ impl<'a> TrigMetaParser<'a> { ObjectValue::BlankNode(_) => Err(TransactError::Parse( "blank nodes not allowed in txn-meta objects".to_string(), )), + ObjectValue::TripleTerm(_) => Err(TransactError::Parse( + "txn-meta does not support triple-term values".to_string(), + )), ObjectValue::LangString { value, lang } => Ok(TxnMetaValue::LangString { value: value.clone(), lang: lang.clone(), @@ -1810,6 +1850,9 @@ impl<'a> TrigMetaParser<'a> { ObjectValue::BlankNode(_) => Err(TransactError::Parse( "blank nodes not allowed in txn-meta objects".to_string(), )), + ObjectValue::TripleTerm(_) => Err(TransactError::Parse( + "txn-meta does not support triple-term values".to_string(), + )), ObjectValue::LangString { value, lang } => Ok(RawObject::LangString { value: value.clone(), lang: lang.clone(), @@ -2770,14 +2813,19 @@ ex:alice ex:note "value with a { brace" . let mut ns = test_registry(); for (label, body, needle) in [ ( - "triple term as value", - "ex:a ex:q <<( ex:s ex:p ex:o )>> .", - "triple terms as values", + "triple term as subject", + "<<( ex:s ex:p ex:o )>> ex:q ex:z .", + "cannot be a subject", ), ( "nested triple term", "ex:r rdf:reifies <<( ex:s ex:p <<( ex:x ex:y ex:z )>> )>> .", - "triple terms as values", + "nested triple terms", + ), + ( + "annotation on a triple-term value", + "ex:a ex:q <<( ex:s ex:p ex:o )>> {| ex:n 1 |} .", + "nested triple terms", ), ( "star inside annotation body", diff --git a/fluree-graph-format/src/jsonld.rs b/fluree-graph-format/src/jsonld.rs index ba786511dc..1e9fad7061 100644 --- a/fluree-graph-format/src/jsonld.rs +++ b/fluree-graph-format/src/jsonld.rs @@ -562,7 +562,7 @@ fn term_to_subject_key( match term { Term::Iri(iri) => Ok(config.compact_id_iri(iri)), Term::BlankNode(id) => Ok(bnode_renamer.rename(id)), - Term::Literal { .. } => Err(LiteralAsNode), + Term::Literal { .. } | Term::TripleTerm(_) => Err(LiteralAsNode), } } @@ -584,6 +584,20 @@ fn term_to_object( datatype, language, } => format_literal(value, datatype, language.as_deref()), + // A JSON-LD-star embedded node, as query results write a term. + Term::TripleTerm(t) => { + let mut node = Map::new(); + if let Ok(s) = term_to_subject_key(&t[0], config, bnode_renamer) { + node.insert("@id".to_string(), JsonValue::String(s)); + } + if let Term::Iri(p) = &t[1] { + node.insert( + config.compact_vocab_iri(p), + term_to_object(&t[2], config, bnode_renamer), + ); + } + json!({ "@id": node }) + } } } @@ -661,7 +675,7 @@ fn add_type_value( let type_iri = match object { Term::Iri(iri) => config.compact_vocab_iri(iri), Term::BlankNode(id) => bnode_renamer.rename(id), - Term::Literal { .. } => return, // Types should be IRIs + Term::Literal { .. } | Term::TripleTerm(_) => return, // Types should be IRIs }; match node.get_mut("@type") { diff --git a/fluree-graph-format/src/rdf_text.rs b/fluree-graph-format/src/rdf_text.rs index 0e1c2d31a1..d62297a50e 100644 --- a/fluree-graph-format/src/rdf_text.rs +++ b/fluree-graph-format/src/rdf_text.rs @@ -249,6 +249,14 @@ fn push_nt_term(out: &mut String, term: &Term) { syntax::push_iri_ref(out, iri); }); } + Term::TripleTerm(t) => { + out.push_str("<<( "); + for term in t.iter() { + push_nt_term(out, term); + out.push(' '); + } + out.push_str(")>>"); + } } } @@ -272,6 +280,14 @@ fn push_turtle_term(out: &mut String, term: &Term, prefixes: &PrefixMap) { prefixes.push_iri(out, iri); }); } + Term::TripleTerm(t) => { + out.push_str("<<( "); + for term in t.iter() { + push_turtle_term(out, term, prefixes); + out.push(' '); + } + out.push_str(")>>"); + } } } diff --git a/fluree-graph-ir/src/sink.rs b/fluree-graph-ir/src/sink.rs index fa92c2bad2..b7843f14dc 100644 --- a/fluree-graph-ir/src/sink.rs +++ b/fluree-graph-ir/src/sink.rs @@ -301,6 +301,32 @@ pub trait GraphSink { false } + /// Whether this sink accepts triple-term values (`<<( s p o )>>` as an + /// object). Parsers MUST check this before calling + /// [`Self::term_triple`]; defaults to `false`. + fn supports_triple_terms(&self) -> bool { + false + } + + /// The triple-term value `<<( subject predicate object )>>`, valid for + /// the current statement like a literal. Only called when + /// [`Self::supports_triple_terms`] returns `true`; the default refuses. + fn term_triple( + &mut self, + subject: TermId, + predicate: TermId, + object: TermId, + ) -> std::result::Result { + let _ = (subject, predicate, object); + debug_assert!( + self.supports_triple_terms(), + "term_triple called on a sink that does not support triple terms" + ); + Err(SinkError::rejected( + "this sink cannot represent triple terms", + )) + } + /// Emit an RDF 1.2 reified-triple event: `reifier` reifies the base /// triple `(subject, predicate, object)`. /// @@ -449,7 +475,10 @@ impl GraphCollectorSink { if cfg!(debug_assertions) { for &slot in &self.literal_slots[..self.literal_cursor] { debug_assert!( - matches!(self.terms[slot as usize], Term::Literal { .. }), + matches!( + self.terms[slot as usize], + Term::Literal { .. } | Term::TripleTerm(_) + ), "retiring non-literal slot {slot}: {:?} — literal_slots is polluted, \ recycling it would clobber a producer-cached term id", self.terms[slot as usize] @@ -548,6 +577,24 @@ impl GraphSink for GraphCollectorSink { self.add_literal_term(term) } + fn supports_triple_terms(&self) -> bool { + true + } + + fn term_triple( + &mut self, + subject: TermId, + predicate: TermId, + object: TermId, + ) -> std::result::Result { + let term = Term::triple( + self.get_term(subject).clone(), + self.get_term(predicate).clone(), + self.get_term(object).clone(), + ); + Ok(self.add_literal_term(term)) + } + /// Retire this statement's literal slots for reuse, and move the rewind /// point past the statement just committed. Recycling is sound because /// `emit_*` has already cloned every term it needed into the graph, and diff --git a/fluree-graph-ir/src/term.rs b/fluree-graph-ir/src/term.rs index 299723690e..0246a1c58a 100644 --- a/fluree-graph-ir/src/term.rs +++ b/fluree-graph-ir/src/term.rs @@ -234,6 +234,10 @@ pub enum Term { /// Language tag (only valid when datatype is rdf:langString) language: Option>, }, + + /// RDF 1.2 triple term `<<( s p o )>>` used as a value: subject, + /// predicate, object. + TripleTerm(Arc<[Term; 3]>), } impl Term { @@ -247,6 +251,11 @@ impl Term { Term::BlankNode(BlankId::new(label)) } + /// Create a triple term `<<( s p o )>>` + pub fn triple(s: Term, p: Term, o: Term) -> Self { + Term::TripleTerm(Arc::new([s, p, o])) + } + /// Create a plain string literal (xsd:string) pub fn string(value: impl AsRef) -> Self { Term::Literal { @@ -393,6 +402,7 @@ impl PartialEq for Term { language: l2, }, ) => v1 == v2 && d1 == d2 && l1 == l2, + (Term::TripleTerm(a), Term::TripleTerm(b)) => a == b, _ => false, } } @@ -415,6 +425,7 @@ impl Hash for Term { datatype.hash(state); language.hash(state); } + Term::TripleTerm(t) => t.hash(state), } } } @@ -427,12 +438,13 @@ impl PartialOrd for Term { impl Ord for Term { fn cmp(&self, other: &Self) -> Ordering { - // Type ordering: BlankNode < Iri < Literal + // Type ordering: BlankNode < Iri < Literal < TripleTerm let type_ord = |t: &Term| -> u8 { match t { Term::BlankNode(_) => 0, Term::Iri(_) => 1, Term::Literal { .. } => 2, + Term::TripleTerm(_) => 3, } }; @@ -457,6 +469,7 @@ impl Ord for Term { language: l2, }, ) => (d1, l1, v1).cmp(&(d2, l2, v2)), + (Term::TripleTerm(a), Term::TripleTerm(b)) => a.cmp(b), _ => Ordering::Equal, // Should not happen } } @@ -503,6 +516,7 @@ impl std::fmt::Display for Term { Ok(()) } } + Term::TripleTerm(t) => write!(f, "<<( {} {} {} )>>", t[0], t[1], t[2]), } } } diff --git a/fluree-graph-json-ld/src/adapter.rs b/fluree-graph-json-ld/src/adapter.rs index d530b02097..1e037b064b 100644 --- a/fluree-graph-json-ld/src/adapter.rs +++ b/fluree-graph-json-ld/src/adapter.rs @@ -218,6 +218,15 @@ fn process_node( fn process_value(value: &Value, sink: &mut S) -> Result { match value { Value::Object(obj) => { + if let Some(Value::Object(term)) = obj.get("@id") { + if obj.len() > 1 { + return Err(AdapterError::InvalidStructure( + "a triple term ({\"@id\": {...}}) is a value and cannot carry properties" + .to_string(), + )); + } + return Ok(ProcessedValue::Single(process_triple_term(term, sink)?)); + } // Check for @id (reference to another node) if let Some(id_val) = obj.get("@id") { // An embedded node object carrying more than a bare `@id` @@ -284,6 +293,46 @@ fn process_value(value: &Value, sink: &mut S) -> Result>`. +fn process_triple_term( + term: &serde_json::Map, + sink: &mut S, +) -> Result { + let invalid = |msg: &str| AdapterError::InvalidStructure(format!("triple term: {msg}")); + if !sink.supports_triple_terms() { + return Err(invalid("this destination does not hold triple-term values")); + } + let subject = match term.get("@id").and_then(Value::as_str) { + Some(id) if id.starts_with("_:") => sink.term_blank(Some(strip_blank_prefix(id))), + Some(id) => sink.term_iri(id), + None => return Err(invalid("@id must name the subject")), + }; + let mut pairs = term.iter().filter(|(k, _)| !k.starts_with('@')); + let (Some((predicate, values)), None) = (pairs.next(), pairs.next()) else { + return Err(invalid("it must describe exactly one triple")); + }; + let value = match values { + Value::Array(items) if items.len() == 1 => &items[0], + Value::Array(_) => return Err(invalid("it must describe exactly one triple")), + value => value, + }; + // A node with properties would assert them, and a term does not nest. + let reference_or_value = match value { + Value::Object(o) => { + o.contains_key("@value") || (o.len() == 1 && o.get("@id").is_some_and(Value::is_string)) + } + _ => true, + }; + if !reference_or_value { + return Err(invalid("its object must be a reference or a value")); + } + let ProcessedValue::Single(object) = process_value(value, sink)? else { + return Err(invalid("its object must be a reference or a value")); + }; + let predicate = sink.term_iri(predicate); + Ok(sink.term_triple(subject, predicate, object)?) +} + /// Process a @list value and return the list items with indices fn process_list(list_val: &Value, sink: &mut S) -> Result { let items = match list_val { diff --git a/fluree-graph-json-ld/src/expand.rs b/fluree-graph-json-ld/src/expand.rs index 07c474f147..f900e79121 100644 --- a/fluree-graph-json-ld/src/expand.rs +++ b/fluree-graph-json-ld/src/expand.rs @@ -629,11 +629,22 @@ fn expand_node_internal( // Handle @id if expanded_key == "@id" || k == "@id" { - if let JsonValue::String(s) = v { - result.insert( - "@id".to_string(), - json!(iri_dispatch(s, &context_with_types, false, strict)?), - ); + match v { + JsonValue::String(s) => { + result.insert( + "@id".to_string(), + json!(iri_dispatch(s, &context_with_types, false, strict)?), + ); + } + // A triple term, `{"@id": {"@id": s, p: o}}`: the + // embedded node naming its triple. + JsonValue::Object(_) => { + result.insert( + "@id".to_string(), + expand_node_internal(v, &context_with_types, &key_idx, strict)?, + ); + } + _ => {} } continue; } diff --git a/fluree-graph-turtle/src/adapter.rs b/fluree-graph-turtle/src/adapter.rs index 3b6728edb1..c87df0b6d1 100644 --- a/fluree-graph-turtle/src/adapter.rs +++ b/fluree-graph-turtle/src/adapter.rs @@ -183,8 +183,9 @@ fn term_to_subject_key(term: &Term) -> String { match term { Term::Iri(iri) => iri.to_string(), Term::BlankNode(id) => format!("_:{}", id.as_str()), - Term::Literal { .. } => { - // Literals shouldn't be subjects in RDF, but handle gracefully + Term::Literal { .. } | Term::TripleTerm(_) => { + // Literals and triple terms shouldn't be subjects in RDF, but + // handle gracefully "_:literal".to_string() } } @@ -201,6 +202,16 @@ fn term_to_iri(term: &Term) -> String { /// Convert an object term to a JSON-LD value object. fn term_to_object_value(term: &Term) -> JsonValue { match term { + // The JSON-LD-star embedded node a triple-term value takes. + Term::TripleTerm(t) => { + let mut node = Map::new(); + node.insert( + "@id".to_string(), + JsonValue::String(term_to_subject_key(&t[0])), + ); + node.insert(term_to_iri(&t[1]), term_to_object_value(&t[2])); + json!({ "@id": node }) + } Term::Iri(iri) => { json!({ "@id": iri.as_ref() }) } @@ -264,7 +275,7 @@ fn term_to_type_value(term: &Term) -> Option { match term { Term::Iri(iri) => Some(JsonValue::String(iri.to_string())), Term::BlankNode(id) => Some(JsonValue::String(format!("_:{}", id.as_str()))), - Term::Literal { .. } => None, + Term::Literal { .. } | Term::TripleTerm(_) => None, } } diff --git a/fluree-graph-turtle/src/parser.rs b/fluree-graph-turtle/src/parser.rs index a2172d65b5..2d85829c7b 100644 --- a/fluree-graph-turtle/src/parser.rs +++ b/fluree-graph-turtle/src/parser.rs @@ -1025,7 +1025,7 @@ impl<'a, 'input, S: GraphSink> Parser<'a, 'input, S> { | TokenKind::Double(_) => self.parse_literal(), TokenKind::KwTrue | TokenKind::KwFalse => self.parse_literal(), TokenKind::ReifiedTripleStart => self.parse_reified_triple(), - TokenKind::TripleTermStart => Err(self.triple_term_deferred_error()), + TokenKind::TripleTermStart => self.parse_triple_term_value(), _ => Err(TurtleError::parse( self.current().start as usize, format!("expected object, found {}", self.current().kind), @@ -1294,13 +1294,29 @@ impl<'a, 'input, S: GraphSink> Parser<'a, 'input, S> { fn triple_term_deferred_error(&self) -> TurtleError { TurtleError::parse( self.current().start as usize, - "RDF 1.2 triple terms as values ('<<( … )>>') are deferred in Turtle \ - ingest except as the object of rdf:reifies; the supported forms are \ - 'r rdf:reifies <<( s p o )>>', reified triples '<< s p o >>' with \ - optional '~ reifier', and annotation blocks '{| … |}'", + "a triple term ('<<( … )>>') is a value and cannot be a subject", ) } + /// `<<( ttSubject predicate ttObject )>>` as a value; the `<<(` token is + /// current. + fn parse_triple_term_value(&mut self) -> Result { + if !self.sink.supports_triple_terms() { + return Err(TurtleError::parse( + self.current().start as usize, + "triple terms ('<<( … )>>') as values are not supported on this ingest path", + )); + } + self.with_nesting(|p| { + p.expect(&TokenKind::TripleTermStart)?; + let subject = p.parse_tt_subject()?; + let predicate = p.parse_predicate()?; + let object = p.parse_tt_object()?; + p.expect(&TokenKind::TripleTermEnd)?; + Ok(p.sink.term_triple(subject, predicate, object)?) + }) + } + /// `r rdf:reifies <<( s p o )>>` — the RDF 1.2 spelling every reifying /// form desugars to, and the only star construct N-Triples/N-Quads have. /// The `<<(` token is current and `subject` is the reifier. Emits exactly @@ -1371,7 +1387,10 @@ impl<'a, 'input, S: GraphSink> Parser<'a, 'input, S> { self.current().kind ), )), - TokenKind::TripleTermStart => Err(self.triple_term_deferred_error()), + TokenKind::TripleTermStart => Err(TurtleError::parse( + self.current().start as usize, + "nested triple terms ('<<( … <<( … )>> )>>') are not supported", + )), _ => self.parse_object(), } } @@ -1515,7 +1534,6 @@ impl<'a, 'input, S: GraphSink> Parser<'a, 'input, S> { self.current().kind ), )), - TokenKind::TripleTermStart => Err(self.triple_term_deferred_error()), TokenKind::ReifiedTripleStart => self.parse_reified_triple(), _ => self.parse_object(), } @@ -2516,7 +2534,7 @@ mod tests { &mut sink, ) .expect_err("nested triple term is a value with no representation"); - assert!(err.to_string().contains("deferred"), "{err}"); + assert!(err.to_string().contains("nested triple terms"), "{err}"); } #[test] @@ -2560,34 +2578,36 @@ mod tests { } #[test] - fn star_triple_term_under_other_predicate_still_deferred() { - // `<<( )>>` is only a value under rdf:reifies; `:q <<( … )>>` (the - // W3C data-0-tripleterms shape) keeps the deferred error. - let mut sink = StarSink::default(); - let err = parse(&format!("{P}:a :q <<( :a :b :c )>> ."), &mut sink) - .expect_err("triple term under a non-reifies predicate"); - assert!(err.to_string().contains("triple terms as values"), "{err}"); - } - - #[test] - fn star_triple_term_rejected_with_deferred_error() { - let mut sink = StarSink::default(); - let err = parse(&format!("{P}:x1 :left <<( :a :b 123 )>> ."), &mut sink) - .expect_err("triple terms as values must be rejected"); - let msg = err.to_string(); - assert!(msg.contains("triple terms as values"), "{msg}"); - assert!(msg.contains("deferred"), "{msg}"); + fn triple_term_value_parses_as_an_object() { + // The W3C data-0-tripleterms shape: `<<( )>>` as an ordinary value, + // here and as a reified triple's object. + let mut sink = GraphCollectorSink::new(); + parse( + &format!("{P}:a :q <<( :a :b :c )>> .\n:f :g << :s :p <<( :x :y 123 )>> >> ."), + &mut sink, + ) + .expect("triple-term values parse"); + let graph = sink.into_graph(); + let iri_term = |suffix: &str| Term::iri(format!("http://example/{suffix}")); + let term = |s: &str, p: &str, o: Term| Term::triple(iri_term(s), iri_term(p), o); + assert!(graph + .triples() + .iter() + .any(|t| t.o == term("a", "b", iri_term("c")))); + let reified = &graph.reifications()[0].triple; + assert!(matches!(&reified.o, Term::TripleTerm(t) if t[0] == iri_term("x"))); } #[test] - fn star_triple_term_in_reified_triple_rejected() { + fn triple_term_value_needs_a_sink_that_holds_one() { let mut sink = StarSink::default(); - let err = parse( - &format!("{P}:f :g << :s :p <<(:x2 :y3 123 )>> >> ."), - &mut sink, - ) - .expect_err("nested triple term must be rejected"); - assert!(err.to_string().contains("triple terms as values")); + let err = parse(&format!("{P}:a :q <<( :a :b :c )>> ."), &mut sink) + .expect_err("this sink holds no triple terms"); + assert!( + err.to_string() + .contains("not supported on this ingest path"), + "{err}" + ); } #[test] diff --git a/testsuite-sparql/src/manifest.rs b/testsuite-sparql/src/manifest.rs index 8183449506..552ba8ce3e 100644 --- a/testsuite-sparql/src/manifest.rs +++ b/testsuite-sparql/src/manifest.rs @@ -333,7 +333,7 @@ fn term_to_string(term: &Term) -> Option { match term { Term::Iri(iri) => Some(iri.to_string()), Term::Literal { value, .. } => Some(value.lexical()), - Term::BlankNode(_) => None, + Term::BlankNode(_) | Term::TripleTerm(_) => None, } } diff --git a/testsuite-sparql/src/result_comparison.rs b/testsuite-sparql/src/result_comparison.rs index c5ebbe5fbd..834e66f6ae 100644 --- a/testsuite-sparql/src/result_comparison.rs +++ b/testsuite-sparql/src/result_comparison.rs @@ -175,6 +175,11 @@ fn terms_match( false } (RdfTerm::Iri(e), RdfTerm::Iri(a)) => e == a, + (RdfTerm::Triple(e), RdfTerm::Triple(a)) => { + terms_match(&e.subject, &a.subject, bnode_map) + && terms_match(&e.predicate, &a.predicate, bnode_map) + && terms_match(&e.object, &a.object, bnode_map) + } _ => false, // Type mismatch } } @@ -320,9 +325,7 @@ fn are_graphs_isomorphic(expected: &[Triple], actual: &[Triple]) -> bool { // Fast path: no blank nodes in either graph — just sort and compare. let has_bnodes = |triples: &[Triple]| { triples.iter().any(|t| { - matches!(t.subject, RdfTerm::BlankNode(_)) - || matches!(t.predicate, RdfTerm::BlankNode(_)) - || matches!(t.object, RdfTerm::BlankNode(_)) + t.subject.has_blank_node() || t.predicate.has_blank_node() || t.object.has_blank_node() }) }; @@ -353,6 +356,7 @@ fn rdf_term_sort_key(a: &RdfTerm, b: &RdfTerm) -> std::cmp::Ordering { RdfTerm::BlankNode(_) => 0, RdfTerm::Iri(_) => 1, RdfTerm::Literal { .. } => 2, + RdfTerm::Triple(_) => 3, } }; discriminant(a).cmp(&discriminant(b)).then_with(|| { @@ -371,6 +375,7 @@ fn rdf_term_sort_key(a: &RdfTerm, b: &RdfTerm) -> std::cmp::Ordering { language: bl, }, ) => av.cmp(bv).then_with(|| ad.cmp(bd)).then_with(|| al.cmp(bl)), + (RdfTerm::Triple(a), RdfTerm::Triple(b)) => triple_sort_key(a, b), _ => std::cmp::Ordering::Equal, // different discriminants already handled } }) diff --git a/testsuite-sparql/src/result_format.rs b/testsuite-sparql/src/result_format.rs index d62da400ae..1f50887b67 100644 --- a/testsuite-sparql/src/result_format.rs +++ b/testsuite-sparql/src/result_format.rs @@ -33,6 +33,22 @@ pub enum RdfTerm { datatype: Option, language: Option, }, + /// A triple term, `<<( s p o )>>`. + Triple(Box), +} + +impl RdfTerm { + pub(crate) fn has_blank_node(&self) -> bool { + match self { + RdfTerm::BlankNode(_) => true, + RdfTerm::Triple(t) => { + t.subject.has_blank_node() + || t.predicate.has_blank_node() + || t.object.has_blank_node() + } + RdfTerm::Iri(_) | RdfTerm::Literal { .. } => false, + } + } } /// An RDF triple in a CONSTRUCT/DESCRIBE result graph. @@ -205,6 +221,7 @@ pub fn project_to_csv_space(results: SparqlResults) -> SparqlResults { language: None, }, RdfTerm::BlankNode(b) => RdfTerm::BlankNode(b), + RdfTerm::Triple(t) => RdfTerm::Triple(t), RdfTerm::Literal { value, .. } => RdfTerm::Literal { value, datatype: None, @@ -1080,6 +1097,11 @@ pub(crate) fn ir_term_to_rdf_term(term: &IrTerm) -> RdfTerm { language: language_opt, } } + IrTerm::TripleTerm(t) => RdfTerm::Triple(Box::new(Triple { + subject: ir_term_to_rdf_term(&t[0]), + predicate: ir_term_to_rdf_term(&t[1]), + object: ir_term_to_rdf_term(&t[2]), + })), } } diff --git a/testsuite-sparql/tests/registers/mod.rs b/testsuite-sparql/tests/registers/mod.rs index ad413db875..e2803bae56 100644 --- a/testsuite-sparql/tests/registers/mod.rs +++ b/testsuite-sparql/tests/registers/mod.rs @@ -378,26 +378,18 @@ pub const SPARQL12_CODEPOINT_ESCAPES: &[&str] = &[]; pub const SPARQL12_SYNTAX_TRIPLE_TERMS_NEGATIVE: &[&str] = &[]; -// SPARQL 1.2 triple-term syntax is fully accepted (accept-then-defer, -// decision D-1): reifier forms (buckets A/D) by PR-W2A; the -// TRIPLE/SUBJECT/PREDICATE/OBJECT/isTRIPLE builtins (bucket B) and bare -// `<<( )>>` triple-term values (bucket C) by PR-W2BC. All parse + validate; -// evaluation is deferred to the first-class-triple-term epic (ROADMAP §2/§4, -// docs/audit/burn-down/sparql12-wave2-triple-terms.md §1.2/§1.3), so the -// sibling SPARQL12_EVAL_TRIPLE_TERMS register still stands. +// SPARQL 1.2 triple-term syntax is fully accepted: reifier forms, the +// TRIPLE/SUBJECT/PREDICATE/OBJECT/isTRIPLE builtins and `<<( )>>` triple-term +// values. What still fails to evaluate is in SPARQL12_EVAL_TRIPLE_TERMS. pub const SPARQL12_SYNTAX_TRIPLE_TERMS_POSITIVE: &[&str] = &[]; -// Blocked on Turtle-star data loading and engine triple-term support — -// audit §4.3 / Phase D. Re-baseline this whole register after Turtle-star -// ingest (PR-W15) lands: the residual blockers are wave-2 query syntax / -// triple-term functions / CONSTRUCT projection / result serialization, -// scoped by the Option-1 first-class-triple-term epic (ROADMAP §2). // Each test appears in EXACTLY ONE reason cluster (PR-1454 review found 10 // entries double-counted across clusters); attribution below re-verified // empirically by unregistering and reading the harness's failure reasons. pub const SPARQL12_EVAL_TRIPLE_TERMS: &[&str] = &[ - // data load: qt:data uses `<<( … )>>` triple-term VALUES, rejected by - // ingest with the specific deferred error — Option-1 epic (10) + // data load: qt:data nests a triple term in another + // (`<<( … <<( … )>> )>>`, or a reifier of a triple whose object is a + // triple term), which ingest refuses (9) "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#results-tripleterms-1j", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#results-tripleterms-1x", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#basic-8", @@ -406,7 +398,6 @@ pub const SPARQL12_EVAL_TRIPLE_TERMS: &[&str] = &[ "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#pattern-11", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#op-1", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#op-2", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#order-1", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#order-2", // data load: blocked on TriG GRAPH-block parsing, orthogonal to star // (D-8) — "expected subject, found 'GRAPH'" (3) @@ -593,12 +584,8 @@ pub const RDF11_TURTLE: &[&str] = &[ ]; pub const RDF12_TURTLE_SYNTAX: &[&str] = &[ - // triple terms as values (`<<( s p o )>>` anywhere but the object of - // `rdf:reifies`, and nested triple terms) are deferred: the graph IR has - // no triple-term Term, so ingest rejects them with the specific error (5) - "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-turtle/syntax#turtle12-3", - "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-turtle/syntax#turtle12-7", - "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-turtle/syntax#turtle12-8", + // nested triple terms (`<<( s p <<( … )>> )>>`) are deferred; ingest + // rejects them with the specific error (2) "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-turtle/syntax#nt-ttl12-3", "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-turtle/syntax#nt-ttl12-nested-1", ]; From c701026f688d2c0b6266dafaf9a4fcc0f58cdb1a Mon Sep 17 00:00:00 2001 From: bplatz Date: Sat, 3 Oct 2026 13:53:55 -0400 Subject: [PATCH 60/92] feat(sparql): CONSTRUCT templates take reified triples A reified triple `<< s p o ~ r? >>` in a template subject or object stands for its reifier and writes its reification of `s p o` without asserting it; with no `~ r` the reifier is a fresh blank node per solution. In the CONSTRUCT WHERE shorthand the template is the pattern, so an anonymous `{| |}` (or a bare `~`) there is a template blank node: the result names a fresh reifier rather than the one the pattern matched. A reified-only template triple whose predicate binds `rdf:reifies` is the triple being reified, so the `?r rdf:reifies ?t` output shortcut now applies to asserted template triples only. W3C: construct-1 and construct-5 pass. construct-3 still fails on its nested triple term. --- docs/query/construct.md | 18 +++++- docs/reference/compatibility.md | 3 +- fluree-db-api/src/format/construct.rs | 4 +- fluree-db-api/tests/it_triple_term_links.rs | 53 +++++++++++++++ fluree-db-sparql/src/lower/annotation.rs | 22 ++++--- fluree-db-sparql/src/lower/construct.rs | 71 +++++++++++++++++++-- fluree-db-sparql/src/lower/select.rs | 2 +- testsuite-sparql/tests/registers/mod.rs | 9 +-- 8 files changed, 153 insertions(+), 29 deletions(-) diff --git a/docs/query/construct.md b/docs/query/construct.md index d6f4ec64eb..93182e8e71 100644 --- a/docs/query/construct.md +++ b/docs/query/construct.md @@ -204,6 +204,22 @@ A `{| ... |}` block without a reifier, or a blank-node reifier (`~ _:r`), mints node for each solution, like `[ ]`. The block's properties become ordinary triples about the reifier. +A reified triple, `<< s p o ~ r >>`, in a template subject or object stands for its reifier +and writes `r`'s reification of `s p o` without asserting it; without `~ r` the reifier is a +fresh blank node per solution: + +```sparql +PREFIX ex: + +# Record each claim without asserting it +CONSTRUCT { << ?s ex:worksFor ?o >> ex:source ex:hrExport } +WHERE { ?s ex:claimsEmployer ?o } +``` + +In the `CONSTRUCT WHERE` shorthand the template is the pattern, and an anonymous `{| ... |}` +(or a bare `~`) in it is a blank node there too: the result names a fresh reifier, not the +one the pattern matched. + In a JSON-LD query, put `@annotation` on the object, as when writing an annotation. An `@annotation` without an `@id` mints a fresh reifier per solution: @@ -367,8 +383,6 @@ details of each. 1.2 defines it; the annotation tail `s p o ~ ?r` writes both. `?r rdf:reifies ?t`, with `?t` bound to a triple term, writes the same as the first. A JSON-LD result writes a reification whose triple it does not carry as the reifier's `@reifies`. - Under any other predicate, a bound triple term is written as a literal holding its N-Triples - text: a result graph holds triple terms only as reifications. - A SPARQL datalog rule whose head (the template) annotates an edge or writes into a named graph is rejected: rules infer default-graph triples only. diff --git a/docs/reference/compatibility.md b/docs/reference/compatibility.md index 089fbc46d0..a459fa0003 100644 --- a/docs/reference/compatibility.md +++ b/docs/reference/compatibility.md @@ -187,7 +187,8 @@ Supported query and update annotation syntax: - Named reifiers: `?s ?p ?o ~ ?r {| ... |}` (IRI, blank-node, or variable reifier) - `rdf:reifies` form with `<<( s p o )>>` triple terms - Annotations in `INSERT DATA` / `DELETE DATA` -- Annotations in `CONSTRUCT` templates (`~ ?r`, `{| ... |}`, and `?r rdf:reifies <<( s p o )>>`), +- Annotations in `CONSTRUCT` templates (`~ ?r`, `{| ... |}`, `<< s p o >>` reified triples and + `?r rdf:reifies <<( s p o )>>`), written by every result format Also supported: triple terms as values in patterns, `INSERT DATA` / `DELETE DATA` / diff --git a/fluree-db-api/src/format/construct.rs b/fluree-db-api/src/format/construct.rs index 007f00297f..4ec6de86ce 100644 --- a/fluree-db-api/src/format/construct.rs +++ b/fluree-db-api/src/format/construct.rs @@ -168,8 +168,10 @@ pub(super) fn instantiate_construct_graph( // `?r rdf:reifies ?t` with a triple term bound to ?t (or a // constant one) writes what `?r rdf:reifies <<( s p o )>>` // does: ?r's reification of the term's triple, which it does - // not assert. + // not assert. A reified-only pattern is the triple a reifier + // reifies, whatever its predicate. let components = match &slots[2] { + _ if !*asserted => None, Slot::Var(v) => batch .get(row, *v) .map(|b| terms.triple_term(b)) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 48b16392ab..917b6f1d3a 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2094,6 +2094,59 @@ async fn triple_terms_render_in_every_result_format() { assert_eq!(parsed(FormatterConfig::typed_json()), typed); } +/// A reified triple in a CONSTRUCT template writes a reification by a fresh +/// blank node per solution, without the triple; the shorthand's anonymous +/// `{| |}` is such a blank node too, not the reifier it matched. +#[tokio::test] +async fn construct_templates_reify_through_fresh_blank_nodes() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger = fluree + .insert_turtle( + support::genesis_ledger(&fluree, "it/triple-term-links:construct-reified"), + CLAIMS, + ) + .await + .expect("insert") + .ledger; + let construct = |sparql: &'static str| { + let fluree = &fluree; + let ledger = &ledger; + async move { + let sparql = format!("PREFIX ex: \n{sparql}"); + support::query_sparql(fluree, ledger, &sparql) + .await + .unwrap_or_else(|e| panic!("{sparql}: {e}")) + .to_construct(&ledger.snapshot) + .expect("JSON-LD")["@graph"] + .clone() + } + }; + let blank = |node: &JsonValue| node["@id"].as_str().is_some_and(|id| id.starts_with("_:")); + + let graph = construct("CONSTRUCT { << ex:x ex:y ex:z >> ex:seen ex:me } WHERE {}").await; + let nodes = graph.as_array().expect("graph"); + assert_eq!( + nodes.len(), + 1, + "the reifier alone, the triple unasserted: {graph}" + ); + assert!(blank(&nodes[0]), "{graph}"); + assert_eq!( + nodes[0]["@reifies"], + json!([{"@id": "ex:x", "ex:y": {"@id": "ex:z"}}]), + "{graph}" + ); + assert_eq!(nodes[0]["ex:seen"], json!([{"@id": "ex:me"}]), "{graph}"); + + let graph = construct("CONSTRUCT WHERE { ex:alice ex:knows ?o {| ex:confidence ?c |} }").await; + let text = graph.to_string(); + assert!(text.contains("0.9"), "the annotation is written: {text}"); + assert!( + !text.contains("ex:claim1"), + "a fresh reifier, not the one the anonymous block matched: {text}" + ); +} + /// `?r rdf:reifies ?t` in a CONSTRUCT template, with a triple term bound to /// ?t, writes what the explicit `<<( s p o )>>` template writes: the /// reification, without asserting its triple. Under any other predicate it diff --git a/fluree-db-sparql/src/lower/annotation.rs b/fluree-db-sparql/src/lower/annotation.rs index a8faf5ba29..b84278d79e 100644 --- a/fluree-db-sparql/src/lower/annotation.rs +++ b/fluree-db-sparql/src/lower/annotation.rs @@ -32,11 +32,15 @@ use std::collections::HashMap; use super::path::PathObject; use super::{LoweringContext, Result}; -/// Prefix used for registry names of synthetic variables that must -/// stay invisible to `SELECT *` and unmatchable by user input. `#` -/// is comment-start in SPARQL, so no user variable can lex with this -/// prefix. -pub(super) const INTERNAL_VAR_PREFIX: &str = "#"; +/// Registry-name prefix of the variable an anonymous reifier (`{| |}`, a +/// bare `~`) lowers to. `#` is comment-start in SPARQL, so no user variable +/// can lex with it, and `SELECT *` hides it. +pub(super) const ANONYMOUS_REIFIER_PREFIX: &str = "?#__ann_"; + +/// The variable an anonymous reifier lowers to. +pub(super) fn is_anonymous_reifier(name: &str) -> bool { + name.starts_with(ANONYMOUS_REIFIER_PREFIX) +} /// Per-BGP memo of already-desugared reified-triple occurrences, keyed /// by source span. A quoted triple shared across several triple @@ -170,11 +174,9 @@ impl LoweringContext<'_, E> { // as a comment-start outside string literals, so no user // variable can ever lex with this name. `lower_select_clause` // filters these out of `SELECT *` expansion. - let var_id = self.vars.get_or_insert(&format!( - "?{}__ann_{}", - INTERNAL_VAR_PREFIX, - self.vars.len() - )); + let var_id = self + .vars + .get_or_insert(&format!("{ANONYMOUS_REIFIER_PREFIX}{}", self.vars.len())); Ok(Ref::Var(var_id)) } } diff --git a/fluree-db-sparql/src/lower/construct.rs b/fluree-db-sparql/src/lower/construct.rs index 37c6ad5abc..652f2768c6 100644 --- a/fluree-db-sparql/src/lower/construct.rs +++ b/fluree-db-sparql/src/lower/construct.rs @@ -7,9 +7,10 @@ use std::sync::Arc; use crate::ast::annotation::{AnnotationVerb, ReifierId}; use crate::ast::query::{ConstructQuery, ConstructTemplate}; +use crate::ast::term::QuotedTriple; use crate::ast::{GraphName, SubjectTerm, Term}; -use fluree_db_query::ir::triple::{Ref, TriplePattern}; +use fluree_db_query::ir::triple::{Ref, Term as IrTerm, TriplePattern}; use fluree_db_query::ir::{ ConstructTemplate as QueryConstructTemplate, Pattern, Query, QueryOutput, }; @@ -38,12 +39,15 @@ impl LoweringContext<'_, E> { // constants, so this prefix uniquely marks template blank nodes. The // WHERE clause never binds them, so the output path mints a fresh blank // node per solution for each (see `ConstructTemplate::bnode_vars`). + // + // The shorthand's template is its WHERE clause, where an anonymous + // reifier (`{| |}`, a bare `~`) is a blank node too. construct_template.bnode_vars = construct_template .var_iter() .filter(|&v| { - self.vars - .try_name(v) - .is_some_and(|name| name.starts_with("_:")) + self.vars.try_name(v).is_some_and(|name| { + name.starts_with("_:") || super::annotation::is_anonymous_reifier(name) + }) }) .collect(); @@ -101,7 +105,12 @@ impl LoweringContext<'_, E> { Some(GraphName::Var(v)) => Some(self.lower_var_ref(v)), None => None, }; - let s = self.lower_subject(&tp.subject)?; + let s = match &tp.subject { + SubjectTerm::QuotedTriple(qt) => { + self.construct_reified_triple(qt, &graph, &mut out)? + } + other => self.lower_subject(other)?, + }; let p = self.lower_predicate(&tp.predicate)?; // `?r rdf:reifies <<( s p o )>>`: the triple term is the reified @@ -136,7 +145,7 @@ impl LoweringContext<'_, E> { // is absent, so `"2024-01-01"^^xsd:date` in a rule head stored // `xsd:string`: `DATATYPE()` said string and `YEAR()` was unbound, // while the identical head written in JSON-LD stored a real date. - let (o, dtc) = self.lower_object_with_constraint(&tp.object)?; + let (o, dtc) = self.construct_object(&tp.object, &graph, &mut out)?; let edge = out.push_pattern(TriplePattern { s, p, o, dtc }, graph.clone()); let Some(annotation) = &tp.annotation else { continue; @@ -160,7 +169,7 @@ impl LoweringContext<'_, E> { )); }; let p = self.lower_predicate(pred)?; - let (o, dtc) = self.lower_object_with_constraint(&entry.object)?; + let (o, dtc) = self.construct_object(&entry.object, &graph, &mut out)?; out.push_pattern( TriplePattern { s: reifier.clone(), @@ -176,6 +185,54 @@ impl LoweringContext<'_, E> { Ok(out) } + /// A template's reified triple `<< s p o ~ r? >>`: its reifier (a fresh + /// blank node per solution when it names none), which reifies `s p o` + /// without asserting it. + fn construct_reified_triple( + &mut self, + qt: &QuotedTriple, + graph: &Option, + out: &mut QueryConstructTemplate, + ) -> Result { + let reifier = match qt.reifier.as_ref().and_then(|r| r.id.as_ref()) { + Some(ReifierId::BlankNode(b)) => { + self.lower_subject(&SubjectTerm::BlankNode(b.clone()))? + } + None => Ref::Var(self.fresh_blank_node_var()), + other => self.lower_reifier_id(other)?, + }; + let s = match &*qt.subject { + SubjectTerm::QuotedTriple(inner) => self.construct_reified_triple(inner, graph, out)?, + other => self.lower_subject(other)?, + }; + let p = self.lower_predicate(&qt.predicate)?; + let (o, dtc) = self.construct_object(&qt.object, graph, out)?; + let triple = out.push_reified_pattern(TriplePattern { s, p, o, dtc }, graph.clone()); + out.push_reification(triple, reifier.clone()); + Ok(reifier) + } + + /// A template object, with a reified triple standing for its reifier. + fn construct_object( + &mut self, + object: &Term, + graph: &Option, + out: &mut QueryConstructTemplate, + ) -> Result<(IrTerm, Option)> { + match object { + Term::QuotedTriple(qt) => Ok(( + IrTerm::from(self.construct_reified_triple(qt, graph, out)?), + None, + )), + Term::TripleTerm(tt) => Err(LowerError::not_implemented( + "a triple term in a CONSTRUCT template is only supported as the object of \ + rdf:reifies", + tt.span, + )), + other => self.lower_object_with_constraint(other), + } + } + /// Extract the CONSTRUCT WHERE shorthand's template from the lowered WHERE /// patterns: every triple, and each edge annotation's link and body. fn extract_template_from_patterns(&self, patterns: &[Pattern]) -> QueryConstructTemplate { diff --git a/fluree-db-sparql/src/lower/select.rs b/fluree-db-sparql/src/lower/select.rs index 501ce16783..7d86f5a7f6 100644 --- a/fluree-db-sparql/src/lower/select.rs +++ b/fluree-db-sparql/src/lower/select.rs @@ -93,7 +93,7 @@ impl LoweringContext<'_, E> { /// must range over the same variables. Three categories are hidden: /// - `?__*` — planner / aggregate / property-path synthetics. /// - `?#*` — annotation-reifier synthetics - /// (see `annotation::INTERNAL_VAR_PREFIX`). + /// (see `annotation::ANONYMOUS_REIFIER_PREFIX`). /// - `_:*` — SPARQL blank-node variables. Per SPARQL §4.1.4 these are /// non-distinguished and not in SELECT scope, so they don't appear in /// `SELECT *` results. Hiding them here also covers blank-node-labelled diff --git a/testsuite-sparql/tests/registers/mod.rs b/testsuite-sparql/tests/registers/mod.rs index e2803bae56..9d26068046 100644 --- a/testsuite-sparql/tests/registers/mod.rs +++ b/testsuite-sparql/tests/registers/mod.rs @@ -406,14 +406,9 @@ pub const SPARQL12_EVAL_TRIPLE_TERMS: &[&str] = &[ "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#update-2", // SRX results: the harness reads no `` result term (1) "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#results-reifiedtriples-1x", - // SPARQL lowering: "RDF-star quoted triples in this position lowering is - // not yet implemented" (CONSTRUCT templates) (2) - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#construct-1", + // results not isomorphic: the template reifies `_:r rdf:reifies <<( … )>>`, + // a nested triple term the result drops (1) "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#construct-3", - // results not isomorphic: CONSTRUCT WHERE reuses the matched reifier for - // an anonymous `{| |}`, where the template mints a fresh blank node per - // solution (1) - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#construct-5", // SPARQL lowering: triple-term values in VALUES data not implemented (1) "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#expr-2", // update-3: the `{| |}` INSERT DATA executes, but the expected post-update From 00f9749bc5b7ff65a622f5b0df56e4dad1ba945a Mon Sep 17 00:00:00 2001 From: bplatz Date: Sat, 3 Oct 2026 14:45:23 -0400 Subject: [PATCH 61/92] feat: nested triple terms A triple term's object may itself be a triple term, as RDF 1.2 allows: `ex:doc ex:mentions <<( ex:alice ex:says <<( ex:s ex:p ex:o )>> )>>`. Every write surface takes one (Turtle, TriG, N-Triples, N-Quads, SPARQL UPDATE, JSON-LD), a triple whose object is a triple term can be annotated, and queries match through the nesting in SPARQL and JSON-LD. - storage: a term's key names its nested term's handle, so the nested term interns first. Novelty registers it first; bulk import spools it first; rebuild and incremental intern a chunk's term table one nesting level per batch (`intern_chunk_terms`). A term whose nested term has no handle keeps no provisional handle either and stays materialized. - query: a nested term lowers to its own term variable and components relation (`lower_term_object`). The components relation now offers every novelty term, of any predicate and nested ones, not only links': when it ran before the pattern binding its term it missed novelty-only term values. - `=` on triple terms is value equality of their components (SPARQL 1.2); triple terms are accepted as VALUES data. - CONSTRUCT and export write nested terms; JSON-LD `@reifies` and `@annotation` take a triple-term object. - An INSERT ... WHERE binding a nested term (e.g. through TRIPLE()) committed it, and indexing then failed on it; it now indexes. W3C harness: reads triple terms in SPARQL JSON and XML results (both sides dropped them, so a triple-term binding could not be compared) and in JSON-LD, and compares a reification as its `rdf:reifies` triple. Now passing: results-tripleterms-1j/1x, results-reifiedtriples-1x, basic-8/9, pattern-10/11, op-1/2, order-2, expr-1/2, construct-3, update-3, and the RDF 1.2 Turtle nested-term syntax and eval tests. --- docs/concepts/edge-annotations.md | 13 +- docs/guides/cookbook-edge-annotations.md | 2 +- docs/query/sparql.md | 2 +- docs/reference/compatibility.md | 14 +- docs/transactions/insert.md | 1 - docs/transactions/turtle.md | 2 +- fluree-db-api/src/export.rs | 4 +- fluree-db-api/src/format/construct.rs | 7 +- fluree-db-api/tests/it_triple_term_links.rs | 356 ++++++++++++++++++ .../src/dict_novelty_safe.rs | 21 +- fluree-db-core/src/dict_novelty.rs | 64 +++- fluree-db-indexer/src/build/rebuild.rs | 25 +- .../run_index/build/incremental_resolve.rs | 71 ++-- .../src/run_index/resolve/resolver.rs | 58 ++- fluree-db-query/src/binary_scan.rs | 5 + fluree-db-query/src/eval/compare.rs | 45 +++ fluree-db-query/src/ir.rs | 4 +- fluree-db-query/src/ir/term_components.rs | 56 ++- fluree-db-query/src/parse/ast.rs | 40 +- fluree-db-query/src/parse/lower.rs | 40 +- fluree-db-query/src/parse/mod.rs | 8 +- fluree-db-query/src/parse/node_map.rs | 21 +- fluree-db-query/src/term_components.rs | 57 +-- fluree-db-sparql/src/lower/annotation.rs | 9 + fluree-db-sparql/src/lower/rdf_star.rs | 12 +- fluree-db-sparql/src/lower/term.rs | 54 ++- fluree-db-sparql/src/parse/query/term.rs | 23 +- fluree-db-sparql/src/parse/query/tests.rs | 2 +- fluree-db-transact/src/flake_sink.rs | 6 +- fluree-db-transact/src/generate/flakes.rs | 12 - fluree-db-transact/src/import_sink.rs | 55 ++- fluree-db-transact/src/lower_sparql_update.rs | 49 +-- .../src/parse/edge_annotations.rs | 19 + fluree-db-transact/src/parse/jsonld.rs | 3 - fluree-db-transact/src/parse/trig_meta.rs | 40 +- fluree-graph-json-ld/src/adapter.rs | 10 +- fluree-graph-turtle/src/parser.rs | 29 +- testsuite-sparql/src/rdf_handlers.rs | 22 +- testsuite-sparql/src/result_format.rs | 135 ++++--- testsuite-sparql/tests/registers/mod.rs | 39 +- 40 files changed, 1038 insertions(+), 397 deletions(-) diff --git a/docs/concepts/edge-annotations.md b/docs/concepts/edge-annotations.md index 1ae8b9c13a..32b5385ff5 100644 --- a/docs/concepts/edge-annotations.md +++ b/docs/concepts/edge-annotations.md @@ -27,7 +27,7 @@ If a fact is naturally about a *node* (Alice's birthdate, Acme's industry), put |---|---|---| | **JSON-LD insert / upsert / update** | `@annotation` (or alias `@edge`) on a value object | Most ergonomic. Covers literal-valued edges (with explicit `@type` / `@language`), parallel annotations, named reifiers, and **named-graph edges** (an annotation on an edge inside a named graph is written into that same graph, keeping the edge's graph identity). A node's `@reifies` (`{"@id": s, p: o}`, or an array of them) makes the node a reifier of that triple **without asserting it**, as `r rdf:reifies <<( s p o )>>` does. | | **SPARQL 1.2 UPDATE** | `INSERT DATA { :s :p :o {\| ... \|} }`, `~ `, optional `INSERT { } WHERE { }` templates | Use this when integrating with SPARQL pipelines or when porting from RDF 1.2 / SPARQL-star. Works inside `GRAPH { }` blocks and under `WITH `; `<< :s :p :o ~ :r >>` reifies without asserting. See [SPARQL 1.2 surface](#sparql-12--rdf-12-surface) below for the per-operation rules. | -| **Turtle / N-Triples / TriG / N-Quads ingest** (`insert`, `upsert`, bulk `import`, graph sync (`fluree sync`, `/sync`), memory import; TriG via `insert` / `upsert` / `import` / `fluree sync` / `/sync`) | RDF 1.2 forms: the annotation syntax `:s :p :o ~ {\| ... \|}`, `<< :s :p :o ~ :r >>` in subject or object position, and the canonical `:r rdf:reifies <<( :s :p :o )>>` (the only star spelling N-Triples and N-Quads have) — in the default graph and inside TriG `GRAPH { }` blocks alike | Same stored link as `@annotation`. An annotation inside a `GRAPH { }` block, or on an N-Quads statement with a graph label, is written into that graph with the edge's graph identity, exactly as JSON-LD `@graph` + `@annotation` does. As in RDF 1.2, only the annotation syntax asserts the triple: `<< s p o >>` and `rdf:reifies <<( s p o )>>` reify it without asserting it, so N-Triples and N-Quads state an annotated triple as the triple plus its reifier's link. Rejected with a specific error: `<<( ... )>>` as a subject, nested triple terms, star constructs inside an annotation body, annotations on collections, and annotations in a TriG `<#txn-meta>` block. Paths that convert to JSON-LD first (`upsert`, `graph sync`, memory import) also reject an annotation on an `rdf:type` edge. See [Turtle ingest](../transactions/turtle.md#edge-annotations-rdf-12--turtle-star). | +| **Turtle / N-Triples / TriG / N-Quads ingest** (`insert`, `upsert`, bulk `import`, graph sync (`fluree sync`, `/sync`), memory import; TriG via `insert` / `upsert` / `import` / `fluree sync` / `/sync`) | RDF 1.2 forms: the annotation syntax `:s :p :o ~ {\| ... \|}`, `<< :s :p :o ~ :r >>` in subject or object position, and the canonical `:r rdf:reifies <<( :s :p :o )>>` (the only star spelling N-Triples and N-Quads have) — in the default graph and inside TriG `GRAPH { }` blocks alike | Same stored link as `@annotation`. An annotation inside a `GRAPH { }` block, or on an N-Quads statement with a graph label, is written into that graph with the edge's graph identity, exactly as JSON-LD `@graph` + `@annotation` does. As in RDF 1.2, only the annotation syntax asserts the triple: `<< s p o >>` and `rdf:reifies <<( s p o )>>` reify it without asserting it, so N-Triples and N-Quads state an annotated triple as the triple plus its reifier's link. Rejected with a specific error: `<<( ... )>>` as a subject, star constructs inside an annotation body, annotations on collections, and annotations in a TriG `<#txn-meta>` block. Paths that convert to JSON-LD first (`upsert`, `graph sync`, memory import) also reject an annotation on an `rdf:type` edge. See [Turtle ingest](../transactions/turtle.md#edge-annotations-rdf-12--turtle-star). | Mint annotations through `@annotation` / `@edge` (JSON-LD) or the RDF 1.2 forms (`~`, `{| |}`, `<< >>`, `rdf:reifies <<( )>>`) in SPARQL UPDATE and Turtle. The [`f:reifies*` predicates](../reference/vocabulary.md#edge-annotation-predicates-reserved) earlier releases stored annotations under are reserved, and the write surfaces reject them; bulk import reads them, as an export written before links carries them, and stores each annotation's link instead. @@ -141,7 +141,7 @@ A few rules that keep the annotation's identity in sync with the base flake: - **Language-tagged literals are language-pinned.** Two annotations on `"chat"@fr` and `"chat"@en` are independent; selector-form retracts and hydration both match on language. - **Hydration promotes annotated literals to value-object form.** A subject expansion (`select: {"?s": ["*"]}`) renders unannotated `ex:name "Alice"` as the scalar `"Alice"`, but renders the annotated form as `{"@value": "Alice", "@annotation": {...}}` so the annotation has somewhere to attach. -The deferred shapes from "Current limits" below (list occurrences, nested triple terms) still apply on the literal path. +The deferred shape from "Current limits" below (list occurrences) still applies on the literal path. ### Querying inline: edge first, metadata second @@ -427,7 +427,7 @@ Notes: - An `annotationBlock` without a preceding `~` mints a fresh anonymous reifier. - A bare `~` (no identifier) is equivalent to `~` + a fresh blank node — useful when you want a reifier variable bound in WHERE but don't care about its IRI. -- `tripleTerm` (the parenthesized `<<( s p o )>>` form) is a value in object position: a reifier's triple under `rdf:reifies`, a stored value under any other predicate (see [Triple terms as values](#triple-terms-as-values)). As a subject, or nested in another triple term, it errors at parse time. +- `tripleTerm` (the parenthesized `<<( s p o )>>` form) is a value in object position: a reifier's triple under `rdf:reifies`, a stored value under any other predicate, or another triple term's object (see [Triple terms as values](#triple-terms-as-values)). As a subject it errors at parse time. - Property-path triples cannot carry an annotation tail. `?s ex:p1/ex:p2 ?o {| ... |}` is rejected — write a simple-predicate triple instead. ### SPARQL UPDATE rules by operation @@ -528,13 +528,14 @@ A triple term is also an ordinary value under any predicate. `ex:doc ex:mentions - SPARQL UPDATE: `INSERT DATA`, `DELETE DATA`, `DELETE WHERE` and templates. - JSON-LD: the triple's node as the value's `@id`, `"ex:mentions": {"@id": {"@id": "ex:s", "ex:p": {"@id": "ex:o"}}}`. -A query matches one as a constant (`?d ex:mentions <<( ex:s ex:p ex:o )>>`) or by its components (`?d ex:mentions <<( ex:s ?p ?o )>>`; in JSON-LD, `{"@id": {"@id": "ex:s", "ex:p": "?o"}}`), and `SUBJECT`, `PREDICATE` and `OBJECT` take a bound one apart (JSON-LD names them `subject`, `predicate` and `object`; see [JSON-LD query](../query/jsonld-query.md#triple-term-functions)). Results, CONSTRUCT output and exports write it back in the same forms (see [Output formats](../query/output-formats.md#triple-terms)), so it round-trips. +A query matches one as a constant (`?d ex:mentions <<( ex:s ex:p ex:o )>>`, or a `VALUES` row) or by its components (`?d ex:mentions <<( ex:s ?p ?o )>>`; in JSON-LD, `{"@id": {"@id": "ex:s", "ex:p": "?o"}}`), and `SUBJECT`, `PREDICATE` and `OBJECT` take a bound one apart (JSON-LD names them `subject`, `predicate` and `object`; see [JSON-LD query](../query/jsonld-query.md#triple-term-functions)). Results, CONSTRUCT output and exports write it back in the same forms (see [Output formats](../query/output-formats.md#triple-terms)), so it round-trips. + +A triple term's object may itself be a triple term — `<<( ex:alice ex:says <<( ex:s ex:p ex:o )>> )>>` — on every surface above, and a triple whose object is a triple term can be annotated like any other. `sameTerm` compares two triple terms as terms; `=` compares them as values, so `<<( :a :b 123 )>> = <<( :a :b 123.0 )>>` holds while `sameTerm` does not. ### Deferred SPARQL shapes (rejected at parse time) These produce a clear error with a span pointing at the offending construct: -- **Nested triple terms.** `<<( :s :p <<( :a :b :c )>> )>>` is rejected. - **Annotation on a property-path triple.** `?s ex:p1/ex:p2 ?o {| ... |}` is rejected — the grammar only attaches annotations to simple-predicate triples. - **Property paths and nested triple terms in a `CONSTRUCT` template's annotation.** A template annotation block (`{| ... |}`) takes simple predicates only, and a template triple term (`?r rdf:reifies <<( ... )>>`) cannot nest. @@ -563,8 +564,6 @@ The bare-quoted-triple form combined with an annotation tail (`<< :s :p :o >> :p Today's surface covers the common LPG / RDF-star use cases. The following are not yet supported and produce a clear validation error rather than silent partial behavior: - **Annotations on list-occurrence triples.** `@list` membership is in scope as a future extension; the on-disk format already reserves space for it. Today, annotating a list element is rejected at parse time. -- **Nested triple terms.** A triple term whose object is another triple term (`<<( :s :p <<( :a :b :c )>> )>>`), and a reifier of a triple whose object is one, are rejected on every write surface. -- **Triple-term constants in `VALUES`.** `VALUES ?t { <<( :s :p :o )>> }` is rejected; bind one with `BIND(<<( :s :p :o )>> AS ?t)` instead. - **Triple-term constants in a `CONSTRUCT` template** other than the object of `rdf:reifies`. A template variable bound to a term (`CONSTRUCT { ?d :mentions ?t }`) writes it. The mandated SPARQL 1.2 `VERSION "1.2"` prologue declaration is **accepted** (lex-and-skipped): the RDF 1.2 surface runs ungated, so a conformant 1.2 client that emits the declaration parses normally. diff --git a/docs/guides/cookbook-edge-annotations.md b/docs/guides/cookbook-edge-annotations.md index 80b8699ab5..9b9eedd952 100644 --- a/docs/guides/cookbook-edge-annotations.md +++ b/docs/guides/cookbook-edge-annotations.md @@ -315,7 +315,7 @@ The edge and every other claim on it stay live. The named claim's body (`ex:conf - **Deleting a claim with `DELETE DATA { … ~ :claim {| … |} }` deletes the edge**, so the annotation syntax stops matching every other claim on it. Retract one claim with the JSON-LD by-id form (see [above](#retract-one-claim-and-keep-the-edge)). - **Don't write `f:reifies*` predicates by hand.** They're reserved and rejected on every write surface; they're also hidden from `?p` scans and `select: "*"`. Use `@annotation` / the annotation tail. (See [Vocabulary](../reference/vocabulary.md#edge-annotation-predicates-reserved).) - **Empty `@annotation: {}`** is a no-op in RDF mode (no subject minted); in LPG mode it mints a property-less relationship with identity. -- **Not yet supported** (all reject cleanly, no silent partial results): annotations on `@list` elements and nested triple terms. See [Current limits](../concepts/edge-annotations.md#current-limits). +- **Not yet supported** (all reject cleanly, no silent partial results): annotations on `@list` elements. See [Current limits](../concepts/edge-annotations.md#current-limits). ## See also diff --git a/docs/query/sparql.md b/docs/query/sparql.md index c30fce98b6..fbd3289ec9 100644 --- a/docs/query/sparql.md +++ b/docs/query/sparql.md @@ -1095,7 +1095,7 @@ Annotation tails are supported in `INSERT DATA`, `DELETE DATA`, and `INSERT { } ### Boundaries (rejected at parse / lowering time) - **Simple-predicate triples only.** `?s ex:p1/ex:p2 ?o {| ... |}` (property-path) is rejected. -- **Triple terms in object position only**, not nested, and not as `VALUES` data. +- **Triple terms in object position only** (a triple term's object may be another one). - **No reserved predicates by hand.** The [system predicates](../reference/vocabulary.md#edge-annotation-predicates-reserved) that back annotations are rejected on every UPDATE clause; mint annotations only through the `~` / `{| |}` surface. - **`CONSTRUCT` template annotation blocks take simple predicates only**, and a template triple term cannot nest. diff --git a/docs/reference/compatibility.md b/docs/reference/compatibility.md index a459fa0003..2254dedb0d 100644 --- a/docs/reference/compatibility.md +++ b/docs/reference/compatibility.md @@ -43,7 +43,7 @@ The vendored W3C RDF 1.1 and RDF 1.2 Turtle suites run in CI (`testsuite-sparql/tests/w3c_rdf.rs`), with known gaps in the skip register. Not yet supported: -- Nested triple terms (`<<( s p <<( ... )>> )>>`) +- Annotation-of-annotation (a `{| ... |}` or `<< ... >>` inside an annotation body) See [Edge annotations](../concepts/edge-annotations.md). @@ -191,13 +191,13 @@ Supported query and update annotation syntax: `?r rdf:reifies <<( s p o )>>`), written by every result format -Also supported: triple terms as values in patterns, `INSERT DATA` / `DELETE DATA` / -`DELETE WHERE` and templates, and the triple-term functions `TRIPLE()`, `SUBJECT()`, -`PREDICATE()`, `OBJECT()` and `isTRIPLE()`. +Also supported: triple terms as values, nested ones included, in patterns, `VALUES`, +`INSERT DATA` / `DELETE DATA` / `DELETE WHERE` and templates; value equality (`=`) on +triple terms; and the triple-term functions `TRIPLE()`, `SUBJECT()`, `PREDICATE()`, +`OBJECT()` and `isTRIPLE()`. Not yet supported: -- Nested triple terms, and triple-term constants in `VALUES` data or (other than the object - of `rdf:reifies`) in a `CONSTRUCT` template +- Triple-term constants in a `CONSTRUCT` template other than as the object of `rdf:reifies` - Nested annotations (annotation-of-annotation) **Specification:** https://www.w3.org/TR/sparql12-query/ @@ -485,7 +485,7 @@ Export Fluree data to: - SPARQL 1.1 Federation: remote `SERVICE` endpoints (local-ledger `SERVICE` is supported) - Remote `LOAD` in SPARQL UPDATE - GeoSPARQL: remaining OGC functions (only `geof:distance` is implemented today) -- RDF 1.2 / SPARQL 1.2: nested triple terms +- RDF 1.2 / SPARQL 1.2: annotation-of-annotation; a `GRAPH ?g` name written as an INSERT template object **Storage:** - Additional cloud providers (GCP, Azure) diff --git a/docs/transactions/insert.md b/docs/transactions/insert.md index 90612dde83..3b5520807f 100644 --- a/docs/transactions/insert.md +++ b/docs/transactions/insert.md @@ -386,7 +386,6 @@ Inline `@annotation` queries return one row per occurrence. **Deferred shapes** error with explicit messages: - Annotations on list-occurrence triples (`@list` membership). -- Nested triple terms (a triple term's object that is itself `{"@id": {...}}`). - More than one predicate-object pair in one `@reifies` block (use an array of blocks to reify several triples). - Annotation-of-annotation (nested `@annotation` inside an annotation body). diff --git a/docs/transactions/turtle.md b/docs/transactions/turtle.md index 5a784d8d4a..77529f078e 100644 --- a/docs/transactions/turtle.md +++ b/docs/transactions/turtle.md @@ -483,7 +483,7 @@ ex:emp1 rdf:reifies <<( ex:alice ex:worksFor ex:acme )>> . Two rules to know: - **Only the annotation syntax asserts the triple.** As RDF 1.2 defines them, `s p o ~ r` and `s p o {| … |}` put `s p o` in the graph and attach the reifier to it, while `<< s p o >>` and `r rdf:reifies <<( s p o )>>` attach the reifier without asserting `s p o`. The reifier's own triples (the annotation body) are ordinary RDF about the reifier. Each anonymous `<< s p o >>` / `{| |}` occurrence mints a fresh reifier — two textual occurrences are two annotations. -- **`<<( ... )>>` is a value.** Under `rdf:reifies` it is a reifier's triple; under any other predicate (`ex:doc ex:mentions <<( ... )>>`) it is stored as a value, without asserting or reifying its triple (see [Triple terms as values](../concepts/edge-annotations.md#triple-terms-as-values)). As a subject, nested inside another triple term, or inside an annotation body, it is rejected with a specific error rather than silently dropped. +- **`<<( ... )>>` is a value.** Under `rdf:reifies` it is a reifier's triple; under any other predicate (`ex:doc ex:mentions <<( ... )>>`) it is stored as a value, without asserting or reifying its triple (see [Triple terms as values](../concepts/edge-annotations.md#triple-terms-as-values)). A triple term's object may be another triple term. As a subject, or inside an annotation body, it is rejected with a specific error rather than silently dropped. TriG and N-Quads accept the same forms inside `GRAPH { }` blocks (and on N-Quads statements with a graph label). The annotation is written into that graph and carries the edge's graph identity, exactly as JSON-LD `@graph` + `@annotation` does: diff --git a/fluree-db-api/src/export.rs b/fluree-db-api/src/export.rs index 2a83ff4741..f957ca33c1 100644 --- a/fluree-db-api/src/export.rs +++ b/fluree-db-api/src/export.rs @@ -1122,8 +1122,8 @@ fn reifies_jsonld( } /// A triple term as the JSON-LD node naming its triple, `{"@id": s, p: o}`; -/// as a value it is wrapped as `{"@id": {...}}`. `None` for a nested term, -/// which has no such node. +/// as a value it is wrapped as `{"@id": {...}}`, a nested term included. +/// `None` when a component has no IRI. fn triple_term_jsonld( term: &fluree_db_core::TripleTermValue, store: &BinaryIndexStore, diff --git a/fluree-db-api/src/format/construct.rs b/fluree-db-api/src/format/construct.rs index 4ec6de86ce..f89946b83f 100644 --- a/fluree-db-api/src/format/construct.rs +++ b/fluree-db-api/src/format/construct.rs @@ -417,15 +417,12 @@ impl TermResolver<'_> { self.term_components(term) } - /// A triple term as the IR terms of its subject, predicate and object, - /// when the graph IR can carry its object (any term but a nested one). + /// A triple term as the IR terms of its subject, predicate and object (a + /// nested term object among them). fn term_components( &mut self, term: &fluree_db_core::TripleTermValue, ) -> Result> { - if matches!(term.o, FlakeValue::TripleTerm(_)) { - return Ok(None); - } let [s, p, o] = super::triple_term_components(term); let s = self.binding(&s, Position::Subject)?; let p = self.binding(&p, Position::Predicate)?; diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 917b6f1d3a..97e8facd14 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2906,3 +2906,359 @@ async fn every_write_path_takes_triple_term_values() { assert_eq!(named, strings(&[&["ex:o"]]), "[{label}] in the named graph"); } } + +/// A triple term nested in another, as a value and inside an annotated +/// edge's term, in Turtle. +const NESTED: &str = "@prefix ex: .\n\ + ex:doc ex:mentions <<( ex:alice ex:says <<( ex:s ex:p ex:o )>> )>> .\n\ + ex:alice ex:says <<( ex:s ex:p \"chat\"@fr )>> {| ex:source ex:hr |} .\n"; + +/// [`NESTED`] reads back from `ledger`, with `docs` subjects mentioning its +/// term, through SPARQL and JSON-LD. +async fn assert_nested( + fluree: &fluree_db_api::Fluree, + ledger: &LedgerState, + label: &str, + docs: usize, +) { + let run = |q: &str| run_link_query(fluree, ledger, q.to_string()); + let got = run("SELECT ?s ?p ?o WHERE { ex:doc ex:mentions ?t \ + BIND(OBJECT(?t) AS ?inner) BIND(SUBJECT(?inner) AS ?s) \ + BIND(PREDICATE(?inner) AS ?p) BIND(OBJECT(?inner) AS ?o) }") + .await; + assert_eq!( + got, + strings(&[&["ex:s", "ex:p", "ex:o"]]), + "[{label}] decomposes" + ); + let got = + run("SELECT ?d WHERE { ?d ex:mentions <<( ex:alice ex:says <<( ex:s ex:p ex:o )>> )>> }") + .await; + assert_eq!( + got.len(), + docs, + "[{label}] a constant nested term matches: {got:?}" + ); + let got = run( + "SELECT ?who ?o WHERE { ex:doc ex:mentions <<( ?who ex:says <<( ex:s ex:p ?o )>> )>> }", + ) + .await; + assert_eq!( + got, + strings(&[&["ex:alice", "ex:o"]]), + "[{label}] by components" + ); + let got = + run("SELECT ?src WHERE { ex:alice ex:says <<( ex:s ex:p ?o )>> {| ex:source ?src |} }") + .await; + assert_eq!( + got, + strings(&[&["ex:hr"]]), + "[{label}] an annotated term-valued edge" + ); + let got = run("SELECT ?src WHERE { << ex:alice ex:says ?t ~ ?r >> ex:source ?src }").await; + assert_eq!(got, strings(&[&["ex:hr"]]), "[{label}] its reifier"); + let got = run("SELECT ?r WHERE { ?r rdf:reifies ?t }").await; + assert_eq!(got.len(), 1, "[{label}] one link: {got:?}"); + + let got = support::query_jsonld_formatted( + fluree, + ledger, + &json!({ + "@context": {"ex": "http://example.org/"}, + "select": ["?d", "?o"], + "where": {"@id": "?d", "ex:mentions": {"@id": { + "@id": "ex:alice", + "ex:says": {"@id": {"@id": "ex:s", "ex:p": "?o"}} + }}} + }), + ) + .await + .unwrap_or_else(|e| panic!("[{label}] JSON-LD nested pattern: {e}")); + assert_eq!( + got.as_array().map(Vec::len), + Some(docs), + "[{label}] JSON-LD matches a nested term: {got}" + ); +} + +/// The predicate of the inner term `ex:doc3` mentions. +async fn fresh(fluree: &fluree_db_api::Fluree, ledger: &LedgerState) -> Vec> { + run_link_query( + fluree, + ledger, + "SELECT ?p WHERE { ex:doc3 ex:mentions <<( ex:bob ex:says <<( ex:x ?p ex:y )>> )>> }" + .to_string(), + ) + .await +} + +/// Nested triple terms are values like any other: they store, index, +/// query, write back and round-trip. +#[tokio::test] +async fn nested_triple_terms_are_values() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/triple-term-links:nested"; + let ledger = fluree + .insert_turtle(support::genesis_ledger(&fluree, ledger_id), NESTED) + .await + .expect("insert nested terms") + .ledger; + assert_nested(&fluree, &ledger, "novelty", 1).await; + support::rebuild_and_publish_index(&fluree, ledger_id).await; + let ledger = fluree.ledger(ledger_id).await.expect("load"); + assert_nested(&fluree, &ledger, "rebuild", 1).await; + + // A nested term whose inner term is new to the index. + let ledger = fluree + .insert_turtle( + ledger, + "@prefix ex: .\n\ + ex:doc2 ex:mentions <<( ex:alice ex:says <<( ex:s ex:p ex:o )>> )>> .\n\ + ex:doc3 ex:mentions <<( ex:bob ex:says <<( ex:x ex:fresh ex:y )>> )>> .", + ) + .await + .expect("insert over an index") + .ledger; + assert_nested(&fluree, &ledger, "novelty over an index", 2).await; + assert_eq!( + fresh(&fluree, &ledger).await, + strings(&[&["ex:fresh"]]), + "novelty over an index" + ); + + for (format, file) in [ + (fluree_db_api::export::ExportFormat::Turtle, "nested.ttl"), + (fluree_db_api::export::ExportFormat::JsonLd, "nested.jsonld"), + ] { + let mut buf = Vec::new(); + fluree + .export(ledger_id) + .format(format) + .write_to(&mut buf) + .await + .unwrap_or_else(|e| panic!("export {file}: {e}")); + let exported = String::from_utf8(buf).expect("utf8"); + let (reimported, reloaded) = import( + &[(file, &exported)], + &format!("it/triple-term-links:n-{file}"), + ) + .await; + assert_nested(&reimported, &reloaded, file, 2).await; + assert_eq!( + fresh(&reimported, &reloaded).await, + strings(&[&["ex:fresh"]]), + "[{file}] {exported}" + ); + } + + support::build_and_publish_index(&fluree, ledger_id).await; + let ledger = fluree.ledger(ledger_id).await.expect("load"); + assert_nested(&fluree, &ledger, "incremental", 2).await; + assert_eq!( + fresh(&fluree, &ledger).await, + strings(&[&["ex:fresh"]]), + "incremental" + ); +} + +/// A components relation anchored on a constant subject can run before the +/// pattern binding its term, and then offers novelty's terms of every +/// predicate, nested ones too, not only its links'. +#[tokio::test] +async fn term_components_offer_novelty_value_terms_when_they_lead() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/triple-term-links:tc-novelty"; + let docs: String = (0..300) + .map(|i| format!("ex:d{i} ex:mentions <<( ex:s{i} ex:p ex:o{i} )>> .\n")) + .collect(); + fluree + .insert_turtle( + support::genesis_ledger(&fluree, ledger_id), + &format!("@prefix ex: .\n{docs}"), + ) + .await + .expect("insert"); + support::rebuild_and_publish_index(&fluree, ledger_id).await; + let ledger = fluree + .insert_turtle( + fluree.ledger(ledger_id).await.expect("load"), + "@prefix ex: .\n\ + ex:dx ex:mentions <<( ex:target ex:q ex:z )>> .\n\ + ex:dy ex:mentions <<( ex:who ex:says <<( ex:target ex:r ex:z )>> )>> .", + ) + .await + .expect("insert over an index") + .ledger; + let sparql = "PREFIX ex: \n\ + SELECT ?d ?p WHERE { ?d ex:mentions <<( ex:target ?p ?o )>> }"; + let plan = fluree + .explain_sparql(&support::graphdb_from_ledger(&ledger), sparql) + .await + .expect("explain") + .to_string(); + assert!( + plan.find("TermComponentsOperator") > plan.find("NestedLoopJoinOperator"), + "the components relation leads: {plan}" + ); + let got = run_link_query( + &fluree, + &ledger, + "SELECT ?d ?p WHERE { ?d ex:mentions <<( ex:target ?p ?o )>> }".to_string(), + ) + .await; + assert_eq!(got, strings(&[&["ex:dx", "ex:q"]])); + let got = run_link_query( + &fluree, + &ledger, + "SELECT ?d ?p WHERE { ?d ex:mentions <<( ?w ex:says <<( ex:target ?p ?o )>> )>> }" + .to_string(), + ) + .await; + assert_eq!(got, strings(&[&["ex:dy", "ex:r"]]), "a nested term"); + let reloaded = fluree.ledger(ledger_id).await.expect("reload"); + let got = run_link_query( + &fluree, + &reloaded, + "SELECT ?d ?p WHERE { ?d ex:mentions <<( ?w ex:says <<( ex:target ?p ?o )>> )>> }" + .to_string(), + ) + .await; + assert_eq!( + got, + strings(&[&["ex:dy", "ex:r"]]), + "a nested term, reloaded" + ); +} + +/// Every write surface takes a nested triple term, and an annotation on a +/// triple whose object is one; the delete forms remove one. +#[tokio::test] +async fn every_write_path_takes_nested_triple_terms() { + let fluree = FlureeBuilder::memory().build_memory(); + let fresh_ledger = |id: &str| support::genesis_ledger(&fluree, id); + let nested_jsonld = json!({ + "@context": {"ex": "http://example.org/"}, + "@graph": [ + {"@id": "ex:doc", "ex:mentions": {"@id": { + "@id": "ex:alice", + "ex:says": {"@id": {"@id": "ex:s", "ex:p": {"@id": "ex:o"}}} + }}}, + {"@id": "ex:alice", "ex:says": { + "@id": {"@id": "ex:s", "ex:p": {"@value": "chat", "@language": "fr"}}, + "@annotation": {"ex:source": {"@id": "ex:hr"}} + }} + ] + }); + let ledger = fluree + .insert(fresh_ledger("it/tt-nested:jsonld"), &nested_jsonld) + .await + .expect("JSON-LD insert") + .ledger; + assert_nested(&fluree, &ledger, "JSON-LD insert", 1).await; + let ledger = fluree + .update( + ledger, + &json!({ + "@context": {"ex": "http://example.org/"}, + "delete": {"@id": "ex:doc", "ex:mentions": {"@id": { + "@id": "ex:alice", + "ex:says": {"@id": {"@id": "ex:s", "ex:p": {"@id": "ex:o"}}} + }}} + }), + ) + .await + .expect("JSON-LD delete") + .ledger; + let mentions = |ledger: LedgerState| { + let fluree = &fluree; + async move { + run_link_query( + fluree, + &ledger, + "SELECT ?t WHERE { ?d ex:mentions ?t }".to_string(), + ) + .await + .len() + } + }; + assert_eq!(mentions(ledger).await, 0, "a JSON-LD delete removes it"); + + let ledger = fluree + .upsert_turtle(fresh_ledger("it/tt-nested:upsert"), NESTED) + .await + .expect("Turtle upsert") + .ledger; + assert_nested(&fluree, &ledger, "Turtle upsert", 1).await; + + let ledger_id = "it/tt-nested:sparql"; + fluree.create_ledger(ledger_id).await.expect("create"); + let update = |body: &str| { + let body = format!("PREFIX ex: \n{body}"); + let fluree = &fluree; + async move { + fluree + .graph(ledger_id) + .transact() + .sparql_update(&body) + .commit() + .await + .unwrap_or_else(|e| panic!("{body}: {e}")); + fluree.ledger(ledger_id).await.expect("load") + } + }; + let ledger = update( + "INSERT DATA { ex:doc ex:mentions <<( ex:alice ex:says <<( ex:s ex:p ex:o )>> )>> . \ + ex:alice ex:says <<( ex:s ex:p \"chat\"@fr )>> {| ex:source ex:hr |} }", + ) + .await; + assert_nested(&fluree, &ledger, "SPARQL INSERT DATA", 1).await; + // A term built in WHERE, nested, is stored, indexed and read back. + let ledger = update( + "INSERT { ex:doc2 ex:mentions ?t } WHERE { \ + BIND(TRIPLE(ex:alice, ex:says, TRIPLE(ex:s, ex:p, ex:o)) AS ?t) }", + ) + .await; + assert_nested(&fluree, &ledger, "SPARQL INSERT WHERE", 2).await; + support::rebuild_and_publish_index(&fluree, ledger_id).await; + let ledger = fluree.ledger(ledger_id).await.expect("load"); + assert_nested(&fluree, &ledger, "SPARQL INSERT WHERE, indexed", 2).await; + let ledger = update( + "DELETE DATA { ex:doc ex:mentions <<( ex:alice ex:says <<( ex:s ex:p ex:o )>> )>> }", + ) + .await; + assert_eq!(mentions(ledger).await, 1, "DELETE DATA removes one"); + let ledger = + update("DELETE WHERE { ?d ex:mentions <<( ?w ex:says <<( ex:s ex:p ?o )>> )>> }").await; + assert_eq!( + mentions(ledger).await, + 0, + "DELETE WHERE matches through the nesting" + ); + + let trig = format!( + "{NESTED}ex:g {{ ex:doc ex:mentions <<( ex:alice ex:says <<( ex:s ex:p ex:o )>> )>> . }}\n" + ); + let inserted = fluree + .insert_turtle(fresh_ledger("it/tt-nested:trig"), &trig) + .await + .expect("TriG insert") + .ledger; + assert_nested(&fluree, &inserted, "TriG insert", 1).await; + let (imported, imported_ledger) = + import(&[("nested.trig", &trig)], "it/tt-nested:import-trig").await; + assert_nested(&imported, &imported_ledger, "TriG import", 1).await; + for (fluree, ledger, label) in [ + (&fluree, &inserted, "TriG insert"), + (&imported, &imported_ledger, "TriG import"), + ] { + let named = run_link_query( + fluree, + ledger, + "SELECT ?o WHERE { GRAPH ex:g { ex:doc ex:mentions <<( ex:alice ex:says <<( ex:s ex:p ?o )>> )>> } }" + .to_string(), + ) + .await; + assert_eq!(named, strings(&[&["ex:o"]]), "[{label}] in the named graph"); + } +} diff --git a/fluree-db-binary-index/src/dict_novelty_safe.rs b/fluree-db-binary-index/src/dict_novelty_safe.rs index d500916265..5e7698c79e 100644 --- a/fluree-db-binary-index/src/dict_novelty_safe.rs +++ b/fluree-db-binary-index/src/dict_novelty_safe.rs @@ -91,16 +91,23 @@ pub fn populate_dict_novelty_safe<'a>( // A term's components need ids for its provisional handle's key. // The term itself is registered whether or not the index holds // it: readers try the persisted dictionary first. + // A nested term goes first: the outer term's key names its handle. FlakeValue::TripleTerm(term) => { - subject(dict_novelty, &term.s, flake.t)?; - match &term.o { - FlakeValue::Ref(sid) => subject(dict_novelty, sid, flake.t)?, - FlakeValue::String(s) | FlakeValue::Json(s) => { - string(dict_novelty, s, flake.t)?; + let mut chain = vec![&**term]; + while let FlakeValue::TripleTerm(inner) = &chain[chain.len() - 1].o { + chain.push(inner); + } + for term in chain.into_iter().rev() { + subject(dict_novelty, &term.s, flake.t)?; + match &term.o { + FlakeValue::Ref(sid) => subject(dict_novelty, sid, flake.t)?, + FlakeValue::String(s) | FlakeValue::Json(s) => { + string(dict_novelty, s, flake.t)?; + } + _ => {} } - _ => {} + dict_novelty.terms.assign_or_lookup_at(term, flake.t); } - dict_novelty.terms.assign_or_lookup_at(term, flake.t); } _ => {} } diff --git a/fluree-db-core/src/dict_novelty.rs b/fluree-db-core/src/dict_novelty.rs index f4c8192dae..b95cca825b 100644 --- a/fluree-db-core/src/dict_novelty.rs +++ b/fluree-db-core/src/dict_novelty.rs @@ -254,29 +254,31 @@ impl DictNovelty { FlakeValue::String(s) | FlakeValue::Json(s) => { self.strings.assign_or_lookup_at(s, flake.t); } - FlakeValue::TripleTerm(term) => { - self.subjects - .assign_or_lookup_at(term.s.namespace_code, &term.s.name, flake.t); - match &term.o { - FlakeValue::Ref(sid) => { - self.subjects.assign_or_lookup_at( - sid.namespace_code, - &sid.name, - flake.t, - ); - } - FlakeValue::String(s) | FlakeValue::Json(s) => { - self.strings.assign_or_lookup_at(s, flake.t); - } - _ => {} - } - self.terms.assign_or_lookup_at(term, flake.t); - } + FlakeValue::TripleTerm(term) => self.register_term(term, flake.t), _ => {} } } } + /// Register a term's subject, object and the term itself; a nested term + /// first, since the outer term's key names its handle. + fn register_term(&mut self, term: &TripleTermValue, t: i64) { + self.subjects + .assign_or_lookup_at(term.s.namespace_code, &term.s.name, t); + match &term.o { + FlakeValue::Ref(sid) => { + self.subjects + .assign_or_lookup_at(sid.namespace_code, &sid.name, t); + } + FlakeValue::String(s) | FlakeValue::Json(s) => { + self.strings.assign_or_lookup_at(s, t); + } + FlakeValue::TripleTerm(inner) => self.register_term(inner, t), + _ => {} + } + self.terms.assign_or_lookup_at(term, t); + } + /// Populate the novelty dictionaries from a slice of flakes. pub fn populate_from_flakes(&mut self, flakes: &[Flake]) { self.populate_from_flakes_iter(flakes); @@ -1101,4 +1103,30 @@ mod tests { assert_eq!(d.terms.find(&term("b", 2)), Some(0)); assert_eq!(d.terms.resolve(0), Some(&term("b", 2))); } + + /// A nested term registers before the term holding it, with its subject. + #[test] + fn nested_terms_register_inner_first() { + let mut d = DictNovelty::with_watermarks(vec![], 0); + let inner = term("inner", 1); + let outer = TripleTermValue { + s: crate::Sid::new(9, "outer"), + p: crate::Sid::new(9, "p"), + o: FlakeValue::TripleTerm(Box::new(inner.clone())), + dt: crate::namespaces::triple_term_datatype_sid().clone(), + lang: None, + }; + d.populate_from_flakes(&[Flake::new( + crate::Sid::new(9, "doc"), + crate::Sid::new(9, "mentions"), + FlakeValue::TripleTerm(Box::new(outer.clone())), + crate::namespaces::triple_term_datatype_sid().clone(), + 1, + true, + None, + )]); + assert_eq!(d.terms.find(&inner), Some(0)); + assert_eq!(d.terms.find(&outer), Some(1)); + assert!(d.subjects.find_subject(9, "inner").is_some()); + } } diff --git a/fluree-db-indexer/src/build/rebuild.rs b/fluree-db-indexer/src/build/rebuild.rs index e2f3e4cf6a..7861f9ccd2 100644 --- a/fluree-db-indexer/src/build/rebuild.rs +++ b/fluree-db-indexer/src/build/rebuild.rs @@ -607,30 +607,27 @@ where } all_attachments.append(&mut attachments); + let handles = crate::run_index::resolve::resolver::intern_chunk_terms( + terms, + &term_registry, + &mut |keys| { + keys.iter() + .map(|&k| term_builder.get_or_insert(k)) + .collect() + }, + ) + .map_err(|e| IndexerError::StorageWrite(format!("chunk {ci}: {e}")))?; let triple_term = fluree_db_core::value_id::ObjKind::TRIPLE_TERM.as_u8(); for record in records.iter_mut() { if record.o_kind != triple_term { continue; } - let term = terms.get(record.o_key as usize).ok_or_else(|| { + record.o_key = *handles.get(record.o_key as usize).ok_or_else(|| { IndexerError::StorageWrite(format!( "term ordinal {} out of range in chunk {ci}", record.o_key )) })?; - let key = fluree_db_core::triple_term::TermKey { - s_id: term.s_id.as_u64(), - p_id: term.p_id, - o_type: term_registry.resolve( - fluree_db_core::value_id::ObjKind::from_u8(term.o_kind), - fluree_db_core::DatatypeDictId::from_u16(term.dt), - term.lang_id, - ), - o_key: term.o_key, - }; - record.o_key = term_builder - .get_or_insert(key) - .map_err(|e| IndexerError::StorageWrite(e.to_string()))?; } // Sort by (g_id, SPOT). diff --git a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs index 0bb324659a..3090897661 100644 --- a/fluree-db-indexer/src/run_index/build/incremental_resolve.rs +++ b/fluree-db-indexer/src/run_index/build/incremental_resolve.rs @@ -742,29 +742,36 @@ pub async fn resolve_incremental_commits_v6( let base_wms: Vec<(u32, u32)> = base_refs.map(|r| r.watermarks.clone()).unwrap_or_default(); let mut builder = fluree_db_binary_index::dict::TermDictBuilder::above_watermarks(&base_wms); + // The window's terms, inner terms first, each level's base handles + // in one batched lookup: a reverse lookup reads a whole tree leaf. + let handles = crate::run_index::resolve::resolver::intern_chunk_terms( + &chunk_terms, + &o_type_registry, + &mut |keys| { + let found = match &base_reader { + Some(reader) => reader.find_handles(keys)?, + None => vec![None; keys.len()], + }; + keys.iter() + .zip(found) + .map(|(&key, found)| match found { + Some(handle) => Ok(handle), + None => builder.get_or_insert(key), + }) + .collect() + }, + )?; let triple_term = ObjKind::TRIPLE_TERM.as_u8(); - let mut record_keys = Vec::new(); - for (idx, record) in v1_records.iter().enumerate() { + for record in &mut v1_records { if record.o_kind != triple_term { continue; } - let term = chunk_terms.get(record.o_key as usize).ok_or_else(|| { + record.o_key = *handles.get(record.o_key as usize).ok_or_else(|| { IncrementalResolveError::Io(std::io::Error::new( std::io::ErrorKind::InvalidData, format!("term ordinal {} out of range", record.o_key), )) })?; - let key = fluree_db_core::triple_term::TermKey { - s_id: term.s_id.as_u64(), - p_id: term.p_id, - o_type: o_type_registry.resolve( - ObjKind::from_u8(term.o_kind), - fluree_db_core::DatatypeDictId::from_u16(term.dt), - term.lang_id, - ), - o_key: term.o_key, - }; - record_keys.push((idx, key)); } let prior = |g_id: u16, ann: u64| { prior_attachments @@ -779,25 +786,22 @@ pub async fn resolve_incremental_commits_v6( IncrementalResolveError::Io(io::Error::other("attachment ops without link ids")) })?) }; - // Resolve every base handle in one batched pass: a reverse lookup per - // term reads a whole tree leaf, and a re-point needs one for the term - // it retracts. A dry replay names the terms the real one will ask for. + // A re-point needs the base handle of the term it retracts; a dry + // replay names the terms the real one will ask for. let mut base_handles = HashMap::new(); - if let Some(reader) = &base_reader { - let mut wanted: Vec<_> = record_keys.iter().map(|&(_, key)| key).collect(); - if let Some(link_ids) = link_ids { - crate::run_index::resolve::link_synth::replay_attachments( - &mut attachments, - prior, - &o_type_registry, - link_ids, - &mut |key| { - wanted.push(key); - Ok(0) - }, - &mut |_| Ok(()), - )?; - } + if let (Some(reader), Some(link_ids)) = (&base_reader, link_ids) { + let mut wanted = Vec::new(); + crate::run_index::resolve::link_synth::replay_attachments( + &mut attachments, + prior, + &o_type_registry, + link_ids, + &mut |key| { + wanted.push(key); + Ok(0) + }, + &mut |_| Ok(()), + )?; wanted.sort_unstable(); wanted.dedup(); for (key, handle) in wanted.iter().zip(reader.find_handles(&wanted)?) { @@ -812,9 +816,6 @@ pub async fn resolve_incremental_commits_v6( None => builder.get_or_insert(key), } }; - for (idx, key) in record_keys { - v1_records[idx].o_key = handle_for(key)?; - } if let Some(link_ids) = link_ids { let emitted = crate::run_index::resolve::link_synth::replay_attachments( &mut attachments, diff --git a/fluree-db-indexer/src/run_index/resolve/resolver.rs b/fluree-db-indexer/src/run_index/resolve/resolver.rs index a05d1c64f5..66b240d4bb 100644 --- a/fluree-db-indexer/src/run_index/resolve/resolver.rs +++ b/fluree-db-indexer/src/run_index/resolve/resolver.rs @@ -1618,7 +1618,6 @@ impl SharedResolverState { .map_err(|e| e.to_string())?; fluree_db_core::triple_term::lexical_term_object(&value) } - RawObject::TripleTerm(_) => return Err("nested triple terms are not supported".into()), _ => None, }; let (o_kind, o_key) = match lexical { @@ -2330,6 +2329,63 @@ pub fn remap_term_record( Ok(()) } +/// The handle of each entry of a chunk's term table, from `intern`, which +/// maps a batch of keys to handles. An entry whose object is a triple term +/// holds that term's ordinal, always a lower one; entries go one nesting level +/// per batch, so an inner term's handle is in place for its outer term's key. +pub fn intern_chunk_terms( + terms: &[RunRecord], + registry: &fluree_db_core::o_type_registry::OTypeRegistry, + intern: &mut dyn FnMut(&[fluree_db_core::triple_term::TermKey]) -> io::Result>, +) -> io::Result> { + let triple_term = ObjKind::TRIPLE_TERM.as_u8(); + let mut depth: Vec = Vec::with_capacity(terms.len()); + for (i, term) in terms.iter().enumerate() { + let level = if term.o_kind == triple_term { + let inner = term.o_key as usize; + if inner >= i { + return Err(io::Error::new( + io::ErrorKind::InvalidData, + format!("term {i} names term {inner}, which does not precede it"), + )); + } + depth[inner] + 1 + } else { + 0 + }; + depth.push(level); + } + let mut handles = vec![0u64; terms.len()]; + let levels = depth.iter().max().map_or(0, |max| max + 1); + for level in 0..levels { + let at: Vec = (0..terms.len()).filter(|&i| depth[i] == level).collect(); + let keys: Vec<_> = at + .iter() + .map(|&i| { + let term = &terms[i]; + fluree_db_core::triple_term::TermKey { + s_id: term.s_id.as_u64(), + p_id: term.p_id, + o_type: registry.resolve( + ObjKind::from_u8(term.o_kind), + DatatypeDictId::from_u16(term.dt), + term.lang_id, + ), + o_key: if term.o_kind == triple_term { + handles[term.o_key as usize] + } else { + term.o_key + }, + } + }) + .collect(); + for (&i, handle) in at.iter().zip(intern(&keys)?) { + handles[i] = handle; + } + } + Ok(handles) +} + fn iso_to_epoch_ms(iso: &str) -> Option { chrono::DateTime::parse_from_rfc3339(iso) .ok() diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index e44e0658c0..a1d40ba5b2 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -4603,6 +4603,11 @@ pub(crate) fn compose_term_handle( Err(e) if matches!(e.kind(), ErrorKind::NotFound | ErrorKind::Unsupported) => {} Err(e) => return Err(e), } + // A provisional handle's key names its nested term's handle, so a term + // whose nested term has none stays materialized too. + if let FlakeValue::TripleTerm(inner) = &term.o { + compose_term_handle(inner, store, dict_novelty)?; + } let provisional = dict_novelty .filter(|dn| dn.is_initialized()) .and_then(|dn| dn.terms.find(term)) diff --git a/fluree-db-query/src/eval/compare.rs b/fluree-db-query/src/eval/compare.rs index f6749d7f32..c2b740e119 100644 --- a/fluree-db-query/src/eval/compare.rs +++ b/fluree-db-query/src/eval/compare.rs @@ -511,6 +511,35 @@ fn cmp_values_inner( } } +fn triple_term(v: &ComparableValue) -> Option<&fluree_db_core::TripleTermValue> { + match v { + ComparableValue::TypedLiteral { + val: FlakeValue::TripleTerm(term), + .. + } => Some(term), + _ => None, + } +} + +/// A triple term's object as a comparable value, with its datatype or tag. +fn term_object(term: &fluree_db_core::TripleTermValue) -> Option { + use fluree_db_core::DatatypeConstraint; + match &term.o { + FlakeValue::Ref(sid) => Some(ComparableValue::Sid(sid.clone())), + FlakeValue::TripleTerm(_) => Some(ComparableValue::TypedLiteral { + val: term.o.clone(), + dtc: None, + }), + o => { + let dtc = match &term.lang { + Some(tag) => DatatypeConstraint::LangTag(std::sync::Arc::from(tag.as_str())), + None => DatatypeConstraint::Explicit(term.dt.clone()), + }; + super::lit_to_comparable(o, &dtc, None) + } + } +} + /// Outcome of RDFterm-equal (`=` / `!=`). #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub(crate) enum EqOutcome { @@ -623,6 +652,22 @@ pub(crate) fn rdf_term_equal_in( /// operator mapping. Returns a three-valued outcome so an incomparable pair is /// a type error (excluding the row) rather than silently `false`/`true`. pub(crate) fn rdf_term_equal(a: &ComparableValue, b: &ComparableValue) -> EqOutcome { + // Two triple terms are equal when their subjects and predicates are the + // same terms and their objects are equal (SPARQL 1.2); a triple term + // equals nothing else. + match (triple_term(a), triple_term(b)) { + (Some(ta), Some(tb)) => { + if ta.s != tb.s || ta.p != tb.p { + return EqOutcome::Ne; + } + return match (term_object(ta), term_object(tb)) { + (Some(oa), Some(ob)) => rdf_term_equal(&oa, &ob), + _ => EqOutcome::TypeError, + }; + } + (Some(_), None) | (None, Some(_)) => return EqOutcome::Ne, + (None, None) => {} + } // XPath op:numeric-equal (F&O §4.2.3): NaN is not equal to anything, // including itself — `NaN = NaN` is false and `NaN != NaN` is true. The // numeric fast path below bottoms out in `numeric_cmp`'s bit-level total diff --git a/fluree-db-query/src/ir.rs b/fluree-db-query/src/ir.rs index 499eaf1bc2..9d356806b2 100644 --- a/fluree-db-query/src/ir.rs +++ b/fluree-db-query/src/ir.rs @@ -62,5 +62,7 @@ pub use projection::{ }; pub use query::{ConstructTemplate, Query, QueryOutput, Restriction, TemplateReification}; pub use reasoning::{ReasoningConfig, ReasoningModes}; -pub use term_components::{lower_reified_link, lower_term_value, Component, TermComponentsPattern}; +pub use term_components::{ + lower_reified_link, lower_term_object, lower_term_value, Component, TermComponentsPattern, +}; pub use triple::{Ref, Term, TriplePattern}; diff --git a/fluree-db-query/src/ir/term_components.rs b/fluree-db-query/src/ir/term_components.rs index c1f359228c..271a3afbb9 100644 --- a/fluree-db-query/src/ir/term_components.rs +++ b/fluree-db-query/src/ir/term_components.rs @@ -111,6 +111,27 @@ pub fn lower_term_value( ); } +/// Lower `<<( s p o )>>` as the object of an enclosing pattern — the object +/// of a triple term, or of an annotated edge: a composed constant (typed +/// `f:tripleTerm`, so an enclosing constant term composes too) or a fresh term +/// variable, with the patterns relating it to the term's components. +pub fn lower_term_object( + term: TriplePattern, + encoder: &E, + vars: &mut VarRegistry, + out: &mut Vec, +) -> (Term, Option) { + let object = term_patterns( + term, + || fresh_term_var(vars), + &|iri| encoder.encode_iri(iri), + out, + ); + let dtc = matches!(object, Term::Value(_)) + .then(|| DatatypeConstraint::Explicit(fluree_db_core::triple_term_datatype_sid().clone())); + (object, dtc) +} + /// The patterns of [`lower_reified_link`], given the `rdf:reifies` ref, the /// term variable (asked for only when the edge is not constant), and how to /// encode an IRI. @@ -122,24 +143,32 @@ pub(crate) fn link_patterns( encode_iri: &dyn Fn(&str) -> Option, out: &mut Vec, ) { - // Fully constant edge: compose the term itself. - if let Some(term) = constant_term(&edge) { - out.push(Pattern::Triple(TriplePattern { - s: annotation_ref, - p: reifies, - o: Term::Value(FlakeValue::TripleTerm(Box::new(term))), - dtc: None, - })); - return; - } - - let t = term_var(); + let mut components = Vec::new(); + let o = term_patterns(edge, term_var, encode_iri, &mut components); out.push(Pattern::Triple(TriplePattern { s: annotation_ref, p: reifies, - o: Term::Var(t), + o, dtc: None, })); + out.extend(components); +} + +/// The term `<<( s p o )>>` of `edge`: the composed term when every position +/// is constant, else a term variable whose relation to the positions goes +/// into `out`. +fn term_patterns( + edge: TriplePattern, + term_var: impl FnOnce() -> VarId, + encode_iri: &dyn Fn(&str) -> Option, + out: &mut Vec, +) -> Term { + // Fully constant edge: compose the term itself. + if let Some(term) = constant_term(&edge) { + return Term::Value(FlakeValue::TripleTerm(Box::new(term))); + } + + let t = term_var(); let accessor = |f: Function| Expression::call(f, vec![Expression::Var(t)]); let same_term = |f: Function, constant: Expression| { @@ -219,6 +248,7 @@ pub(crate) fn link_patterns( { out.push(Pattern::TermComponents(tc)); } + Term::Var(t) } /// A `?__term_N` variable no pattern uses yet. diff --git a/fluree-db-query/src/parse/ast.rs b/fluree-db-query/src/parse/ast.rs index 2443a80319..e511364335 100644 --- a/fluree-db-query/src/parse/ast.rs +++ b/fluree-db-query/src/parse/ast.rs @@ -1159,7 +1159,7 @@ pub enum UnresolvedPattern { /// or named IRI). annotation: UnresolvedTerm, /// The base edge being reified (subject, predicate, object). - edge: UnresolvedTriplePattern, + edge: UnresolvedTermPattern, /// Patterns about the annotation subject (lowered from the /// non-`@`-keyword properties of the enclosing node). body: Vec, @@ -1169,10 +1169,42 @@ pub enum UnresolvedPattern { TripleTermValue { subject: UnresolvedTerm, predicate: UnresolvedTerm, - term: UnresolvedTriplePattern, + term: UnresolvedTermPattern, }, } +/// The triple of a triple term `<<( s p o )>>`, whose object may be another +/// triple term. +#[derive(Debug, Clone, PartialEq)] +pub struct UnresolvedTermPattern { + pub s: UnresolvedTerm, + pub p: UnresolvedTerm, + pub o: UnresolvedTermObject, +} + +/// The object of an [`UnresolvedTermPattern`]. +#[derive(Debug, Clone, PartialEq)] +pub enum UnresolvedTermObject { + Value { + o: UnresolvedTerm, + dtc: Option, + }, + Term(Box), +} + +impl From for UnresolvedTermPattern { + fn from(tp: UnresolvedTriplePattern) -> Self { + Self { + s: tp.s, + p: tp.p, + o: UnresolvedTermObject::Value { + o: tp.o, + dtc: tp.dtc, + }, + } + } +} + impl UnresolvedPattern { /// Create a triple pattern pub fn triple(pattern: UnresolvedTriplePattern) -> Self { @@ -1325,11 +1357,11 @@ impl UnresolvedQuery { UnresolvedPattern::Graph { patterns: inner, .. } => collect(inner, out), - UnresolvedPattern::EdgeAnnotation { edge, body, .. } - | UnresolvedPattern::AnnotationTarget { edge, body, .. } => { + UnresolvedPattern::EdgeAnnotation { edge, body, .. } => { out.push(edge); collect(body, out); } + UnresolvedPattern::AnnotationTarget { body, .. } => collect(body, out), UnresolvedPattern::Filter(_) | UnresolvedPattern::Bind { .. } | UnresolvedPattern::Unwind { .. } diff --git a/fluree-db-query/src/parse/lower.rs b/fluree-db-query/src/parse/lower.rs index 3251d3ce27..19caac4ef1 100644 --- a/fluree-db-query/src/parse/lower.rs +++ b/fluree-db-query/src/parse/lower.rs @@ -8,8 +8,8 @@ use super::ast::{ UnresolvedConstructTemplate, UnresolvedDatatypeConstraint, UnresolvedExpression, UnresolvedForwardItem, UnresolvedHydrationSpec, UnresolvedNestedSelectSpec, UnresolvedOptions, UnresolvedPathExpr, UnresolvedPattern, UnresolvedProjection, UnresolvedQuery, UnresolvedRoot, - UnresolvedSortDirection, UnresolvedSortSpec, UnresolvedTerm, UnresolvedTriplePattern, - UnresolvedValue, + UnresolvedSortDirection, UnresolvedSortSpec, UnresolvedTerm, UnresolvedTermObject, + UnresolvedTermPattern, UnresolvedTriplePattern, UnresolvedValue, }; use super::encode::{IriEncoder, NoEncoder}; use super::error::{ParseError, Result}; @@ -459,8 +459,8 @@ pub fn lower_unresolved_pattern( body, } => { let lowered_annotation = lower_ref_term(annotation, encoder, vars)?; - let lowered_edge = lower_triple_pattern(edge, encoder, vars)?; let mut out = Vec::new(); + let lowered_edge = lower_term_pattern(edge, encoder, vars, &mut out)?; crate::ir::lower_reified_link( lowered_annotation, lowered_edge, @@ -478,8 +478,8 @@ pub fn lower_unresolved_pattern( } => { let subject = lower_ref_term(subject, encoder, vars)?; let predicate = lower_ref_term(predicate, encoder, vars)?; - let term = lower_triple_pattern(term, encoder, vars)?; let mut out = Vec::new(); + let term = lower_term_pattern(term, encoder, vars, &mut out)?; crate::ir::lower_term_value(subject, predicate, term, encoder, vars, &mut out); Ok(out) } @@ -600,6 +600,38 @@ pub fn coerce_value_by_datatype(value: FlakeValue, datatype_iri: &str) -> Result } /// Lower an unresolved triple pattern to a resolved TriplePattern +/// A triple term's triple; a nested term object lowers to a term variable +/// (or a composed constant) whose patterns go into `out`. +fn lower_term_pattern( + tp: &UnresolvedTermPattern, + encoder: &E, + vars: &mut VarRegistry, + out: &mut Vec, +) -> Result { + match &tp.o { + UnresolvedTermObject::Value { o, dtc } => lower_triple_pattern( + &UnresolvedTriplePattern { + s: tp.s.clone(), + p: tp.p.clone(), + o: o.clone(), + dtc: dtc.clone(), + }, + encoder, + vars, + ), + UnresolvedTermObject::Term(inner) => { + let inner = lower_term_pattern(inner, encoder, vars, out)?; + let (o, dtc) = crate::ir::lower_term_object(inner, encoder, vars, out); + Ok(TriplePattern { + s: lower_ref_term(&tp.s, encoder, vars)?, + p: lower_ref_term(&tp.p, encoder, vars)?, + o, + dtc, + }) + } + } +} + fn lower_triple_pattern( pattern: &UnresolvedTriplePattern, encoder: &E, diff --git a/fluree-db-query/src/parse/mod.rs b/fluree-db-query/src/parse/mod.rs index de2fc9aa4f..5738084c0d 100644 --- a/fluree-db-query/src/parse/mod.rs +++ b/fluree-db-query/src/parse/mod.rs @@ -4296,7 +4296,13 @@ mod tests { assert!( matches!(&edge.p, UnresolvedTerm::Iri(p) if p.as_ref() == "http://example.org/worksFor") ); - assert!(matches!(&edge.o, UnresolvedTerm::Var(v) if v.as_ref() == "?org")); + assert!(matches!( + &edge.o, + crate::parse::ast::UnresolvedTermObject::Value { + o: UnresolvedTerm::Var(v), + .. + } if v.as_ref() == "?org" + )); // Body: 2 facts about the annotation (ex:role, ex:since). assert_eq!(body.len(), 2, "body should have ex:role + ex:since"); } diff --git a/fluree-db-query/src/parse/node_map.rs b/fluree-db-query/src/parse/node_map.rs index bf28e03f57..f44d4f76da 100644 --- a/fluree-db-query/src/parse/node_map.rs +++ b/fluree-db-query/src/parse/node_map.rs @@ -6,8 +6,9 @@ use super::ast::{ UnresolvedDatatypeConstraint, UnresolvedIndexSearchPattern, UnresolvedIndexSearchTarget, - UnresolvedPathExpr, UnresolvedPattern, UnresolvedQuery, UnresolvedTerm, - UnresolvedTriplePattern, UnresolvedVectorSearchPattern, UnresolvedVectorSearchTarget, + UnresolvedPathExpr, UnresolvedPattern, UnresolvedQuery, UnresolvedTerm, UnresolvedTermObject, + UnresolvedTermPattern, UnresolvedTriplePattern, UnresolvedVectorSearchPattern, + UnresolvedVectorSearchTarget, }; use super::error::{ParseError, Result}; use super::policy::JsonLdParseCtx; @@ -1305,7 +1306,8 @@ fn parse_literal_edge_annotation( } /// Lower a node-map describing one triple (an `@reifies` value, or a -/// triple-term value's `@id`) to a single `UnresolvedTriplePattern`. +/// triple-term value's `@id`) to its pattern; its object may be a triple +/// term. /// /// Reuses the regular `parse_node_map` machinery via a buffer, then /// asserts the result is exactly one triple. `what` names the form in @@ -1317,7 +1319,7 @@ fn parse_reifies_edge( subject_counter: &mut u32, nested_counter: &mut u32, object_var_parsing: bool, -) -> Result { +) -> Result { let JsonValue::Object(rmap) = value else { return Err(ParseError::InvalidWhere(format!( "{what} must be a node-map describing the base triple" @@ -1343,7 +1345,16 @@ fn parse_reifies_edge( } match buffer.patterns.into_iter().next().unwrap() { - UnresolvedPattern::Triple(tp) => Ok(tp), + UnresolvedPattern::Triple(tp) => Ok(tp.into()), + UnresolvedPattern::TripleTermValue { + subject, + predicate, + term, + } => Ok(UnresolvedTermPattern { + s: subject, + p: predicate, + o: UnresolvedTermObject::Term(Box::new(term)), + }), _ => Err(ParseError::InvalidWhere(format!( "{what} must describe a basic triple pattern; \ property paths, lists, and other shapes are deferred to v2" diff --git a/fluree-db-query/src/term_components.rs b/fluree-db-query/src/term_components.rs index efaeb0af8d..f71d588f1a 100644 --- a/fluree-db-query/src/term_components.rs +++ b/fluree-db-query/src/term_components.rs @@ -92,39 +92,44 @@ impl TermComponentsOperator { } } - /// The terms novelty's links assert that the dictionary does not hold. + /// The terms novelty holds that the dictionary does not: those of any + /// predicate, and those nested in them. The relation may offer a term no + /// asserted flake still holds; the pattern binding the term joins it out. fn collect_novelty_terms(&self, ctx: &ExecutionContext<'_>) -> Result> { - let reifies = fluree_db_core::rdf_reifies_sid(); - let (first, rhs) = crate::fast_path_common::predicate_walk_bounds(reifies); + let dict_novelty = ctx.dict_novelty.as_ref(); let mut seen: HashSet = HashSet::new(); - let mut visit = |overlay: &dyn fluree_db_core::OverlayProvider, g_id, to_t| { - overlay.for_each_overlay_flake( - g_id, - fluree_db_core::IndexType::Post, - Some(&first), - Some(&rhs), - false, - to_t, - &mut |f| { - if f.op && f.p == *reifies { - if let FlakeValue::TripleTerm(term) = &f.o { - seen.insert((**term).clone()); + match dict_novelty.filter(|dn| dn.is_initialized()) { + Some(dn) => seen.extend(dn.terms.iter().map(|(_, term)| term.clone())), + None => { + let mut visit = |overlay: &dyn fluree_db_core::OverlayProvider, g_id, to_t| { + overlay.for_each_overlay_flake( + g_id, + fluree_db_core::IndexType::Post, + None, + None, + false, + to_t, + &mut |f| { + let mut value = &f.o; + while let FlakeValue::TripleTerm(term) = value { + seen.insert((**term).clone()); + value = &term.o; + } + }, + ); + }; + match ctx.active_graphs() { + crate::dataset::ActiveGraphs::Single => { + visit(ctx.overlay(), ctx.binary_g_id, ctx.to_t); + } + crate::dataset::ActiveGraphs::Many(graphs) => { + for g in graphs { + visit(g.overlay, g.g_id, g.to_t); } } - }, - ); - }; - match ctx.active_graphs() { - crate::dataset::ActiveGraphs::Single => { - visit(ctx.overlay(), ctx.binary_g_id, ctx.to_t); - } - crate::dataset::ActiveGraphs::Many(graphs) => { - for g in graphs { - visit(g.overlay, g.g_id, g.to_t); } } } - let dict_novelty = ctx.dict_novelty.as_ref(); let mut out = Vec::new(); for term in seen { let encoded = match &self.store { diff --git a/fluree-db-sparql/src/lower/annotation.rs b/fluree-db-sparql/src/lower/annotation.rs index b84278d79e..6ecb360bdb 100644 --- a/fluree-db-sparql/src/lower/annotation.rs +++ b/fluree-db-sparql/src/lower/annotation.rs @@ -145,6 +145,15 @@ impl LoweringContext<'_, E> { let r = self.lower_reified_triple(qt, cache, out)?; Ok((r.into(), None)) } + SparqlTerm::TripleTerm(tt) => { + let term = self.lower_triple_term(tt, out)?; + Ok(fluree_db_query::ir::lower_term_object( + term, + self.encoder, + self.vars, + out, + )) + } other if with_constraint => self.lower_object_with_constraint(other), other => self.lower_object_with_term_constraint(other), } diff --git a/fluree-db-sparql/src/lower/rdf_star.rs b/fluree-db-sparql/src/lower/rdf_star.rs index 748f9873d7..d9f373745e 100644 --- a/fluree-db-sparql/src/lower/rdf_star.rs +++ b/fluree-db-sparql/src/lower/rdf_star.rs @@ -93,15 +93,9 @@ impl LoweringContext<'_, E> { other => self.lower_subject(other)?, }; let p = self.lower_predicate(&tp.predicate)?; - // A triple-term value: the triple and the term's components. - if let SparqlTerm::TripleTerm(tt) = &tp.object { - if tp.annotation.is_some() { - return Err(LowerError::not_implemented( - "an annotation on a triple whose object is a triple term \ - (a nested triple term)", - tt.span, - )); - } + // A triple-term value: the triple and the term's components. An + // annotated one is an edge like any other, below. + if let (SparqlTerm::TripleTerm(tt), None) = (&tp.object, &tp.annotation) { let term = self.lower_triple_term(tt, &mut result)?; fluree_db_query::ir::lower_term_value( s, diff --git a/fluree-db-sparql/src/lower/term.rs b/fluree-db-sparql/src/lower/term.rs index 66d7ba3714..d6bcf7bbb6 100644 --- a/fluree-db-sparql/src/lower/term.rs +++ b/fluree-db-sparql/src/lower/term.rs @@ -523,12 +523,54 @@ impl LoweringContext<'_, E> { "RDF 1.2 reified triples (`<< s p o >>`) as VALUES data", qt.span, )), - // SPARQL 1.2 triple-term value as VALUES data: parse-accepted, - // lower-deferred (burn-down D-1). - SparqlTerm::TripleTerm(tt) => Err(LowerError::not_implemented( - "SPARQL 1.2 triple-term values (`<<( s p o )>>`) as VALUES data", - tt.span, - )), + SparqlTerm::TripleTerm(tt) => { + let node = |binding: Binding| match binding { + Binding::Sid { sid, .. } => Ok(sid), + _ => Err(LowerError::not_implemented( + "a triple term in VALUES data whose subject is not an IRI", + tt.span, + )), + }; + let s = match &tt.subject { + SubjectTerm::Iri(iri) => { + node(self.term_to_binding(&SparqlTerm::Iri(iri.clone()))?)? + } + _ => node(Binding::Unbound)?, + }; + let p = match &tt.predicate { + PredicateTerm::Iri(iri) => { + node(self.term_to_binding(&SparqlTerm::Iri(iri.clone()))?)? + } + PredicateTerm::Var(_) => node(Binding::Unbound)?, + }; + let (o, dt, lang) = match self.term_to_binding(&tt.object)? { + Binding::Sid { sid, .. } => ( + FlakeValue::Ref(sid), + fluree_db_core::edge::id_datatype_sid(), + None, + ), + Binding::Lit { val, dtc, .. } => { + let lang = dtc.lang_tag().map(str::to_string); + (val, dtc.datatype().clone(), lang) + } + _ => { + return Err(LowerError::not_implemented( + "a triple term in VALUES data whose object is not a constant", + tt.span, + )) + } + }; + Ok(Binding::lit( + FlakeValue::TripleTerm(Box::new(fluree_db_core::TripleTermValue { + s, + p, + o, + dt, + lang, + })), + fluree_db_core::triple_term_datatype_sid().clone(), + )) + } } } } diff --git a/fluree-db-sparql/src/parse/query/term.rs b/fluree-db-sparql/src/parse/query/term.rs index 2976caeb2e..d275bed69c 100644 --- a/fluree-db-sparql/src/parse/query/term.rs +++ b/fluree-db-sparql/src/parse/query/term.rs @@ -941,12 +941,10 @@ impl super::Parser<'_> { /// Parse a single triple term `<<( s p o )>>` after the opening /// `TripleTermStart` token has been verified by the caller. /// - /// Strict v1 rules: - /// - Triple-term subject must be an IRI, blank node, or variable - /// (no nested triple terms). + /// - Triple-term subject must be an IRI, blank node, or variable. /// - Triple-term predicate must be a simple predicate (no paths). - /// - Triple-term object must be an ordinary term (no nested triple - /// terms, no annotation tails). + /// - Triple-term object is a term or another triple term, with no + /// annotation tail. fn parse_triple_term(&mut self) -> Option { let start = self.stream.current_span(); if !self.stream.match_token(&TokenKind::TripleTermStart) { @@ -957,27 +955,18 @@ impl super::Parser<'_> { self.reject_collection_in_quoted_context()?; let subject = self.parse_subject()?; - // The `rdf:reifies` object stays strict per v1 (pr-w2a): its inner - // subject may not be a nested triple term or reified triple. Now - // that `parse_subject` accepts `<<(` as a value, guard both variants - // (bare triple-term values in general BGP positions may nest — that - // is the separate `parse_triple_term_value` path). + // `ttSubject` is an IRI, blank node or variable (`parse_subject` + // also accepts the reified-triple and triple-term forms). if matches!( subject, SubjectTerm::QuotedTriple(_) | SubjectTerm::TripleTerm(_) ) { self.stream - .error_at_current("nested triple terms are not supported in v1"); + .error_at_current("a triple term's subject cannot be a triple term"); return None; } let predicate = self.parse_simple_predicate()?; - // Reject nested triple terms in object position. - if self.stream.check(&TokenKind::TripleTermStart) { - self.stream - .error_at_current("nested triple terms are not supported in v1"); - return None; - } // Reified triples are not grammatical inside a triple term // either (`ttObject` has no `ReifiedTriple` production). if self.stream.check(&TokenKind::TripleStart) { diff --git a/fluree-db-sparql/src/parse/query/tests.rs b/fluree-db-sparql/src/parse/query/tests.rs index 255e18525f..d88efa26ae 100644 --- a/fluree-db-sparql/src/parse/query/tests.rs +++ b/fluree-db-sparql/src/parse/query/tests.rs @@ -2633,7 +2633,7 @@ fn nested_triple_term_in_subject_is_rejected() { &format!( "{RDF_PREFIX}{EX_PREFIX}SELECT * WHERE {{ ?ann rdf:reifies <<( <<( ex:a ex:b ex:c )>> ex:p ex:o )>> . }}" ), - "nested triple terms", + "cannot be a triple term", ); } diff --git a/fluree-db-transact/src/flake_sink.rs b/fluree-db-transact/src/flake_sink.rs index 97a79359af..ec548081d9 100644 --- a/fluree-db-transact/src/flake_sink.rs +++ b/fluree-db-transact/src/flake_sink.rs @@ -217,8 +217,7 @@ impl<'a> FlakeSink<'a> { } } -/// The triple-term value of a resolved `<<( s p o )>>`. A nested triple term -/// as its object is refused. +/// The triple-term value of a resolved `<<( s p o )>>`. pub(crate) fn triple_term_value( s: Option, p: Option, @@ -229,9 +228,6 @@ pub(crate) fn triple_term_value( "a triple term's subject and predicate must be IRIs or blank nodes", )); }; - if matches!(o, FlakeValue::TripleTerm(_)) { - return Err(SinkError::rejected("nested triple terms are not supported")); - } Ok(FlakeValue::TripleTerm(Box::new( fluree_db_core::TripleTermValue { s, diff --git a/fluree-db-transact/src/generate/flakes.rs b/fluree-db-transact/src/generate/flakes.rs index 380ba9b5ef..fffb132ae1 100644 --- a/fluree-db-transact/src/generate/flakes.rs +++ b/fluree-db-transact/src/generate/flakes.rs @@ -525,11 +525,6 @@ impl<'a> FlakeGenerator<'a> { bindings: &Batch, row: usize, ) -> Result> { - if matches!(term.o, TemplateTerm::TripleTerm(_)) { - return Err(TransactError::InvalidTerm( - "a triple term cannot nest another".to_string(), - )); - } let s = self.resolve_subject(&term.s, bindings, row)?; let p = self.resolve_predicate(&term.p, bindings, row)?; let explicit_dt = term @@ -629,13 +624,6 @@ pub(crate) fn reified_triple_link( ann: &Sid, t: i64, ) -> Result { - if matches!(o, FlakeValue::TripleTerm(_)) { - return Err(TransactError::UnsupportedFeature( - "reifying a triple whose object is a triple term needs a nested triple \ - term, which is not supported" - .to_string(), - )); - } let dt = dtc.datatype().clone(); validate_value_dt_pair(&o, &dt)?; let term = fluree_db_core::TripleTermValue { diff --git a/fluree-db-transact/src/import_sink.rs b/fluree-db-transact/src/import_sink.rs index 9b89a4034b..6a688e6ced 100644 --- a/fluree-db-transact/src/import_sink.rs +++ b/fluree-db-transact/src/import_sink.rs @@ -590,35 +590,37 @@ mod inner { self.write_term_record(ann, None, term, t) } - /// Spool `s p <<( … )>>`: the term's components as a term-table - /// pseudo-record, and the statement with the term's ordinal as its - /// object. `p` is `rdf:reifies` when `None`. - fn write_term_record( + /// A term's term-table entry, written after its inner term's when it + /// nests one (whose ordinal is then its object): the entry's ordinal, + /// or `None` when its object has no spool encoding. + fn spool_term( &mut self, - s: &Sid, - p: Option<&Sid>, term: &fluree_db_core::TripleTermValue, t: i64, - ) -> Result<(), CommitCodecError> { + ) -> Result, CommitCodecError> { let s_id = self.assign_subject_id(&term.s); let p_id = self.assign_predicate_id(&term.p); let dt_id = self.assign_datatype_id(&term.dt)?; // An arena handle names a value only within one graph and // predicate; a term keys the object by its canonical form. - let resolved = match fluree_db_core::triple_term::lexical_term_object(&term.o) { - Some((o_type, form)) => { - let o_kind = if o_type == fluree_db_core::o_type::OType::VECTOR { - ObjKind::VECTOR_ID - } else { - ObjKind::NUM_BIG - }; - let id = self.assign_string_id(&form); - Some((o_kind.as_u8(), ObjKey::encode_u32_id(id).as_u64())) - } - None => self.resolve_object_value(&term.o, p_id), + let resolved = if let FlakeValue::TripleTerm(inner) = &term.o { + self.spool_term(inner, t)? + .map(|ordinal| (ObjKind::TRIPLE_TERM.as_u8(), ordinal)) + } else if let Some((o_type, form)) = + fluree_db_core::triple_term::lexical_term_object(&term.o) + { + let o_kind = if o_type == fluree_db_core::o_type::OType::VECTOR { + ObjKind::VECTOR_ID + } else { + ObjKind::NUM_BIG + }; + let id = self.assign_string_id(&form); + Some((o_kind.as_u8(), ObjKey::encode_u32_id(id).as_u64())) + } else { + self.resolve_object_value(&term.o, p_id) }; let Some((o_kind, o_key)) = resolved else { - return Ok(()); + return Ok(None); }; let lang_id = term .lang @@ -638,7 +640,22 @@ mod inner { lang_id, i: LIST_INDEX_NONE, }); + Ok(Some(ordinal)) + } + /// Spool `s p <<( … )>>`: the term's components as a term-table + /// pseudo-record, and the statement with the term's ordinal as its + /// object. `p` is `rdf:reifies` when `None`. + fn write_term_record( + &mut self, + s: &Sid, + p: Option<&Sid>, + term: &fluree_db_core::TripleTermValue, + t: i64, + ) -> Result<(), CommitCodecError> { + let Some(ordinal) = self.spool_term(term, t)? else { + return Ok(()); + }; let link_p = match (p, self.rdf_reifies_pid) { (Some(p), _) => self.assign_predicate_id(p), (None, Some(p)) => p, diff --git a/fluree-db-transact/src/lower_sparql_update.rs b/fluree-db-transact/src/lower_sparql_update.rs index 51b044c42a..594452bd28 100644 --- a/fluree-db-transact/src/lower_sparql_update.rs +++ b/fluree-db-transact/src/lower_sparql_update.rs @@ -1251,8 +1251,10 @@ fn rewrite_blank_nodes_to_vars(pattern: &QuadPattern) -> QuadPattern { } }; rewrite(&mut out.subject, &mut out.object); - if let Term::TripleTerm(tt) = &mut out.object { + let mut object = &mut out.object; + while let Term::TripleTerm(tt) = object { rewrite(&mut tt.subject, &mut tt.object); + object = &mut tt.object; } out }; @@ -1520,19 +1522,6 @@ fn lower_triple_to_template( let result = literal_to_template(lit, prologue, ns)?; (result.term, result.dtc) } - Term::TripleTerm(tt) => { - let s = subject_to_template(&tt.subject, prologue, ns, vars, bnodes)?; - let p = predicate_to_template(&tt.predicate, prologue, ns, vars)?; - let (o, dtc) = match &tt.object { - Term::Literal(lit) => { - let result = literal_to_template(lit, prologue, ns)?; - (result.term, result.dtc) - } - other => (object_to_template(other, prologue, ns, vars, bnodes)?, None), - }; - let term = TemplateTripleTerm { s, p, o, dtc }; - (TemplateTerm::TripleTerm(Box::new(term)), None) - } other => (object_to_template(other, prologue, ns, vars, bnodes)?, None), }; @@ -1642,16 +1631,7 @@ fn lower_triple_to_delete_template_delete_where( ) -> Result { let subject = delete_where_subject_template(&triple.subject, prologue, ns, vars, bnodes)?; let predicate = delete_where_predicate_template(&triple.predicate, prologue, ns, vars)?; - let (object, dtc) = match &triple.object { - Term::TripleTerm(tt) => { - let s = delete_where_subject_template(&tt.subject, prologue, ns, vars, bnodes)?; - let p = delete_where_predicate_template(&tt.predicate, prologue, ns, vars)?; - let (o, dtc) = delete_where_object_template(&tt.object, prologue, ns, vars, bnodes)?; - let term = TemplateTripleTerm { s, p, o, dtc }; - (TemplateTerm::TripleTerm(Box::new(term)), None) - } - other => delete_where_object_template(other, prologue, ns, vars, bnodes)?, - }; + let (object, dtc) = delete_where_object_template(&triple.object, prologue, ns, vars, bnodes)?; Ok(TripleTemplate { subject, @@ -1763,9 +1743,11 @@ fn delete_where_object_template( span: qt.span, }); } + // `lower_delete_where` sends a triple-term object down the + // graph-pattern lane. Term::TripleTerm(tt) => { return Err(LowerError::UnsupportedFeature { - feature: "a triple term nested in a triple term", + feature: "a triple term on the triple-only DELETE WHERE lane", span: tt.span, }); } @@ -1948,10 +1930,19 @@ fn object_to_template( feature: "RDF 1.2 reified triple (`<< s p o >>`) in SPARQL UPDATE (deferred)", span: qt.span, }), - Term::TripleTerm(tt) => Err(LowerError::UnsupportedFeature { - feature: "a triple term nested in a triple term", - span: tt.span, - }), + Term::TripleTerm(tt) => { + let s = subject_to_template(&tt.subject, prologue, ns, vars, bnodes)?; + let p = predicate_to_template(&tt.predicate, prologue, ns, vars)?; + let (o, dtc) = match &tt.object { + Term::Literal(lit) => { + let result = literal_to_template(lit, prologue, ns)?; + (result.term, result.dtc) + } + other => (object_to_template(other, prologue, ns, vars, bnodes)?, None), + }; + let term = TemplateTripleTerm { s, p, o, dtc }; + Ok(TemplateTerm::TripleTerm(Box::new(term))) + } } } diff --git a/fluree-db-transact/src/parse/edge_annotations.rs b/fluree-db-transact/src/parse/edge_annotations.rs index ab7784c2c3..9501a52eb1 100644 --- a/fluree-db-transact/src/parse/edge_annotations.rs +++ b/fluree-db-transact/src/parse/edge_annotations.rs @@ -94,6 +94,17 @@ pub(crate) enum ReifiedObjectShape { /// to the same `lang` the base flake carries via `flake.m.lang`. language: Option, }, + /// Object is a triple term: the node naming its triple, + /// `{"@id": s, p: o}`. + TripleTerm(Value), +} + +/// The triple-term shape of an object written `{"@id": {"@id": s, p: o}}`. +fn triple_term_shape(map: &Map, id_alias: &str) -> Option { + map.get("@id") + .or_else(|| map.get(id_alias)) + .filter(|id| id.is_object()) + .map(|node| ReifiedObjectShape::TripleTerm(node.clone())) } /// Reject the deferred JSON-LD wrapper shapes that can't carry an edge @@ -347,6 +358,7 @@ pub(crate) fn emit_reifies_object_payload(shape: &ReifiedObjectShape) -> (Value, ReifiedObjectShape::Literal { value, language, .. } => (value.clone(), language.clone()), + ReifiedObjectShape::TripleTerm(node) => (json!({ "@id": node }), None), } } @@ -1010,6 +1022,8 @@ fn lift_annotations_under_predicate( // becomes a silent no-op. reject_context_coercion_on_annotated_literal(map, predicate, ctx)?; classify_reified_object(map)? + } else if let Some(shape) = triple_term_shape(map, ctx.id_key.as_str()) { + shape } else { let id_alias = ctx.id_key.as_str(); let object_id = map @@ -1493,6 +1507,9 @@ fn reifies_slots(block: Value) -> Result> { } Value::Object(object) => match object.get("@id") { Some(Value::String(id)) if object.len() == 1 => ReifiedObjectShape::Iri(id.clone()), + Some(node @ Value::Object(_)) if object.len() == 1 => { + ReifiedObjectShape::TripleTerm(node.clone()) + } _ => { return Err(TransactError::Parse( "@reifies object must be an @id reference or a value".to_string(), @@ -2046,6 +2063,8 @@ fn intercept_annotations_for_predicate( // cannot silently diverge from the base flake's. reject_context_coercion_on_annotated_literal(map, predicate, walk.json_ld)?; classify_reified_object(map)? + } else if let Some(shape) = triple_term_shape(map, walk.json_ld.id_key.as_str()) { + shape } else { let object_id = ensure_subject_id(map, walk, ctx); ReifiedObjectShape::Iri(object_id) diff --git a/fluree-db-transact/src/parse/jsonld.rs b/fluree-db-transact/src/parse/jsonld.rs index 3352f7e0e8..6ad7788ced 100644 --- a/fluree-db-transact/src/parse/jsonld.rs +++ b/fluree-db-transact/src/parse/jsonld.rs @@ -1755,9 +1755,6 @@ fn parse_expanded_triple_term_with_ctx( Some(o) if objects.is_empty() && asserted.is_empty() && o.list_index.is_none() => o, _ => return Err(invalid("it must describe exactly one triple")), }; - if matches!(o.term, TemplateTerm::TripleTerm(_)) { - return Err(invalid("a triple term cannot nest another")); - } Ok(ParsedValue::new(TemplateTerm::TripleTerm(Box::new( TemplateTripleTerm { s, diff --git a/fluree-db-transact/src/parse/trig_meta.rs b/fluree-db-transact/src/parse/trig_meta.rs index 979aee9f9c..ffe45f3b22 100644 --- a/fluree-db-transact/src/parse/trig_meta.rs +++ b/fluree-db-transact/src/parse/trig_meta.rs @@ -963,9 +963,7 @@ impl<'a> TrigMetaParser<'a> { fn triple_term_value_error(&self) -> TransactError { TransactError::Parse( - "a triple term ('<<( … )>>') is a value: it cannot be a subject, and \ - nested triple terms are not supported" - .to_string(), + "a triple term ('<<( … )>>') is a value and cannot be a subject".to_string(), ) } @@ -1028,10 +1026,7 @@ impl<'a> TrigMetaParser<'a> { self.advance(); // `<<` let subject = self.parse_subject()?; let predicate = self.parse_predicate()?; - let object = match self.current().kind { - TokenKind::TripleTermStart => return Err(self.triple_term_value_error()), - _ => self.parse_object()?, - }; + let object = self.parse_object()?; let reifier = if self.check(&TokenKind::Tilde) { self.advance(); self.parse_reifier_term()? @@ -1084,7 +1079,6 @@ impl<'a> TrigMetaParser<'a> { }; let predicate = self.parse_predicate()?; let object = match self.current().kind { - TokenKind::TripleTermStart => return Err(self.triple_term_value_error()), TokenKind::ReifiedTripleStart => { return Err(TransactError::Parse( "reified triples ('<< … >>') are not allowed inside a triple term".to_string(), @@ -1300,18 +1294,6 @@ impl<'a> TrigMetaParser<'a> { loop { if self.check(&TokenKind::TripleTermStart) && self.predicate_is_reifies(predicate)? { self.parse_reifies_triple_term(subject)?; - } else if self.check(&TokenKind::TripleTermStart) { - if self.annotation_depth > 0 { - return Err(self.annotation_of_annotation_error()); - } - let (s, p, o) = self.parse_triple_term_parts()?; - if matches!( - self.current().kind, - TokenKind::Tilde | TokenKind::AnnotationOpen - ) { - return Err(self.triple_term_value_error()); - } - objects.push(ObjectValue::TripleTerm(Box::new((s, p, o)))); } else { let object = self.parse_object()?; if matches!( @@ -1426,7 +1408,13 @@ impl<'a> TrigMetaParser<'a> { } TermValue::BlankNode(label) => ObjectValue::BlankNode(label), }), - TokenKind::TripleTermStart => Err(self.triple_term_value_error()), + TokenKind::TripleTermStart => { + if self.annotation_depth > 0 { + return Err(self.annotation_of_annotation_error()); + } + let (s, p, o) = self.parse_triple_term_parts()?; + Ok(ObjectValue::TripleTerm(Box::new((s, p, o)))) + } _ => Err(TransactError::Parse(format!( "expected object, found {}", self.current().kind @@ -2817,16 +2805,6 @@ ex:alice ex:note "value with a { brace" . "<<( ex:s ex:p ex:o )>> ex:q ex:z .", "cannot be a subject", ), - ( - "nested triple term", - "ex:r rdf:reifies <<( ex:s ex:p <<( ex:x ex:y ex:z )>> )>> .", - "nested triple terms", - ), - ( - "annotation on a triple-term value", - "ex:a ex:q <<( ex:s ex:p ex:o )>> {| ex:n 1 |} .", - "nested triple terms", - ), ( "star inside annotation body", "ex:s ex:p ex:o {| ex:q ex:z {| ex:n 1 |} |} .", diff --git a/fluree-graph-json-ld/src/adapter.rs b/fluree-graph-json-ld/src/adapter.rs index 1e037b064b..8acc945c17 100644 --- a/fluree-graph-json-ld/src/adapter.rs +++ b/fluree-graph-json-ld/src/adapter.rs @@ -316,15 +316,15 @@ fn process_triple_term( Value::Array(_) => return Err(invalid("it must describe exactly one triple")), value => value, }; - // A node with properties would assert them, and a term does not nest. + // A node with properties would assert them. let reference_or_value = match value { - Value::Object(o) => { - o.contains_key("@value") || (o.len() == 1 && o.get("@id").is_some_and(Value::is_string)) - } + Value::Object(o) => o.contains_key("@value") || (o.len() == 1 && o.contains_key("@id")), _ => true, }; if !reference_or_value { - return Err(invalid("its object must be a reference or a value")); + return Err(invalid( + "its object must be a reference, a value or a triple term", + )); } let ProcessedValue::Single(object) = process_value(value, sink)? else { return Err(invalid("its object must be a reference or a value")); diff --git a/fluree-graph-turtle/src/parser.rs b/fluree-graph-turtle/src/parser.rs index 2d85829c7b..1c500ee833 100644 --- a/fluree-graph-turtle/src/parser.rs +++ b/fluree-graph-turtle/src/parser.rs @@ -1282,8 +1282,6 @@ impl<'a, 'input, S: GraphSink> Parser<'a, 'input, S> { // annotation blocks `s p o {| … |}` / `s p o ~ reifier {| … |}`. // // Deliberately rejected with specific deferred errors: - // - `<<( … )>>` triple terms as values (no Fluree representation yet; - // the triple-term-as-value epic owns this), // - star constructs nested inside an annotation body // (annotation-of-annotation — mirrors the JSON-LD `@annotation` // lowering's v1 deferral), @@ -1325,8 +1323,7 @@ impl<'a, 'input, S: GraphSink> Parser<'a, 'input, S> { /// /// Grammar: `tripleTerm ::= '<<(' ttSubject predicate ttObject ')>>'`, /// `ttSubject ::= iri | BlankNode`, `ttObject ::= iri | BlankNode | - /// literal | tripleTerm`. A nested triple term in object position is a - /// value with no Fluree representation and keeps the deferred error. + /// literal | tripleTerm`. fn parse_reifies_triple_term(&mut self, reifier: TermId) -> Result<()> { self.with_nesting(|p| { p.check_star_allowed("triple term ('<<( … )>>')")?; @@ -1387,10 +1384,6 @@ impl<'a, 'input, S: GraphSink> Parser<'a, 'input, S> { self.current().kind ), )), - TokenKind::TripleTermStart => Err(TurtleError::parse( - self.current().start as usize, - "nested triple terms ('<<( … <<( … )>> )>>') are not supported", - )), _ => self.parse_object(), } } @@ -2524,17 +2517,25 @@ mod tests { } #[test] - fn star_rdf_reifies_nested_triple_term_still_deferred() { - let mut sink = StarSink::default(); - let err = parse( + fn nested_triple_terms_parse() { + let mut sink = GraphCollectorSink::new(); + parse( &format!( "{P}PREFIX rdf: \n\ - :r rdf:reifies <<( :s :p <<( :x :y 1 )>> )>> ." + :r rdf:reifies <<( :s :p <<( :x :y 1 )>> )>> .\n\ + :d :q <<( :s :p <<( :x :y :z )>> )>> ." ), &mut sink, ) - .expect_err("nested triple term is a value with no representation"); - assert!(err.to_string().contains("nested triple terms"), "{err}"); + .expect("nested triple terms parse"); + let graph = sink.into_graph(); + let reified = &graph.reifications()[0].triple; + // `:s :p <<( :x :y 1 )>>` is the reified triple. + assert!(matches!(&reified.o, Term::TripleTerm(_))); + assert!(graph.triples().iter().any(|t| matches!( + &t.o, + Term::TripleTerm(outer) if matches!(&outer[2], Term::TripleTerm(_)) + ))); } #[test] diff --git a/testsuite-sparql/src/rdf_handlers.rs b/testsuite-sparql/src/rdf_handlers.rs index 51c4c06bbb..3ecec9f735 100644 --- a/testsuite-sparql/src/rdf_handlers.rs +++ b/testsuite-sparql/src/rdf_handlers.rs @@ -27,8 +27,7 @@ use crate::files::read_file_to_string; use crate::manifest::Test; use crate::result_comparison::{are_results_isomorphic, format_results_diff}; use crate::result_format::{ - ir_term_to_rdf_term, RdfTerm, SparqlResults, Triple, REIFIES_OBJECT, REIFIES_PREDICATE, - REIFIES_SUBJECT, + ir_term_to_rdf_term, reification_triples, RdfTerm, SparqlResults, Triple, }; use crate::vocab::rdft; @@ -147,7 +146,7 @@ fn evaluate_eval(test: &Test) -> Result<()> { /// Convert a parsed graph to harness triples, re-expanding Fluree's /// `list_index` collection encoding into the `rdf:first` / `rdf:rest` chains /// the expected N-Triples spell out, and each reifier attachment into -/// `REIFIES_*` triples. +/// `rdf:reifies` triples. /// /// The Turtle parser emits `( a b )` in object position as one triple per /// element carrying `list_index` (the transaction layer stores lists that @@ -171,17 +170,12 @@ fn graph_to_rdf_triples(graph: &Graph) -> Vec { } } for r in graph.reifications() { - for (predicate, term) in [ - (REIFIES_SUBJECT, &r.triple.s), - (REIFIES_PREDICATE, &r.triple.p), - (REIFIES_OBJECT, &r.triple.o), - ] { - out.push(Triple { - subject: ir_term_to_rdf_term(&r.reifier), - predicate: RdfTerm::Iri(predicate.to_string()), - object: ir_term_to_rdf_term(term), - }); - } + out.extend(reification_triples( + ir_term_to_rdf_term(&r.reifier), + ir_term_to_rdf_term(&r.triple.s), + ir_term_to_rdf_term(&r.triple.p), + ir_term_to_rdf_term(&r.triple.o), + )); } let mut next_cell = 0usize; for ((s, p), mut items) in lists { diff --git a/testsuite-sparql/src/result_format.rs b/testsuite-sparql/src/result_format.rs index 1f50887b67..4caba52bf0 100644 --- a/testsuite-sparql/src/result_format.rs +++ b/testsuite-sparql/src/result_format.rs @@ -422,6 +422,27 @@ pub fn parse_srx(xml: &str) -> Result { }, } + /// An open ``: the position its next term fills, and its terms. + #[derive(Default)] + struct TripleFrame { + slot: usize, + parts: [Option; 3], + } + + /// Put a finished term in the innermost open triple, else the binding. + fn place( + term: RdfTerm, + frames: &mut [TripleFrame], + current_binding_name: &Option, + current_solution: &mut Option>, + ) { + if let Some(frame) = frames.last_mut() { + frame.parts[frame.slot] = Some(term); + } else if let (Some(name), Some(solution)) = (current_binding_name, current_solution) { + solution.insert(name.clone(), term); + } + } + /// Complete a finished element — on a real `Event::End`, or immediately /// for a self-closing `Event::Empty` (which emits NO matching End event, /// so its completion must never wait for one). Returns `Some` when the @@ -433,6 +454,7 @@ pub fn parse_srx(xml: &str) -> Result { current_binding_name: &mut Option, current_solution: &mut Option>, current_term: &mut Option, + frames: &mut Vec, ) -> Option { match local_name { b"result" => { @@ -445,41 +467,43 @@ pub fn parse_srx(xml: &str) -> Result { } b"uri" => { if let Some(TermKind::Uri) = current_term { - if let Some(name) = current_binding_name.as_ref() { - if let Some(solution) = current_solution.as_mut() { - solution.insert(name.clone(), RdfTerm::Iri(text_buf.to_string())); - } - } + let term = RdfTerm::Iri(text_buf.to_string()); + place(term, frames, current_binding_name, current_solution); } *current_term = None; } b"bnode" => { if let Some(TermKind::Bnode) = current_term { - if let Some(name) = current_binding_name.as_ref() { - if let Some(solution) = current_solution.as_mut() { - solution.insert(name.clone(), RdfTerm::BlankNode(text_buf.to_string())); - } - } + let term = RdfTerm::BlankNode(text_buf.to_string()); + place(term, frames, current_binding_name, current_solution); } *current_term = None; } b"literal" => { if let Some(TermKind::Literal { datatype, language }) = current_term.clone() { - if let Some(name) = current_binding_name.as_ref() { - if let Some(solution) = current_solution.as_mut() { - solution.insert( - name.clone(), - RdfTerm::Literal { - value: text_buf.to_string(), - datatype, - language, - }, - ); - } - } + let term = RdfTerm::Literal { + value: text_buf.to_string(), + datatype, + language, + }; + place(term, frames, current_binding_name, current_solution); } *current_term = None; } + b"triple" => { + if let Some(TripleFrame { + parts: [Some(subject), Some(predicate), Some(object)], + .. + }) = frames.pop() + { + let term = RdfTerm::Triple(Box::new(Triple { + subject, + predicate, + object, + })); + place(term, frames, current_binding_name, current_solution); + } + } b"boolean" => { let val = text_buf.trim(); return Some(SparqlResults::Boolean(val == "true" || val == "1")); @@ -492,6 +516,7 @@ pub fn parse_srx(xml: &str) -> Result { let mut current_term: Option = None; let mut text_buf = String::new(); let mut in_boolean = false; + let mut frames: Vec = Vec::new(); loop { let event = reader.read_event(); @@ -552,6 +577,16 @@ pub fn parse_srx(xml: &str) -> Result { in_boolean = true; text_buf.clear(); } + b"triple" => frames.push(TripleFrame::default()), + slot @ (b"subject" | b"predicate" | b"object") => { + if let Some(frame) = frames.last_mut() { + frame.slot = match slot { + b"subject" => 0, + b"predicate" => 1, + _ => 2, + }; + } + } _ => {} } // A self-closing element (``, ``, …) @@ -565,6 +600,7 @@ pub fn parse_srx(xml: &str) -> Result { &mut current_binding_name, &mut current_solution, &mut current_term, + &mut frames, ) { return Ok(result); } @@ -578,6 +614,7 @@ pub fn parse_srx(xml: &str) -> Result { &mut current_binding_name, &mut current_solution, &mut current_term, + &mut frames, ) { return Ok(result); } @@ -658,6 +695,14 @@ pub fn parse_srj(json: &str) -> Result { fn parse_srj_term(value: &serde_json::Value) -> Option { let obj = value.as_object()?; let term_type = obj.get("type")?.as_str()?; + if term_type == "triple" { + let triple = obj.get("value")?; + return Some(RdfTerm::Triple(Box::new(Triple { + subject: parse_srj_term(triple.get("subject")?)?, + predicate: parse_srj_term(triple.get("predicate")?)?, + object: parse_srj_term(triple.get("object")?)?, + }))); + } let val = obj.get("value")?.as_str()?; match term_type { @@ -715,32 +760,27 @@ pub fn parse_expected_graph(url: &str) -> Result> { Ok(graph_triples(&sink.into_graph())) } -/// Stand-in predicates that spell a reifier attachment as ordinary triples, -/// so the isomorphism check sees which triple each reifier names. -pub(crate) const REIFIES_SUBJECT: &str = "urn:fluree:testsuite:reifies-subject"; -pub(crate) const REIFIES_PREDICATE: &str = "urn:fluree:testsuite:reifies-predicate"; -pub(crate) const REIFIES_OBJECT: &str = "urn:fluree:testsuite:reifies-object"; - -/// `reifier`'s attachment to `(s, p, o)` as its three stand-in triples. +/// `reifier`'s attachment to `(s, p, o)`: `reifier rdf:reifies <<( s p o )>>`. pub(crate) fn reification_triples( reifier: RdfTerm, s: RdfTerm, p: RdfTerm, o: RdfTerm, -) -> [Triple; 3] { - [ - (REIFIES_SUBJECT, s), - (REIFIES_PREDICATE, p), - (REIFIES_OBJECT, o), - ] - .map(|(stand_in, term)| Triple { - subject: reifier.clone(), - predicate: RdfTerm::Iri(stand_in.to_string()), - object: term, - }) +) -> [Triple; 1] { + [Triple { + subject: reifier, + predicate: RdfTerm::Iri(RDF_REIFIES.to_string()), + object: RdfTerm::Triple(Box::new(Triple { + subject: s, + predicate: p, + object: o, + })), + }] } -/// A parsed graph's triples, with each reification as its stand-in triples. +const RDF_REIFIES: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#reifies"; + +/// A parsed graph's triples, with each reification as its `rdf:reifies` triple. fn graph_triples(graph: &IrGraph) -> Vec { let mut out: Vec = graph .iter() @@ -1198,7 +1238,7 @@ impl JsonLdContext { /// Expects a JSON-LD `@graph` array (or a single node object). Each node has /// `@id` as the subject; every other key is a predicate whose values are objects. /// Compact IRIs are expanded against the result's `@context`. A value's -/// `@annotation` and a node's `@reifies` become reification stand-in triples, +/// `@annotation` and a node's `@reifies` become `rdf:reifies` triples, /// as the expected graph's reifications do. pub fn fluree_construct_to_sparql_results(json: &serde_json::Value) -> Result { let ctx = JsonLdContext::parse(json); @@ -1329,6 +1369,17 @@ fn id_term(id: &str, ctx: &JsonLdContext) -> RdfTerm { /// and plain string/number values. fn json_ld_value_to_rdf_term(val: &serde_json::Value, ctx: &JsonLdContext) -> Option { if let Some(obj) = val.as_object() { + // Triple term: {"@id": {"@id": s, p: o}} + if let Some(node) = obj.get("@id").and_then(|v| v.as_object()) { + let subject = id_term(node.get("@id")?.as_str()?, ctx); + let (p, o) = node.iter().find(|(k, _)| !k.starts_with('@'))?; + let object = json_ld_value_to_rdf_term(json_values(o).first()?, ctx)?; + return Some(RdfTerm::Triple(Box::new(Triple { + subject, + predicate: RdfTerm::Iri(ctx.expand_vocab(p)), + object, + }))); + } // Node reference: {"@id": "http://..."} if let Some(id) = obj.get("@id").and_then(|v| v.as_str()) { return Some(match id.strip_prefix("_:") { diff --git a/testsuite-sparql/tests/registers/mod.rs b/testsuite-sparql/tests/registers/mod.rs index 9d26068046..9c9822559e 100644 --- a/testsuite-sparql/tests/registers/mod.rs +++ b/testsuite-sparql/tests/registers/mod.rs @@ -387,34 +387,10 @@ pub const SPARQL12_SYNTAX_TRIPLE_TERMS_POSITIVE: &[&str] = &[]; // entries double-counted across clusters); attribution below re-verified // empirically by unregistering and reading the harness's failure reasons. pub const SPARQL12_EVAL_TRIPLE_TERMS: &[&str] = &[ - // data load: qt:data nests a triple term in another - // (`<<( … <<( … )>> )>>`, or a reifier of a triple whose object is a - // triple term), which ingest refuses (9) - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#results-tripleterms-1j", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#results-tripleterms-1x", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#basic-8", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#basic-9", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#pattern-10", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#pattern-11", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#op-1", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#op-2", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#order-2", - // data load: blocked on TriG GRAPH-block parsing, orthogonal to star - // (D-8) — "expected subject, found 'GRAPH'" (3) - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#expr-1", + // update: a `GRAPH ?g` name binds a raw IRI, which an INSERT template + // cannot write as an object (2) "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#update-1", "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#update-2", - // SRX results: the harness reads no `` result term (1) - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#results-reifiedtriples-1x", - // results not isomorphic: the template reifies `_:r rdf:reifies <<( … )>>`, - // a nested triple term the result drops (1) - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#construct-3", - // SPARQL lowering: triple-term values in VALUES data not implemented (1) - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#expr-2", - // update-3: the `{| |}` INSERT DATA executes, but the expected post-update - // TriG-star graph carries reifier semantics the harness comparison does - // not model (expected 2 triples, got 0 in the default graph) (1) - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#update-3", ]; pub const SPARQL12_EXPRESSION: &[&str] = &[ @@ -578,18 +554,9 @@ pub const RDF11_TURTLE: &[&str] = &[ "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-turtle/manifest.ttl#turtle-syntax-ln-dots", ]; -pub const RDF12_TURTLE_SYNTAX: &[&str] = &[ - // nested triple terms (`<<( s p <<( … )>> )>>`) are deferred; ingest - // rejects them with the specific error (2) - "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-turtle/syntax#nt-ttl12-3", - "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-turtle/syntax#nt-ttl12-nested-1", -]; +pub const RDF12_TURTLE_SYNTAX: &[&str] = &[]; pub const RDF12_TURTLE_EVAL: &[&str] = &[ - // action rejected: nested triple terms (`<<( s p <<( … )>> )>>`) are - // deferred (2) - "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-turtle/eval#turtle12-tt-03", - "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-turtle/eval#turtle12-tt-04", // action rejected: annotation-of-annotation (a `{| |}` or `<< >>` inside an // annotation body) is the deferred v1 shape, mirroring JSON-LD (2) "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-turtle/eval#turtle12-annotation-04", From 96116f13632966978dee06d6f1d912cd2276e2d8 Mon Sep 17 00:00:00 2001 From: bplatz Date: Sat, 3 Oct 2026 14:56:27 -0400 Subject: [PATCH 62/92] fix(transact): INSERT templates write built and graph-name IRIs A WHERE clause hands a template a raw IRI when it binds a `GRAPH ?g` name or builds one with `IRI()` / `BIND(IRI(...))`. Flake generation refused a raw IRI in every template position, so `INSERT { ?s ex:source ?g } WHERE { GRAPH ?g { ... } }` and `INSERT { ?s ex:id ?new } WHERE { BIND(IRI(...) AS ?new) }` failed. A raw IRI is now encoded as a node like any other IRI; a raw blank-node label from a graph source is still refused, since it names no stored node. W3C harness: an expected-state `.trig` file (the SPARQL 1.2 update tests) names its graphs in GRAPH blocks; it was read as Turtle. update-1 and update-2 now pass, and the triple-term eval register is empty. --- fluree-db-api/tests/it_named_graphs.rs | 48 ++++++++++ fluree-db-api/tests/it_transact_update.rs | 36 ++++---- fluree-db-transact/src/generate/flakes.rs | 36 +++++--- testsuite-sparql/Cargo.lock | 1 + testsuite-sparql/Cargo.toml | 1 + testsuite-sparql/src/query_handler.rs | 24 +++-- testsuite-sparql/src/result_format.rs | 107 ++++++++++++++++++++++ testsuite-sparql/tests/registers/mod.rs | 7 +- 8 files changed, 218 insertions(+), 42 deletions(-) diff --git a/fluree-db-api/tests/it_named_graphs.rs b/fluree-db-api/tests/it_named_graphs.rs index fa858a1815..c5cadb1e21 100644 --- a/fluree-db-api/tests/it_named_graphs.rs +++ b/fluree-db-api/tests/it_named_graphs.rs @@ -537,6 +537,54 @@ async fn user_graph_iris(fluree: &fluree_db_api::Fluree, ledger_id: &str) -> Vec out } +/// A `GRAPH ?g` name is an IRI an INSERT template can write as a subject or +/// an object. +#[tokio::test] +async fn test_sparql_update_writes_graph_names_as_terms() { + let fluree = FlureeBuilder::memory() + .with_ledger_cache_config(LedgerManagerConfig::default()) + .build_memory(); + let ledger_id = "it/sparql-update-graph-name-terms:main"; + let ledger = genesis_ledger(&fluree, ledger_id); + let ledger = run_sparql_update( + &fluree, + ledger, + r#"INSERT DATA { + GRAPH { "x" . } + GRAPH { "y" . } + }"#, + ) + .await + .ledger; + run_sparql_update( + &fluree, + ledger, + r"INSERT { ?s ?g . ?g ?s } + WHERE { GRAPH ?g { ?s ?o } }", + ) + .await; + assert_eq!( + graph_values( + &fluree, + ledger_id, + "https://example.org/a", + "https://example.org/source" + ) + .await, + vec!["https://example.org/g/1"] + ); + assert_eq!( + graph_values( + &fluree, + ledger_id, + "https://example.org/g/2", + "https://example.org/holds" + ) + .await, + vec!["https://example.org/b"] + ); +} + #[tokio::test] async fn test_sparql_update_graph_variable_rewrites_in_place() { // #1513: DELETE/INSERT with `GRAPH ?g` templates rewrites each match in diff --git a/fluree-db-api/tests/it_transact_update.rs b/fluree-db-api/tests/it_transact_update.rs index 09189db70d..d2bfde1945 100644 --- a/fluree-db-api/tests/it_transact_update.rs +++ b/fluree-db-api/tests/it_transact_update.rs @@ -1126,13 +1126,12 @@ async fn update_where_bind_error_handling_runtime_type_mismatch() { ); } +/// `IRI("bad:thing")` names the absolute IRI `bad:thing` (scheme `bad`); a +/// template writes it as a node, like any IRI the WHERE clause builds. #[tokio::test] -async fn update_where_bind_error_handling_invalid_iri() { - // IRI("bad:thing") produces a Binding::Iri at the expression level, but - // the transact layer rejects raw IRIs that can't be resolved to a SID - // for flake generation. +async fn update_where_bind_iri_writes_the_iri() { let fluree = FlureeBuilder::memory().build_memory(); - let db0 = LedgerSnapshot::genesis("it/transact-update:error-invalid-iri"); + let db0 = LedgerSnapshot::genesis("it/transact-update:bind-iri"); let ledger0 = LedgerState::new(db0, Novelty::new(0)); let seeded = fluree @@ -1151,7 +1150,7 @@ async fn update_where_bind_error_handling_invalid_iri() { .await .expect("seed error fns"); - let run_err = fluree + let out = fluree .update( seeded.ledger, &json!({ @@ -1164,19 +1163,20 @@ async fn update_where_bind_error_handling_invalid_iri() { "values": ["?s", [{"@value":"ex:error","@type":"@id"}]] }), ) - .await; + .await + .expect("a built IRI is written"); - assert!( - run_err.is_err(), - "expected transact error for unresolvable IRI" - ); - if let Err(err) = run_err { - assert!( - err.to_string().contains("Raw IRI") - || err.to_string().contains("cannot be used as object"), - "unexpected error: {err}" - ); - } + let q = json!({ + "@context": ctx_ex(), + "select": "?e", + "where": {"id": "ex:error", "ex:error": "?e"} + }); + let got = support::query_jsonld(&fluree, &out.ledger, &q) + .await + .expect("query") + .to_jsonld(&out.ledger.snapshot) + .expect("to_jsonld"); + assert_eq!(got, json!(["bad:thing"])); } #[tokio::test] diff --git a/fluree-db-transact/src/generate/flakes.rs b/fluree-db-transact/src/generate/flakes.rs index fffb132ae1..255919a41c 100644 --- a/fluree-db-transact/src/generate/flakes.rs +++ b/fluree-db-transact/src/generate/flakes.rs @@ -366,9 +366,7 @@ impl<'a> FlakeGenerator<'a> { Binding::EncodedPid { .. } => Err(TransactError::InvalidTerm( "Subject must be a Sid; EncodedPid cannot be used as subject".to_string(), )), - Binding::Iri(_) => Err(TransactError::InvalidTerm( - "Raw IRI from graph source cannot be used as subject for flake generation".to_string(), - )), + Binding::Iri(iri) => self.raw_iri_sid(iri).map(Some), } } else { Ok(None) @@ -424,9 +422,7 @@ impl<'a> FlakeGenerator<'a> { Binding::EncodedPid { .. } => Err(TransactError::InvalidTerm( "Predicate must be a Sid; EncodedPid must be materialized before flake generation".to_string(), )), - Binding::Iri(_) => Err(TransactError::InvalidTerm( - "Raw IRI from graph source cannot be used as predicate for flake generation".to_string(), - )), + Binding::Iri(iri) => self.raw_iri_sid(iri).map(Some), } } else { Ok(None) @@ -470,9 +466,10 @@ impl<'a> FlakeGenerator<'a> { Binding::Sid { sid, .. } => { Ok((Some(FlakeValue::Ref(sid.clone())), Some(DT_ID.clone()))) } - Binding::IriMatch { primary_sid, .. } => { - Ok((Some(FlakeValue::Ref(primary_sid.clone())), Some(DT_ID.clone()))) - } + Binding::IriMatch { primary_sid, .. } => Ok(( + Some(FlakeValue::Ref(primary_sid.clone())), + Some(DT_ID.clone()), + )), Binding::Lit { val, dtc, .. } => { Ok((Some(val.clone()), Some(dtc.datatype().clone()))) } @@ -489,11 +486,15 @@ impl<'a> FlakeGenerator<'a> { Binding::Grouped(_) => Err(TransactError::InvalidTerm( "Object cannot be a grouped value (GROUP BY output)".to_string(), )), - Binding::Path { .. } | Binding::Rel(_) | Binding::List(_) | Binding::Map(_) => Err(TransactError::InvalidTerm( + Binding::Path { .. } + | Binding::Rel(_) + | Binding::List(_) + | Binding::Map(_) => Err(TransactError::InvalidTerm( "Object cannot be a path or list value".to_string(), )), - Binding::Iri(_) => Err(TransactError::InvalidTerm( - "Raw IRI from graph source cannot be used as object for flake generation".to_string(), + Binding::Iri(iri) => Ok(( + Some(FlakeValue::Ref(self.raw_iri_sid(iri)?)), + Some(DT_ID.clone()), )), } } else { @@ -517,6 +518,17 @@ impl<'a> FlakeGenerator<'a> { } } + /// A raw IRI binding — a `GRAPH ?g` name, or a graph source's IRI — as a + /// node. A raw blank-node label names no stored node. + fn raw_iri_sid(&mut self, iri: &str) -> Result { + if iri.starts_with("_:") { + return Err(TransactError::InvalidTerm(format!( + "a blank node from a graph source ({iri}) cannot be written" + ))); + } + Ok(self.ns_registry.sid_for_iri(iri)) + } + /// A triple term's positions, resolved as a flake's own are; `None` when /// one is unbound. fn resolve_triple_term( diff --git a/testsuite-sparql/Cargo.lock b/testsuite-sparql/Cargo.lock index 194921ec35..3c43d5a5e1 100644 --- a/testsuite-sparql/Cargo.lock +++ b/testsuite-sparql/Cargo.lock @@ -2434,6 +2434,7 @@ dependencies = [ "anyhow", "fluree-db-api", "fluree-db-sparql", + "fluree-db-transact", "fluree-graph-ir", "fluree-graph-turtle", "quick-xml", diff --git a/testsuite-sparql/Cargo.toml b/testsuite-sparql/Cargo.toml index 53d6c4b2bd..d397bacb65 100644 --- a/testsuite-sparql/Cargo.toml +++ b/testsuite-sparql/Cargo.toml @@ -14,6 +14,7 @@ publish = false # Fluree crates (for testing) fluree-db-api = { path = "../fluree-db-api", features = ["native"] } fluree-db-sparql = { path = "../fluree-db-sparql" } +fluree-db-transact = { path = "../fluree-db-transact" } fluree-graph-turtle = { path = "../fluree-graph-turtle" } fluree-graph-ir = { path = "../fluree-graph-ir" } diff --git a/testsuite-sparql/src/query_handler.rs b/testsuite-sparql/src/query_handler.rs index acb62898c1..ab2fbe2507 100644 --- a/testsuite-sparql/src/query_handler.rs +++ b/testsuite-sparql/src/query_handler.rs @@ -14,8 +14,9 @@ use crate::manifest::Test; use crate::rdfxml; use crate::result_comparison::{are_results_isomorphic, format_results_diff}; use crate::result_format::{ - fluree_construct_to_sparql_results, fluree_json_to_sparql_results, parse_expected_graph, - parse_expected_results, project_to_csv_space, RdfTerm, SparqlResults, Triple, + fluree_construct_to_sparql_results, fluree_json_to_sparql_results, parse_expected_dataset, + parse_expected_graph, parse_expected_results, project_to_csv_space, RdfTerm, SparqlResults, + Triple, }; use crate::subprocess::{run_in_subprocess, TestDescriptor}; @@ -560,10 +561,10 @@ pub async fn run_update_eval_test( let ledger = fetch_state(&fluree).await?; - // 3. Compare default graph state - let expected_default = match result_data_url { - Some(url) => parse_expected_graph(url)?, - None => Vec::new(), + // 3. Compare default graph state. A TriG result file also names graphs. + let (expected_default, result_dataset_graphs) = match result_data_url { + Some(url) => parse_expected_dataset(url)?, + None => (Vec::new(), Vec::new()), }; let actual_default = read_graph_triples(&fluree, &ledger, None).await?; compare_graph( @@ -586,6 +587,16 @@ pub async fn run_update_eval_test( Some(expected_url), )?; } + for (graph_name, expected) in &result_dataset_graphs { + let actual = read_graph_triples(&fluree, &ledger, Some(graph_name)).await?; + compare_graph( + test_id, + &format!("named graph <{graph_name}>"), + expected.clone(), + actual, + result_data_url, + )?; + } // 5. No unexpected non-empty named graphs. // @@ -611,6 +622,7 @@ pub async fn run_update_eval_test( let expected_names: std::collections::HashSet<&str> = result_graph_data .iter() .map(|(name, _)| name.as_str()) + .chain(result_dataset_graphs.iter().map(|(name, _)| name.as_str())) .collect(); for name in &list_named_graphs(&fluree, &ledger).await? { if expected_names.contains(name.as_str()) { diff --git a/testsuite-sparql/src/result_format.rs b/testsuite-sparql/src/result_format.rs index 4caba52bf0..94d44eb410 100644 --- a/testsuite-sparql/src/result_format.rs +++ b/testsuite-sparql/src/result_format.rs @@ -760,6 +760,113 @@ pub fn parse_expected_graph(url: &str) -> Result> { Ok(graph_triples(&sink.into_graph())) } +/// A dataset's default graph and its named graphs, by name. +pub type ExpectedDataset = (Vec, Vec<(String, Vec)>); + +/// An expected-state file's default graph and named graphs. A `.trig` file +/// names its graphs in `GRAPH` blocks; any other file is one default graph. +pub fn parse_expected_dataset(url: &str) -> Result { + use fluree_db_transact::{RawObject, RawTerm}; + if !url.ends_with(".trig") { + return Ok((parse_expected_graph(url)?, Vec::new())); + } + let content = + read_file_to_string(url).with_context(|| format!("Reading expected graph file: {url}"))?; + let with_base = format!("@base <{url}> .\n{content}"); + let phase1 = fluree_db_transact::parse_trig_phase1(&with_base) + .map_err(|e| anyhow::anyhow!("Parsing expected TriG {url}: {e}"))?; + let mut sink = GraphCollectorSink::new(); + parse_turtle(&phase1.turtle, &mut sink) + .with_context(|| format!("Parsing expected graph: {url}"))?; + let default = graph_triples(&sink.into_graph()); + + let mut named = Vec::new(); + for block in &phase1.named_graphs { + let node = |term: &RawTerm| -> Result { + Ok(match term { + RawTerm::Iri(iri) => match iri.strip_prefix("_:") { + Some(label) => RdfTerm::BlankNode(label.to_string()), + None => RdfTerm::Iri(iri.clone()), + }, + RawTerm::PrefixedName { prefix, local } => { + let ns = block + .prefixes + .get(prefix.as_str()) + .with_context(|| format!("undefined prefix {prefix}: in {url}"))?; + RdfTerm::Iri(format!("{ns}{local}")) + } + }) + }; + fn object(o: &RawObject, node: &dyn Fn(&RawTerm) -> Result) -> Result { + let typed = |value: String, dt: &str| RdfTerm::Literal { + value, + datatype: Some(format!("http://www.w3.org/2001/XMLSchema#{dt}")), + language: None, + }; + Ok(match o { + RawObject::Iri(iri) => node(&RawTerm::Iri(iri.clone()))?, + RawObject::PrefixedName { prefix, local } => node(&RawTerm::PrefixedName { + prefix: prefix.clone(), + local: local.clone(), + })?, + RawObject::String(s) => RdfTerm::Literal { + value: s.clone(), + datatype: None, + language: None, + }, + RawObject::Integer(n) => typed(n.to_string(), "integer"), + RawObject::Double(d) => typed(d.to_string(), "double"), + RawObject::Boolean(b) => typed(b.to_string(), "boolean"), + RawObject::TypedLiteral { value, datatype } => RdfTerm::Literal { + value: value.clone(), + datatype: Some(datatype.clone()), + language: None, + }, + RawObject::LangString { value, lang } => RdfTerm::Literal { + value: value.clone(), + datatype: None, + language: Some(lang.clone()), + }, + RawObject::TripleTerm { + subject, + predicate, + object: o, + } => RdfTerm::Triple(Box::new(Triple { + subject: node(subject)?, + predicate: node(predicate)?, + object: object(o, node)?, + })), + }) + } + let mut triples = Vec::new(); + for t in &block.triples { + let subject = node( + t.subject + .as_ref() + .context("named graph triple without subject")?, + )?; + let predicate = node(&t.predicate)?; + for o in &t.objects { + triples.push(Triple { + subject: subject.clone(), + predicate: predicate.clone(), + object: object(o, &node)?, + }); + } + } + for r in &block.reified { + triples.extend(reification_triples( + node(&r.reifier)?, + node(&r.subject)?, + node(&r.predicate)?, + object(&r.object, &node)?, + )); + } + named.push((block.iri.clone(), triples)); + } + Ok((default, named)) +} + /// `reifier`'s attachment to `(s, p, o)`: `reifier rdf:reifies <<( s p o )>>`. pub(crate) fn reification_triples( reifier: RdfTerm, diff --git a/testsuite-sparql/tests/registers/mod.rs b/testsuite-sparql/tests/registers/mod.rs index 9c9822559e..eb8ed40c37 100644 --- a/testsuite-sparql/tests/registers/mod.rs +++ b/testsuite-sparql/tests/registers/mod.rs @@ -386,12 +386,7 @@ pub const SPARQL12_SYNTAX_TRIPLE_TERMS_POSITIVE: &[&str] = &[]; // Each test appears in EXACTLY ONE reason cluster (PR-1454 review found 10 // entries double-counted across clusters); attribution below re-verified // empirically by unregistering and reading the harness's failure reasons. -pub const SPARQL12_EVAL_TRIPLE_TERMS: &[&str] = &[ - // update: a `GRAPH ?g` name binds a raw IRI, which an INSERT template - // cannot write as an object (2) - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#update-1", - "https://w3c.github.io/rdf-tests/sparql/sparql12/eval-triple-terms/manifest#update-2", -]; +pub const SPARQL12_EVAL_TRIPLE_TERMS: &[&str] = &[]; pub const SPARQL12_EXPRESSION: &[&str] = &[ // not-not: the D-EBV fix makes !!?v unbind for the language-tagged, From 68509b6fe75f567cb39398cfa5e3927d60f6f01e Mon Sep 17 00:00:00 2001 From: bplatz Date: Sat, 3 Oct 2026 19:23:57 -0400 Subject: [PATCH 63/92] perf(query): an annotated edge checks its triple instead of joining it Annotation syntax (SPARQL `{| |}`, JSON-LD `@annotation`, Cypher relationship properties) joins its edge so that a reifier of an unasserted triple does not match. The term's components already bind every position of that edge, so the join only checks existence, but as a joined triple the planner drove from it: LDBC IC7 hash-joined a scan of every LIKES edge before the term lookup, 67 ms -> 480 ms (IC5 1.5x). The edge is now an EXISTS, which the planner places after the components bind it and runs as a semijoin. Locally (debug build, LDBC SF0.1) IC7 is 2-8x faster than with the joined edge, IC5 within 4% of no check at all; answers match the golden results. --- fluree-db-query/src/execute/where_plan.rs | 31 ++++++++++++++++++----- 1 file changed, 25 insertions(+), 6 deletions(-) diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index ca6ba24a06..2fde422c45 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -139,15 +139,17 @@ fn expand_one_into(pattern: Pattern, out: &mut Vec, one_graph: bool) { // is in scope and the chain joins its enclosing block. // // The order is the planner's tie-break: the body and the link are - // probes by reifier, the term's components then decode from the - // bound term, and the edge is a bound existence probe last. Estimates still decide where they differ (a + // probes by reifier, and the term's components then decode from + // the bound term. Estimates still decide where they differ (a // constant subject anchors the components through the term // dictionary). // - // Annotation syntax asserts its triple (RDF 1.2), so the edge is - // joined: a reifier may reify a triple that is not asserted, and - // such a link must not match `s p o {| … |}`. - let base_edge = Pattern::Triple(edge.clone()); + // Annotation syntax asserts its triple (RDF 1.2): a reifier may + // reify a triple that is not asserted, and such a link must not + // match `s p o {| … |}`. The components bind every position of the + // edge, so the edge only checks existence; as a joined triple the + // planner drove from it, scanning the whole predicate (LDBC IC7). + let base_edge = Pattern::Exists(vec![Pattern::Triple(edge.clone())]); let mut chain: Vec = Vec::new(); // 1. Body patterns (recursively expanded so nested annotations — @@ -6607,6 +6609,23 @@ mod tests { .count() } + /// The annotated edge only checks that its triple is asserted: the term's + /// components bind every position, and as a joined triple the planner + /// drove from it. + #[test] + fn annotated_edge_is_an_existence_check() { + let expanded = expand_edge_annotation_patterns(&[annotated_hop(0, 1, 2, 3)]); + let chain = unwrap_default_graph_source(&expanded[0]); + let knows = |p: &Pattern| matches!(p, Pattern::Triple(tp) if matches!(&tp.p, Ref::Sid(sid) if &*sid.name == "knows")); + assert!(!chain.iter().any(knows), "no joined edge: {chain:?}"); + assert!( + chain + .iter() + .any(|p| matches!(p, Pattern::Exists(inner) if matches!(inner.as_slice(), [e] if knows(e)))), + "the edge is an EXISTS: {chain:?}" + ); + } + #[test] fn sink_copies_annotation_body_filter_inside_the_wrapper() { let patterns = vec![annotated_hop(0, 1, 2, 3), gt(3, 0.97)]; From 665f89dfc157e435cc2d8f085d85959e01a2f5bd Mon Sep 17 00:00:00 2001 From: bplatz Date: Sat, 3 Oct 2026 21:42:26 -0400 Subject: [PATCH 64/92] perf(query): link probes bound by a term handle run in bulk A nested-loop join whose right side is `?r rdf:reifies ?t`, with `?t` bound to a term handle by the left side, opened one scan per left row. Each scan decoded the handle to a triple-term value and then re-derived the handle from that value through the subject and term dictionaries. The bound-object lane already probes OPST once per batch for ref keys, and now takes triple-term handles too. One flush probes one object type; a batch that mixes refs and handles flushes at each type change. The novelty merge filters overlay ops to the flush's type, and overlay ops already key terms by the same handle. Annotation benchmark (61M triples), local: S21 9.67s -> 2.31s, S17 1.58s -> 1.03s, S18 1.93s -> 1.16s (main: 2.48s, 1.37s, 1.50s). --- fluree-db-api/tests/it_triple_term_links.rs | 137 ++++++++++++++++++++ fluree-db-query/src/fast_path_common.rs | 10 +- fluree-db-query/src/join.rs | 60 ++++++--- 3 files changed, 184 insertions(+), 23 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 97e8facd14..52c41ae1ac 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -3262,3 +3262,140 @@ async fn every_write_path_takes_nested_triple_terms() { assert_eq!(named, strings(&[&["ex:o"]]), "[{label}] in the named graph"); } } + +/// A link probe bound by a term handle — from a reified pattern's term +/// lookup, or from a term-valued edge — runs as one bulk object probe per +/// batch, and merges novelty links: a new link on an indexed term, a term +/// only novelty holds, and a retracted link. A batch mixing ref and term +/// keys probes each under its own type, under a predicate that holds both. +#[tokio::test(flavor = "current_thread")] +async fn term_bound_links_probe_in_bulk() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/triple-term-links:bulk-probe"; + let turtle = + |body: &str| format!("VERSION \"1.2\"\n@prefix ex: .\n{body}\n"); + // Enough other links that the anchored term, not the link scan, leads. + let padding: String = (0..50) + .map(|i| format!("ex:p{i} ex:partOf ex:q{i} {{| ex:source ex:pad |}} .\n")) + .collect(); + fluree + .insert_turtle( + support::genesis_ledger(&fluree, ledger_id), + &turtle(&format!( + "<< ex:s1 ex:partOf ex:anchor ~ ex:r1 >> ex:source ex:d1 .\n\ + << ex:s2 ex:partOf ex:anchor ~ ex:r2 >> ex:source ex:d2 .\n\ + << ex:s3 ex:partOf ex:anchor ~ ex:r3 >> ex:source ex:d3 .\n\ + << ex:s3 ex:partOf ex:other ~ ex:r4 >> ex:source ex:d4 .\n\ + ex:doc1 ex:mentions <<( ex:s1 ex:partOf ex:anchor )>> .\n\ + ex:doc2 ex:mentions ex:r1 .\n\ + ex:doc3 ex:mentions <<( ex:s3 ex:partOf ex:anchor )>> .\n\ + ex:a1 ex:cites ex:r1 .\n\ + ex:a2 ex:cites <<( ex:s1 ex:partOf ex:anchor )>> .\n\ + ex:a3 ex:cites <<( ex:s3 ex:partOf ex:anchor )>> .\n{padding}", + )), + ) + .await + .expect("base"); + fluree + .reindex(ledger_id, fluree_db_api::ReindexOptions::default()) + .await + .expect("reindex base"); + let ledger = fluree.ledger(ledger_id).await.expect("reload"); + + let queries = [ + "SELECT ?s ?src WHERE { << ?s ex:partOf ex:anchor >> ex:source ?src } ORDER BY ?s ?src", + "SELECT ?doc ?r WHERE { ?doc ex:mentions ?t . ?r rdf:reifies ?t } ORDER BY ?doc ?r", + "SELECT ?doc ?a WHERE { ?doc ex:mentions ?t . ?a ex:cites ?t } ORDER BY ?doc ?a", + ]; + // Register the span callsites before this thread's subscriber reads them. + run_link_query(&fluree, &ledger, queries[0].to_string()).await; + let (store, _guard) = support::span_capture::init_test_tracing(); + tracing::callsite::rebuild_interest_cache(); + let answers = |ledger: LedgerState, novelty: bool| { + let (fluree, store) = (&fluree, &store); + async move { + let mut answers = Vec::new(); + for q in queries { + let before = store.find_spans("join_flush_batched_object_binary").len(); + answers.push(run_link_query(fluree, &ledger, q.to_string()).await); + assert!( + store.find_spans("join_flush_batched_object_binary").len() > before, + "{q}: the link probe should run in bulk" + ); + } + if novelty { + assert!( + store.has_event("join batched object flush merged novelty overlay"), + "the bulk probe should merge novelty links" + ); + } + answers + } + }; + + assert_eq!( + answers(ledger, false).await, + vec![ + strings(&[ + &["ex:s1", "ex:d1"], + &["ex:s2", "ex:d2"], + &["ex:s3", "ex:d3"] + ]), + strings(&[&["ex:doc1", "ex:r1"], &["ex:doc3", "ex:r3"]]), + strings(&[ + &["ex:doc1", "ex:a2"], + &["ex:doc2", "ex:a1"], + &["ex:doc3", "ex:a3"] + ]), + ] + ); + + fluree + .insert_turtle( + fluree.ledger(ledger_id).await.expect("reload"), + &turtle( + "<< ex:s1 ex:partOf ex:anchor ~ ex:r1b >> ex:source ex:d1b .\n\ + << ex:s9 ex:partOf ex:anchor ~ ex:r9 >> ex:source ex:d9 .", + ), + ) + .await + .expect("novelty links"); + fluree + .graph(ledger_id) + .transact() + .sparql_update( + "PREFIX ex: \n\ + PREFIX rdf: \n\ + DELETE DATA { ex:r2 rdf:reifies <<( ex:s2 ex:partOf ex:anchor )>> }", + ) + .commit() + .await + .expect("retract a link"); + let expected = vec![ + strings(&[ + &["ex:s1", "ex:d1"], + &["ex:s1", "ex:d1b"], + &["ex:s3", "ex:d3"], + &["ex:s9", "ex:d9"], + ]), + strings(&[ + &["ex:doc1", "ex:r1"], + &["ex:doc1", "ex:r1b"], + &["ex:doc3", "ex:r3"], + ]), + strings(&[ + &["ex:doc1", "ex:a2"], + &["ex:doc2", "ex:a1"], + &["ex:doc3", "ex:a3"], + ]), + ]; + let ledger = fluree.ledger(ledger_id).await.expect("reload"); + assert_eq!(answers(ledger, true).await, expected, "under novelty"); + + fluree + .reindex(ledger_id, fluree_db_api::ReindexOptions::default()) + .await + .expect("reindex"); + let ledger = fluree.ledger(ledger_id).await.expect("reload"); + assert_eq!(answers(ledger, false).await, expected, "after reindex"); +} diff --git a/fluree-db-query/src/fast_path_common.rs b/fluree-db-query/src/fast_path_common.rs index 51fa14ca7a..f33b7ae72e 100644 --- a/fluree-db-query/src/fast_path_common.rs +++ b/fluree-db-query/src/fast_path_common.rs @@ -2479,13 +2479,15 @@ pub struct ObjectProbeOps { impl ObjectProbeOps { /// Filter `ops` (one predicate's resolved ops, any sort order) to the - /// `IRI_REF` subset and index it object-major. Returns `None` when no op + /// `o_type` subset and index it object-major. Returns `None` when no op /// can affect the lane — callers then run their unmodified scan. - pub fn new(ops: &[fluree_db_binary_index::read::types::OverlayOp]) -> Option { - let iri_ref = OType::IRI_REF.as_u16(); + pub fn new( + ops: &[fluree_db_binary_index::read::types::OverlayOp], + o_type: u16, + ) -> Option { let mut subset: Vec = ops .iter() - .filter(|o| o.o_type == iri_ref) + .filter(|o| o.o_type == o_type) .map(|o| ObjectProbeOp { o_key: o.o_key, s_id: o.s_id, diff --git a/fluree-db-query/src/join.rs b/fluree-db-query/src/join.rs index d06d3382a3..483d2be9c8 100644 --- a/fluree-db-query/src/join.rs +++ b/fluree-db-query/src/join.rs @@ -26,7 +26,9 @@ use crate::var_registry::VarId; use async_trait::async_trait; use fluree_db_binary_index::{BinaryGraphView, BinaryIndexStore}; use fluree_db_core::clock::Instant; +use fluree_db_core::o_type::OType; use fluree_db_core::subject_id::SubjectId; +use fluree_db_core::value_id::ObjKind; use fluree_db_core::{DatatypeDictId, GraphId, IndexType, ObjectBounds, Sid, BATCHED_JOIN_SIZE}; use rustc_hash::{FxBuildHasher, FxHashMap}; use std::collections::{HashSet, VecDeque}; @@ -454,8 +456,8 @@ fn is_batched_eligible( /// - subject is a new unbound variable (no BindInstruction for Subject) /// - no object bounds or datatype/language constraints /// -/// This enables scanning OPST in bulk for a set of bound ref objects rather than -/// opening one scan per left row. +/// This enables scanning OPST in bulk for a set of bound ref (or triple-term) +/// objects rather than opening one scan per left row. fn is_batched_object_eligible( bind_instructions: &[BindInstruction], right_pattern: &TriplePattern, @@ -602,6 +604,9 @@ pub struct NestedLoopJoinOperator { /// Accumulated entries for batched processing: (stored_batch_idx, row_idx, subject_s_id) /// Stores the raw s_id directly to avoid dictionary round-trips with EncodedSid. batched_accumulator: Vec<(usize, usize, u64)>, + /// Object lane: the `OType` every accumulated key carries. Refs and + /// triple-term handles never share a flush. + batched_object_o_type: u16, /// Accumulator size that triggers a flush. `BATCHED_JOIN_SIZE` throughout /// unless `set_row_budget` starts it near the budget, so a small `LIMIT` /// doesn't buffer ~100k left rows before producing anything. Each flush @@ -901,6 +906,7 @@ impl NestedLoopJoinOperator { batched_predicate, batched_overlay_mode: ProbeLanePlan::Clean, batched_accumulator: Vec::new(), + batched_object_o_type: OType::IRI_REF.as_u16(), flush_schedule: FlushSchedule::fixed(BATCHED_JOIN_SIZE), stored_left_batches: Vec::new(), batched_output: VecDeque::new(), @@ -1552,7 +1558,7 @@ impl Operator for NestedLoopJoinOperator { self.subject_left_col.unwrap() }; - let resolved: Option = { + let resolved: Option<(u16, u64)> = { let left_batch = self.current_left_batch.as_ref().unwrap(); let store = ctx.binary_store.as_deref(); // Persisted reverse dict first, then DictNovelty: a subject @@ -1577,10 +1583,21 @@ impl Operator for NestedLoopJoinOperator { }) }) }; + let iri_ref = OType::IRI_REF.as_u16(); match left_batch.get_by_col(left_row, left_col) { - Binding::EncodedSid { s_id, .. } => Some(*s_id), - Binding::Sid { sid, .. } => resolve_subject(sid), - Binding::IriMatch { primary_sid, .. } => resolve_subject(primary_sid), + Binding::EncodedSid { s_id, .. } => Some((iri_ref, *s_id)), + Binding::Sid { sid, .. } => resolve_subject(sid).map(|k| (iri_ref, k)), + Binding::IriMatch { primary_sid, .. } => { + resolve_subject(primary_sid).map(|k| (iri_ref, k)) + } + // A term handle keys OPST like a ref; per row it would + // decode the term only for the scan to re-derive it. + Binding::EncodedLit { o_kind, o_key, .. } + if self.batched_object_eligible + && *o_kind == ObjKind::TRIPLE_TERM.as_u8() => + { + Some((OType::TRIPLE_TERM.as_u16(), *o_key)) + } Binding::Unbound => None, _ => { // For subject/predicate bindings we already screened invalid types. @@ -1590,7 +1607,15 @@ impl Operator for NestedLoopJoinOperator { } }; - if let Some(key) = resolved { + if let Some((o_type, key)) = resolved { + if o_type != self.batched_object_o_type { + if !self.batched_accumulator.is_empty() { + ctx.check_cancelled()?; + self.flush_batched_accumulator_for_ctx(ctx).await?; + ctx.check_cancelled()?; + } + self.batched_object_o_type = o_type; + } let batch_idx = self.ensure_current_batch_stored(); self.batched_accumulator.push((batch_idx, left_row, key)); if self.batched_accumulator.len() >= self.flush_schedule.size() { @@ -1954,7 +1979,7 @@ impl NestedLoopJoinOperator { /// Decide how the batched lanes handle the active overlay this call. /// /// The subject-probe and exists lanes merge overlay ops per probed - /// subject; the object (OPST) lane merges its `IRI_REF` subset per probed + /// subject; the object (OPST) lane merges its key type's subset per probed /// object. Decline cases route to the overlay-correct per-row fallback /// BEFORE any accumulation, so a flush never reroutes mid-stream. fn compute_batched_overlay_mode(&self, ctx: &ExecutionContext<'_>) -> Result { @@ -2094,8 +2119,6 @@ impl NestedLoopJoinOperator { mut probe_ops: Option<&mut ProbeOps>, on_match: &mut dyn FnMut(&[usize], &Binding) -> Result<()>, ) -> Result<()> { - use fluree_db_core::o_type::OType; - let scan_start = Instant::now(); let mut leaflets_scanned: u64 = 0; @@ -2694,7 +2717,6 @@ impl NestedLoopJoinOperator { cmp_v2_for_order, read_ordered_key_v2, RunRecordV2, }; use fluree_db_binary_index::RunSortOrder; - use fluree_db_core::o_type::OType; if self.batched_accumulator.is_empty() { return Ok(()); @@ -2731,7 +2753,7 @@ impl NestedLoopJoinOperator { // One reconciler per flush (see `flush_batched_accumulator_binary`). let mut probe_ops = match &self.batched_overlay_mode { - ProbeLanePlan::Merge(ops) => ObjectProbeOps::new(ops), + ProbeLanePlan::Merge(ops) => ObjectProbeOps::new(ops, self.batched_object_o_type), _ => None, }; @@ -2743,7 +2765,7 @@ impl NestedLoopJoinOperator { // We build a set of leaf indices that contain any of our object IDs, then scan // those leaves. This avoids re-opening and re-decoding leaflets once per object // (which is the dominant cost in `BinaryCursor`-per-object approaches). - let iri_ref = OType::IRI_REF.as_u16(); + let o_type = self.batched_object_o_type; let cmp = cmp_v2_for_order(RunSortOrder::Opst); let mut objs: Vec = o_to_accum.keys().copied().collect(); @@ -2763,7 +2785,7 @@ impl NestedLoopJoinOperator { p_id, t: 0, o_i: 0, - o_type: iri_ref, + o_type, g_id: ctx.binary_g_id, }; let max_key = RunRecordV2 { @@ -2772,7 +2794,7 @@ impl NestedLoopJoinOperator { p_id, t: u32::MAX, o_i: u32::MAX, - o_type: iri_ref, + o_type, g_id: ctx.binary_g_id, }; let r = branch.find_leaves_in_range(&min_key, &max_key, cmp); @@ -2799,7 +2821,7 @@ impl NestedLoopJoinOperator { if entry.p_const.is_some() && entry.p_const != Some(p_id) { continue; } - if entry.o_type_const.is_some() && entry.o_type_const != Some(iri_ref) { + if entry.o_type_const.is_some() && entry.o_type_const != Some(o_type) { continue; } @@ -2858,7 +2880,7 @@ impl NestedLoopJoinOperator { // which is extremely rare for OPST; fall back to row-scan in that case. for row in 0..batch.row_count { let ot = batch.o_type.get_or(row, 0); - if ot != iri_ref { + if ot != o_type { continue; } let pid = batch.p_id.get_or(row, 0); @@ -2895,7 +2917,7 @@ impl NestedLoopJoinOperator { // Fast path: if o_type/p_id are const and already filtered by leaflet // metadata, we can skip per-row checks. - let ot_const_ok = batch.o_type.is_const() && batch.o_type.get_or(0, 0) == iri_ref; + let ot_const_ok = batch.o_type.is_const() && batch.o_type.get_or(0, 0) == o_type; let pid_const_ok = batch.p_id.is_const() && batch.p_id.get_or(0, 0) == p_id; // Start scanning at the first possible match within this leaflet. @@ -2934,7 +2956,7 @@ impl NestedLoopJoinOperator { for r in row..run_end { if !ot_const_ok { let ot = batch.o_type.get_or(r, 0); - if ot != iri_ref { + if ot != o_type { continue; } } From 0de62a331e9ecba0e7e239113bc95a5368bf6484 Mon Sep 17 00:00:00 2001 From: bplatz Date: Mon, 14 Sep 2026 16:38:53 -0400 Subject: [PATCH 65/92] perf(query): batch wildcard predicate joins with shared streaming lookups Ported from perf/batched-wildcard-joins onto the RDF 1.2 link model. - The lane admission runs the shared history and policy gates first. - Wildcard rows hide only the legacy `f:reifies*` predicates, so `rdf:reifies` links are ordinary rows. - Triple-term keys keep to the fixed-predicate object lane, since the wildcard cursor reads refs only. --- fluree-db-api/Cargo.toml | 6 + .../tests/it_wildcard_batched_join.rs | 562 ++++++++++++++++++ .../src/read/batched_lookup.rs | 381 ++++++------ fluree-db-query/src/execute/operator_tree.rs | 3 +- fluree-db-query/src/join.rs | 55 +- fluree-db-query/src/join/wildcard.rs | 201 +++++++ 6 files changed, 1007 insertions(+), 201 deletions(-) create mode 100644 fluree-db-api/tests/it_wildcard_batched_join.rs create mode 100644 fluree-db-query/src/join/wildcard.rs diff --git a/fluree-db-api/Cargo.toml b/fluree-db-api/Cargo.toml index 61bef03d55..0e2382156c 100644 --- a/fluree-db-api/Cargo.toml +++ b/fluree-db-api/Cargo.toml @@ -187,6 +187,12 @@ path = "tests/grp_import.rs" name = "grp_triple_terms" path = "tests/grp_triple_terms.rs" +# Standalone: wildcard join routing and the process-global fast-path switch. +[[test]] +name = "it_wildcard_batched_join" +path = "tests/it_wildcard_batched_join.rs" +required-features = ["native"] + # Standalone (NOT in grp_import): re-execs itself and lowers RLIMIT_NOFILE # hard+soft in the child, which must not share a process with other tests. [[test]] diff --git a/fluree-db-api/tests/it_wildcard_batched_join.rs b/fluree-db-api/tests/it_wildcard_batched_join.rs new file mode 100644 index 0000000000..028d0ccb86 --- /dev/null +++ b/fluree-db-api/tests/it_wildcard_batched_join.rs @@ -0,0 +1,562 @@ +//! Standalone: routing capture and the process-global fast-path switch. +#![cfg(feature = "native")] +mod support; + +use fluree_db_api::{set_fast_paths_disabled, Fluree, FlureeBuilder, ReindexOptions}; +use fluree_db_ledger::LedgerState; +use serde_json::{json, Value}; + +const PREFIX: &str = "PREFIX ex: "; +const EVENT: &str = "batched wildcard join engaged"; + +#[derive(Clone, Copy, Debug)] +enum Routing { + MustFire, + MustNotFire, +} + +struct Reset; +impl Drop for Reset { + fn drop(&mut self) { + set_fast_paths_disabled(false); + } +} + +async fn check( + fluree: &Fluree, + ledger: &LedgerState, + query: &str, + expected_len: usize, + routing: Routing, +) -> Value { + let query = format!("{PREFIX}{query}"); + let (spans, _guard) = support::span_capture::init_test_tracing(); + set_fast_paths_disabled(false); + let fast = support::query_sparql(fluree, ledger, &query) + .await + .expect(&query) + .to_sparql_json(&ledger.snapshot) + .unwrap(); + let fast_rows = normalized(&fast); + assert_eq!(fast_rows.len(), expected_len, "{query}: {fast}"); + let fired = spans.has_event(EVENT); + assert_eq!( + fired, + matches!(routing, Routing::MustFire), + "{routing:?}: {query}" + ); + drop(_guard); + let (spans, _guard) = support::span_capture::init_test_tracing(); + set_fast_paths_disabled(true); + let generic = support::query_sparql(fluree, ledger, &query) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + assert_eq!(fast_rows, normalized(&generic), "{query}"); + if query.contains("ORDER BY") { + assert_eq!( + fast["results"]["bindings"], generic["results"]["bindings"], + "ordered output" + ); + } + assert!(!spans.has_event(EVENT), "disabled lane fired: {query}"); + set_fast_paths_disabled(false); + fast +} + +fn normalized(result: &Value) -> Vec { + let mut rows: Vec<_> = result["results"]["bindings"] + .as_array() + .unwrap() + .iter() + .map(|r| serde_json::to_string(r).unwrap()) + .collect(); + rows.sort(); + rows +} + +#[tokio::test(flavor = "current_thread")] +async fn wildcard_joins_preserve_facts_multiplicity_and_fallbacks() { + assert!(std::env::var_os("FLUREE_DISABLE_QUERY_FAST_PATHS").is_none()); + let _reset = Reset; + let dir = tempfile::tempdir().unwrap(); + let fluree = FlureeBuilder::file(dir.path().to_string_lossy().to_string()) + .build() + .unwrap(); + let alias = "wildcard/joins:main"; + let ledger = fluree.create_ledger(alias).await.unwrap(); + fluree.insert(ledger, &json!({ + "@context": {"ex":"http://example.org/"}, + "@graph": [ + {"@id":"ex:a", "ex:name":"A", + "ex:edge":{"@id":"ex:b", "@annotation":{"@id":"ex:ann", "ex:source":"paper"}}, + "ex:label":[{"@value":"bonjour", "@language":"fr"}, {"@value":"hello", "@language":"en"}], + "ex:items":{"@list":[10,10,20]}}, + {"@id":"ex:b", "ex:name":"B", "ex:owner":{"@id":"ex:a"}}, + {"@id":"ex:c", "ex:edge":{"@id":"ex:a"}, "ex:other":{"@id":"ex:a"}} + ] + })).await.unwrap(); + fluree + .reindex(alias, ReindexOptions::default()) + .await + .unwrap(); + let ledger = fluree.ledger(alias).await.unwrap(); + use Routing::*; + let outgoing = "SELECT ?s ?p ?o WHERE { VALUES ?s { ex:a ex:a ex:b } ?s ?p ?o }"; + let rows = check(&fluree, &ledger, outgoing, 16, MustFire).await; + // Two equal list values at different indices must survive, for each of + // the two duplicate driving rows. Language tags must survive too. + let bindings = rows["results"]["bindings"].as_array().unwrap(); + assert_eq!( + bindings + .iter() + .filter(|r| r["p"]["value"] == "http://example.org/items" && r["o"]["value"] == "10") + .count(), + 4 + ); + assert_eq!( + bindings + .iter() + .filter(|r| r["o"]["xml:lang"] == "fr") + .count(), + 2 + ); + check( + &fluree, + &ledger, + "SELECT ?s ?p ?o WHERE { VALUES ?o { ex:a ex:a ex:b } ?s ?p ?o }", + 7, + MustFire, + ) + .await; + check( + &fluree, + &ledger, + "SELECT ?s ?p WHERE { VALUES ?o { ex:c ex:c } ?s ?p ?o }", + 0, + MustFire, + ) + .await; + check( + &fluree, + &ledger, + "SELECT ?s WHERE { VALUES ?s { ex:a ex:a ex:b } ?s ?p ?o }", + 16, + MustFire, + ) + .await; + check(&fluree, &ledger, + "SELECT ?p ?o WHERE { VALUES ?s { ex:a ex:b } ?s ?p ?o FILTER(STR(?p) = 'http://example.org/name') }", 2, MustFire).await; + check( + &fluree, + &ledger, + "SELECT ?p ?o WHERE { VALUES ?s { ex:a ex:b } ?s ?p ?o FILTER(?s = ex:b) }", + 2, + MustFire, + ) + .await; + check( + &fluree, + &ledger, + "SELECT ?s ?p ?o WHERE { VALUES ?s { ex:a ex:b } ?s ?p ?o } ORDER BY ?s ?p ?o LIMIT 3", + 3, + MustFire, + ) + .await; + check( + &fluree, + &ledger, + "SELECT ?p ?o WHERE { VALUES ?s { ex:a } ?s ?p ?s }", + 0, + MustNotFire, + ) + .await; + check( + &fluree, + &ledger, + "SELECT ?p WHERE { VALUES ?s { ex:a } ?s ?p ?p }", + 0, + MustNotFire, + ) + .await; + check( + &fluree, + &ledger, + "SELECT ?o WHERE { VALUES (?s ?p) { (ex:a ex:name) } ?s ?p ?o }", + 1, + MustNotFire, + ) + .await; + check( + &fluree, + &ledger, + "SELECT ?s ?p WHERE { VALUES ?o { 'A' } ?s ?p ?o }", + 1, + MustNotFire, + ) + .await; + check( + &fluree, + &ledger, + "SELECT ?s ?p ?o WHERE { VALUES ?s { ex:a UNDEF ex:b } ?s ?p ?o }", + 22, // the UNDEF row's full scan includes the annotation's rdf:reifies link + MustFire, + ) + .await; + // A single matched fact fans out across several output batches. Resuming + // must retain the duplicate-driver position as well as the cursor position. + let many = format!( + "SELECT ?p ?o WHERE {{ VALUES ?s {{ {} }} ?s ?p ?o }}", + "ex:a ".repeat(2100) + ); + check(&fluree, &ledger, &many, 14700, MustFire).await; + check( + &fluree, + &ledger, + "SELECT ?x WHERE { VALUES ?s { ex:a ex:b } ?s ?p ?o BIND(CONCAT(STR(?s), STR(?p)) AS ?x) }", + 9, + MustFire, + ) + .await; + // Per driving row: the reifier's `rdf:reifies` link and its annotation, + // and no internal encoding predicate. + let annotation = check( + &fluree, + &ledger, + "SELECT ?d ?e WHERE { VALUES ?s { ex:a ex:a } << ?s ex:edge ex:b >> ?d ?e }", + 4, + MustFire, + ) + .await; + let mut preds: Vec<&str> = annotation["results"]["bindings"] + .as_array() + .unwrap() + .iter() + .map(|r| r["d"]["value"].as_str().unwrap()) + .collect(); + preds.sort_unstable(); + assert_eq!( + preds, + [ + "http://example.org/source", + "http://example.org/source", + "http://www.w3.org/1999/02/22-rdf-syntax-ns#reifies", + "http://www.w3.org/1999/02/22-rdf-syntax-ns#reifies", + ] + ); + + // JSON-LD twin uses the shared IR and must engage the same lane. + let json_query = json!({"@context":{"ex":"http://example.org/"}, + "select":["?s","?p","?o"], "where":[{"@id":"?s", "?p":"?o"}], + "values":["?s", [ + {"@value":"ex:a","@type":"@id"}, + {"@value":"ex:a","@type":"@id"}, + {"@value":"ex:b","@type":"@id"}]]}); + let (spans, _guard) = support::span_capture::init_test_tracing(); + let twin = support::query_jsonld(&fluree, &ledger, &json_query) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + assert_eq!(normalized(&rows), normalized(&twin)); + assert!(spans.has_event(EVENT), "JSON-LD twin must fire"); + drop(_guard); + + // Incoming JSON-LD twin, including duplicate driving keys. + let mut incoming = json_query.clone(); + incoming["values"][0] = json!("?o"); + let (spans, _guard) = support::span_capture::init_test_tracing(); + let incoming_rows = support::query_jsonld(&fluree, &ledger, &incoming) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + assert_eq!(normalized(&incoming_rows).len(), 7); + assert!(spans.has_event(EVENT), "incoming JSON-LD twin must fire"); + drop(_guard); + + // UNDEF and literals can be interleaved with batchable keys. No pending + // fallback row may lose its retained left batch when a flush takes ownership. + check( + &fluree, + &ledger, + "SELECT ?s ?p ?o WHERE { VALUES ?o { ex:a 'A' ex:a } ?s ?p ?o }", + 7, + MustFire, + ) + .await; + check( + &fluree, + &ledger, + "SELECT ?p ?o WHERE { VALUES ?s { ex:ann ex:ann } ?s ?p ?o }", + 4, + MustFire, + ) + .await; + + let mut inspection = json_query.clone(); + inspection["values"] = json!(["?s", [ + {"@value":"ex:ann","@type":"@id"}, {"@value":"ex:ann","@type":"@id"}]]); + inspection["opts"] = json!({"includeSystemFacts":true}); + let (spans, _guard) = support::span_capture::init_test_tracing(); + let exposed = support::query_jsonld(&fluree, &ledger, &inspection) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + // Annotations store no encoding facts, so inspection sees the same four. + assert_eq!(normalized(&exposed).len(), 4); + assert!(spans.has_event(EVENT)); + set_fast_paths_disabled(true); + let generic = support::query_jsonld(&fluree, &ledger, &inspection) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + assert_eq!(normalized(&exposed), normalized(&generic)); + set_fast_paths_disabled(false); + drop(_guard); + + // A restrictive property policy must use the filtered scan, even at HEAD. + let mut restricted = json_query.clone(); + restricted["from"] = json!(alias); + restricted["opts"] = json!({"default-allow":true, "policy":[{ + "@id":"ex:hideName", "f:required":true, "f:action":"f:view", + "f:onProperty":[{"@id":"http://example.org/name"}], "f:allow":false + }]}); + let (spans, _guard) = support::span_capture::init_test_tracing(); + let hidden = fluree + .query_connection(&restricted) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + assert_eq!(normalized(&hidden).len(), 13); + assert!(!spans.has_event(EVENT), "restricted policy must decline"); + assert!(hidden["results"]["bindings"] + .as_array() + .unwrap() + .iter() + .all(|r| r["p"]["value"] != "http://example.org/name")); + drop(_guard); + // Multiple ledgers have independent dictionaries and must decline. + let other_alias = "wildcard/other:main"; + let other = fluree.create_ledger(other_alias).await.unwrap(); + fluree + .insert( + other, + &json!({"@context":{"ex":"http://example.org/"}, + "@id":"ex:outside", "ex:extra":99}), + ) + .await + .unwrap(); + fluree + .reindex(other_alias, ReindexOptions::default()) + .await + .unwrap(); + let mut dataset = json_query.clone(); + dataset["from"] = json!([alias, other_alias]); + dataset["values"] = json!(["?s", [ + {"@value":"ex:b","@type":"@id"}, {"@value":"ex:b","@type":"@id"}]]); + let (spans, _guard) = support::span_capture::init_test_tracing(); + let union = fluree + .query_connection(&dataset) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + assert_eq!(normalized(&union).len(), 4); + assert!(!spans.has_event(EVENT), "multi-ledger dataset must decline"); + drop(_guard); + let mut reasoning = json_query.clone(); + reasoning["reasoning"] = json!("rdfs"); + let (spans, _guard) = support::span_capture::init_test_tracing(); + let inferred = support::query_jsonld(&fluree, &ledger, &reasoning) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + assert_eq!(normalized(&inferred), normalized(&twin)); + assert!(!spans.has_event(EVENT), "reasoning must decline"); + drop(_guard); + let base_t = ledger.t(); + + // Any live novelty declines before accumulating: both directions see + // fresh asserts and retracts through the ordinary overlay-aware scan. + let receipt = fluree + .update( + ledger, + &json!({ + "@context":{"ex":"http://example.org/"}, + "where":{"@id":"ex:a","ex:name":"?old"}, + "delete":[{"@id":"ex:a","ex:name":"?old"}], + "insert":[{"@id":"ex:a","ex:name":"NEW"}, + {"@id":"ex:new","ex:edge":{"@id":"ex:a"}}] + }), + ) + .await + .unwrap(); + let rows = check(&fluree, &receipt.ledger, outgoing, 16, MustNotFire).await; + assert!(rows["results"]["bindings"] + .as_array() + .unwrap() + .iter() + .any(|r| r["o"]["value"] == "NEW")); + check( + &fluree, + &receipt.ledger, + "SELECT ?s ?p WHERE { VALUES ?o { ex:a ex:a } ?s ?p ?o }", + 8, + MustNotFire, + ) + .await; + fluree + .reindex(alias, ReindexOptions::default()) + .await + .unwrap(); + let head = fluree.ledger(alias).await.unwrap(); + // Past snapshots decline after a new index contains the retraction. + let mut historical = json_query.clone(); + historical["from"] = json!({"@id":alias, "t":base_t}); + let (spans, _guard) = support::span_capture::init_test_tracing(); + let past = fluree + .query_connection(&historical) + .await + .unwrap() + .to_sparql_json(&head.snapshot) + .unwrap(); + assert_eq!(normalized(&past), normalized(&twin)); + assert!(!spans.has_event(EVENT), "past snapshot must decline"); + drop(_guard); + historical["from"] = json!(format!("{alias}@t:{base_t}")); + historical["to"] = json!(format!("{alias}@t:latest")); + let (spans, _guard) = support::span_capture::init_test_tracing(); + let history = fluree + .query_connection(&historical) + .await + .unwrap() + .to_sparql_json(&head.snapshot) + .unwrap(); + assert!(normalized(&history).len() > 16); + assert!(!spans.has_event(EVENT), "history range must decline"); +} + +/// Reproducible local A/B probe benchmark. Run separately in dev-fast/release; +/// the fixture and expected counts are independent of StarBench and its ledger. +#[tokio::test(flavor = "current_thread")] +#[ignore = "local performance experiment; run with --profile dev-fast --ignored --nocapture"] +async fn wildcard_join_benchmark() { + use std::fmt::Write; + use std::time::Instant; + let _reset = Reset; + let dir = tempfile::tempdir().unwrap(); + let data = tempfile::tempdir().unwrap(); + let mut ttl = String::from("@prefix ex: .\n"); + let n: usize = std::env::var("FLUREE_WILDCARD_BENCH_KEYS") + .map(|s| s.parse().expect("positive key count")) + .unwrap_or(20_000); + assert!(n > 0); + for i in 0..n { + writeln!(ttl, "ex:driver{i} ex:pick ex:node{i} .\nex:node{i} ex:p1 1 ; ex:p2 2 ; ex:p3 ex:value .\nex:missing{i} ex:marker true .").unwrap(); + } + std::fs::write(data.path().join("data.ttl"), ttl).unwrap(); + let fluree = FlureeBuilder::file(dir.path().to_string_lossy().to_string()) + .build() + .unwrap(); + let alias = "wildcard/bench:main"; + fluree + .create(alias) + .import(data.path()) + .threads(1) + .memory_budget_mb(256) + .execute() + .await + .unwrap(); + let ledger = fluree.ledger(alias).await.unwrap(); + for (name, body, expected) in [ + ("outgoing", "?driver ex:pick ?s . ?s ?p ?o", n * 3), + ("incoming-misses", "?o ex:marker true . ?s ?p ?o", 0), + ("incoming-hits", "?o ex:p1 1 . ?s ?p ?o", n), + ] { + let query = format!("{PREFIX} SELECT (COUNT(*) AS ?n) WHERE {{ {body} }}"); + set_fast_paths_disabled(false); + let (spans, tracing_guard) = support::span_capture::init_test_tracing(); + support::query_sparql(&fluree, &ledger, &query) + .await + .unwrap(); + assert!(spans.has_event(EVENT), "benchmark must engage: {name}"); + drop(tracing_guard); + let mut times = [Vec::new(), Vec::new()]; + for round in 0..8 { + // Alternate mode order; exclude the first warm-up pair. + for mode in [round % 2, 1 - round % 2] { + set_fast_paths_disabled(mode == 1); + let start = Instant::now(); + let result = support::query_sparql(&fluree, &ledger, &query) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + let elapsed = start.elapsed().as_secs_f64(); + assert_eq!( + result["results"]["bindings"][0]["n"]["value"], + expected.to_string() + ); + if round != 0 { + times[mode].push(elapsed); + } + } + } + for t in &mut times { + t.sort_by(f64::total_cmp); + } + println!( + "{name}: batched={:.6}s generic={:.6}s speedup={:.2}x (median of 7, {n} keys)", + times[0][3], + times[1][3], + times[1][3] / times[0][3] + ); + } +} + +/// Term-valued objects drive the wildcard incoming probe through the per-row +/// scan; the ref keys beside them still batch. +#[tokio::test(flavor = "current_thread")] +async fn wildcard_incoming_probe_leaves_term_keys_to_the_scan() { + let _reset = Reset; + let dir = tempfile::tempdir().unwrap(); + let fluree = FlureeBuilder::file(dir.path().to_string_lossy().to_string()) + .build() + .unwrap(); + let alias = "wildcard/terms:main"; + let ledger = fluree.create_ledger(alias).await.unwrap(); + fluree + .insert_turtle( + ledger, + "VERSION \"1.2\"\n@prefix ex: .\n\ + << ex:s1 ex:partOf ex:anchor ~ ex:r1 >> ex:source ex:d1 .\n\ + ex:doc1 ex:mentions <<( ex:s1 ex:partOf ex:anchor )>> .\n\ + ex:doc2 ex:mentions ex:r1 .\n\ + ex:a1 ex:cites ex:r1 .\n\ + ex:a2 ex:cites <<( ex:s1 ex:partOf ex:anchor )>> .\n", + ) + .await + .unwrap(); + fluree + .reindex(alias, ReindexOptions::default()) + .await + .unwrap(); + let ledger = fluree.ledger(alias).await.unwrap(); + // doc1's term: its mention, a2's citation and r1's link; doc2's ref: + // its mention and a1's citation. + check( + &fluree, + &ledger, + "SELECT ?doc ?s ?p WHERE { VALUES ?doc { ex:doc1 ex:doc2 } \ + ?doc ex:mentions ?t . ?s ?p ?t }", + 5, + Routing::MustFire, + ) + .await; +} diff --git a/fluree-db-binary-index/src/read/batched_lookup.rs b/fluree-db-binary-index/src/read/batched_lookup.rs index e50d436d8d..ba2918b4d6 100644 --- a/fluree-db-binary-index/src/read/batched_lookup.rs +++ b/fluree-db-binary-index/src/read/batched_lookup.rs @@ -278,118 +278,25 @@ pub fn batched_lookup_subject_properties( subjects: &[u64], to_t: i64, ) -> io::Result { - let mut out: HashMap> = HashMap::new(); - if subjects.is_empty() { - return Ok(out); - } - - let mut sorted_subjects = subjects.to_vec(); - sorted_subjects.sort_unstable(); - sorted_subjects.dedup(); - - let Some(branch) = store.branch_for_order(g_id, RunSortOrder::Spot) else { - return Ok(out); - }; - let branch = Arc::clone(branch); - - const MAX_SPAN: u64 = 100_000; - const MAX_CHUNK: usize = 1000; - let chunks = chunk_subjects(&sorted_subjects, MAX_SPAN, MAX_CHUNK); - - // Need s_id (to filter), p_id, o_type, o_key (to reconstruct datatype/lang/ref). - let mut needed = ColumnSet::EMPTY; - needed.insert(ColumnId::SId); - needed.insert(ColumnId::PId); - needed.insert(ColumnId::OType); - needed.insert(ColumnId::OKey); - let projection = ColumnProjection { - output: needed, - internal: ColumnSet::EMPTY, - }; - - #[cfg(any(target_arch = "wasm32", feature = "residency"))] - register_routed_wants( - store, - &branch, - RunSortOrder::Spot, - chunks.iter().map(|chunk| { - let (min_s, max_s) = (chunk[0], *chunk.last().unwrap()); - ( - RunRecordV2 { - s_id: SubjectId::from_u64(min_s), - o_key: 0, - p_id: 0, - t: 0, - o_i: 0, - o_type: 0, - g_id, - }, - RunRecordV2 { - s_id: SubjectId::from_u64(max_s), - o_key: u64::MAX, - p_id: u32::MAX, - t: 0, - o_i: u32::MAX, - o_type: u16::MAX, - g_id, - }, - ) - }), + let mut out: SubjectPropertyFlakes = HashMap::new(); + let mut cursor = BatchedWildcardCursor::new( + Arc::clone(store), + g_id, + subjects, + to_t, + WildcardDirection::Outgoing, + property_projection(), ); - - for chunk in &chunks { - let min_s = chunk[0]; - let max_s = *chunk.last().unwrap(); - - let min_key = RunRecordV2 { - s_id: SubjectId::from_u64(min_s), - o_key: 0, - p_id: 0, - t: 0, - o_i: 0, - o_type: 0, - g_id, - }; - let max_key = RunRecordV2 { - s_id: SubjectId::from_u64(max_s), - o_key: u64::MAX, - p_id: u32::MAX, - t: 0, - o_i: u32::MAX, - o_type: u16::MAX, - g_id, - }; - - let mut cursor = BinaryCursor::new( - Arc::clone(store), - RunSortOrder::Spot, - Arc::clone(&branch), - &min_key, - &max_key, - BinaryFilter::default(), - projection, - ); - cursor.set_to_t(to_t); - - // Batches are leaflet-granular and spill far past the wanted - // subjects, so testing every row is the dominant cost for small - // subject sets (a single-node hydration paid a full-leaflet - // membership scan). SPOT's primary sort key is `s_id`, so gallop - // instead: binary-search each wanted subject's contiguous run and - // copy only those rows. Spillover rows for a neighboring chunk's - // subjects are excluded by construction (only this chunk's ids are - // searched), preserving the no-double-collect invariant the - // per-chunk membership set used to provide. - while let Some(batch) = cursor.next_batch()? { - for_each_subject_run(&batch, chunk, |s_id, i, batch| { - let p_id = batch.p_id.get_or(i, 0); - let o_type = batch.o_type.get_or(i, 0); - let o_key = batch.o_key.get(i); - out.entry(s_id).or_default().push((p_id, o_type, o_key)); - }); + while let Some(matches) = cursor.next_batch()? { + for (s_id, i) in matches.rows { + let batch = &matches.batch; + out.entry(s_id).or_default().push(( + batch.p_id.get_or(i, 0), + batch.o_type.get_or(i, 0), + batch.o_key.get(i), + )); } } - Ok(out) } @@ -487,121 +394,201 @@ pub fn batched_lookup_inbound_refs( to_t: i64, ) -> io::Result>> { let mut out: HashMap> = HashMap::new(); - if objects.is_empty() { - return Ok(out); + let mut cursor = BatchedWildcardCursor::new( + Arc::clone(store), + g_id, + objects, + to_t, + WildcardDirection::IncomingRefs, + property_projection(), + ); + while let Some(matches) = cursor.next_batch()? { + for (o_key, i) in matches.rows { + out.entry(o_key) + .or_default() + .push((matches.batch.p_id.get_or(i, 0), matches.batch.s_id.get(i))); + } } + for v in out.values_mut() { + v.sort_unstable(); + v.dedup(); + } + Ok(out) +} - let mut sorted = objects.to_vec(); - sorted.sort_unstable(); - sorted.dedup(); - - let Some(branch) = store.branch_for_order(g_id, RunSortOrder::Opst) else { - return Ok(out); - }; - let branch = Arc::clone(branch); - let iri_ref = OType::IRI_REF.as_u16(); - - const MAX_SPAN: u64 = 100_000; - const MAX_CHUNK: usize = 1000; - // chunk_subjects operates on a sorted &[u64] span — identical logic applies - // to object o_key spans, so reuse it verbatim. - let chunks = chunk_subjects(&sorted, MAX_SPAN, MAX_CHUNK); - - // Need s_id (inbound subject), p_id, o_key (filter to set), o_type (ref check). +fn property_projection() -> ColumnProjection { let mut needed = ColumnSet::EMPTY; needed.insert(ColumnId::SId); needed.insert(ColumnId::PId); needed.insert(ColumnId::OType); needed.insert(ColumnId::OKey); - let projection = ColumnProjection { + ColumnProjection { output: needed, internal: ColumnSet::EMPTY, - }; + } +} - #[cfg(any(target_arch = "wasm32", feature = "residency"))] - register_routed_wants( - store, - &branch, - RunSortOrder::Opst, - chunks.iter().map(|chunk| { - let (min_o, max_o) = (chunk[0], *chunk.last().unwrap()); - ( - RunRecordV2 { - s_id: SubjectId::from_u64(0), - o_key: min_o, - p_id: 0, - t: 0, - o_i: 0, - o_type: iri_ref, - g_id, - }, - RunRecordV2 { - s_id: SubjectId::from_u64(u64::MAX), - o_key: max_o, - p_id: u32::MAX, - t: 0, - o_i: u32::MAX, - o_type: iri_ref, - g_id, - }, - ) - }), - ); +/// Which endpoint of a wildcard-predicate lookup is bound. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum WildcardDirection { + Outgoing, + /// Only reference-valued objects; literal probes must use another path. + IncomingRefs, +} - for chunk in &chunks { - let min_o = chunk[0]; - let max_o = *chunk.last().unwrap(); +impl WildcardDirection { + fn order(self) -> RunSortOrder { + match self { + Self::Outgoing => RunSortOrder::Spot, + Self::IncomingRefs => RunSortOrder::Opst, + } + } +} - // OPST order = (o_type, o_key, o_i, p_id, s_id); pin o_type = IRI_REF. - let min_key = RunRecordV2 { - s_id: SubjectId::from_u64(0), - o_key: min_o, +/// One leaflet and the `(bound key, row index)` matches within it. Keeping the +/// source columns preserves list position, time, and datatype metadata for +/// query consumers; stats and hydration can project only the columns they need. +pub struct WildcardMatches { + pub batch: super::column_types::ColumnBatch, + pub rows: Vec<(u64, usize)>, +} + +/// Streaming, gap-aware wildcard lookup shared by joins, stats and hydration. +/// Reads persisted snapshot state at `to_t`; the caller must merge or decline +/// novelty and enforce policy/graph visibility. Does not deduplicate facts. +/// Each call yields at most one leaflet, including empty match sets so callers +/// can check cancellation while traversing gaps. Output follows index order, +/// not the order (or multiplicity) of the input keys. +pub struct BatchedWildcardCursor { + store: Arc, + branch: Option>, + g_id: GraphId, + to_t: i64, + direction: WildcardDirection, + projection: ColumnProjection, + chunks: Vec>, + chunk_idx: usize, + cursor: Option, +} + +impl BatchedWildcardCursor { + pub fn new( + store: Arc, + g_id: GraphId, + keys: &[u64], + to_t: i64, + direction: WildcardDirection, + projection: ColumnProjection, + ) -> Self { + let mut sorted = keys.to_vec(); + sorted.sort_unstable(); + sorted.dedup(); + let chunks: Vec> = chunk_subjects(&sorted, 100_000, 1000) + .into_iter() + .map(<[u64]>::to_vec) + .collect(); + let branch = store.branch_for_order(g_id, direction.order()).cloned(); + #[cfg(any(target_arch = "wasm32", feature = "residency"))] + if let Some(branch) = &branch { + register_routed_wants( + &store, + branch, + direction.order(), + chunks + .iter() + .map(|chunk| Self::bounds(g_id, direction, chunk)), + ); + } + Self { + store, + branch, + g_id, + to_t, + direction, + projection, + chunks, + chunk_idx: 0, + cursor: None, + } + } + + fn bounds( + g_id: GraphId, + direction: WildcardDirection, + keys: &[u64], + ) -> (RunRecordV2, RunRecordV2) { + let mut lo = RunRecordV2 { + s_id: SubjectId(0), + o_key: 0, p_id: 0, t: 0, o_i: 0, - o_type: iri_ref, + o_type: 0, g_id, }; - let max_key = RunRecordV2 { + let mut hi = RunRecordV2 { s_id: SubjectId::from_u64(u64::MAX), - o_key: max_o, + o_key: u64::MAX, p_id: u32::MAX, t: 0, o_i: u32::MAX, - o_type: iri_ref, + o_type: u16::MAX, g_id, }; - - let mut cursor = BinaryCursor::new( - Arc::clone(store), - RunSortOrder::Opst, - Arc::clone(&branch), - &min_key, - &max_key, - BinaryFilter::default(), - projection, - ); - cursor.set_to_t(to_t); - - // OPST sorts by (o_type, o_key, ...): binary-search the IRI_REF - // o_type run (boundary leaflets can carry neighboring o_types), - // then gallop the chunk's wanted o_keys within it. Spillover rows - // for a neighboring chunk's objects are excluded by construction. - while let Some(batch) = cursor.next_batch()? { - let (lo, hi) = o_type_run(&batch, iri_ref); - for_each_wanted_run(&batch.o_key, lo, hi, chunk, |o_key, i| { - out.entry(o_key) - .or_default() - .push((batch.p_id.get_or(i, 0), batch.s_id.get(i))); - }); + match direction { + WildcardDirection::Outgoing => { + lo.s_id = SubjectId::from_u64(keys[0]); + hi.s_id = SubjectId::from_u64(*keys.last().unwrap()); + } + WildcardDirection::IncomingRefs => { + lo.o_type = OType::IRI_REF.as_u16(); + hi.o_type = lo.o_type; + lo.o_key = keys[0]; + hi.o_key = *keys.last().unwrap(); + } } + (lo, hi) } - for v in out.values_mut() { - v.sort_unstable(); - v.dedup(); + pub fn next_batch(&mut self) -> io::Result> { + let Some(branch) = &self.branch else { + return Ok(None); + }; + while let Some(chunk) = self.chunks.get(self.chunk_idx) { + if self.cursor.is_none() { + let (lo, hi) = Self::bounds(self.g_id, self.direction, chunk); + let mut cursor = BinaryCursor::new( + Arc::clone(&self.store), + self.direction.order(), + Arc::clone(branch), + &lo, + &hi, + BinaryFilter::default(), + self.projection, + ); + cursor.set_to_t(self.to_t); + self.cursor = Some(cursor); + } + if let Some(batch) = self.cursor.as_mut().unwrap().next_batch()? { + let mut rows = Vec::new(); + match self.direction { + WildcardDirection::Outgoing => { + for_each_subject_run(&batch, chunk, |key, row, _| rows.push((key, row))); + } + WildcardDirection::IncomingRefs => { + let (lo, hi) = o_type_run(&batch, OType::IRI_REF.as_u16()); + for_each_wanted_run(&batch.o_key, lo, hi, chunk, |key, row| { + rows.push((key, row)); + }); + } + } + return Ok(Some(WildcardMatches { batch, rows })); + } + self.cursor = None; + self.chunk_idx += 1; + } + Ok(None) } - Ok(out) } /// Residency mode: record every non-resident leaf the routed key ranges will diff --git a/fluree-db-query/src/execute/operator_tree.rs b/fluree-db-query/src/execute/operator_tree.rs index b9df4ad4dd..2a34655a2e 100644 --- a/fluree-db-query/src/execute/operator_tree.rs +++ b/fluree-db-query/src/execute/operator_tree.rs @@ -2243,7 +2243,8 @@ fn detect_union_star_count_all( /// query with fast paths on and off and asserts identical results, and as /// an operational escape hatch when triaging a suspected fast-path bug. /// It is NOT a tuning knob: runtime operator-internal optimizations -/// (cursor selection, batched joins) are unaffected. +/// (cursor selection, fixed-predicate batched joins) are unaffected. The +/// wildcard-predicate join lane also honors this switch for differential tests. static FAST_PATHS_DISABLED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); diff --git a/fluree-db-query/src/join.rs b/fluree-db-query/src/join.rs index 483d2be9c8..93a998df80 100644 --- a/fluree-db-query/src/join.rs +++ b/fluree-db-query/src/join.rs @@ -4,6 +4,8 @@ //! where left results drive right scans. It enforces var unification - shared //! vars between left and right must match exactly. +mod wildcard; + use crate::binary_scan::EmitMask; use crate::binding::{Batch, Binding, RowAccess}; use crate::context::ExecutionContext; @@ -24,6 +26,7 @@ use crate::operator::{ }; use crate::var_registry::VarId; use async_trait::async_trait; +use fluree_db_binary_index::read::batched_lookup::WildcardDirection; use fluree_db_binary_index::{BinaryGraphView, BinaryIndexStore}; use fluree_db_core::clock::Instant; use fluree_db_core::o_type::OType; @@ -585,6 +588,8 @@ pub struct NestedLoopJoinOperator { active_right_left_row: usize, /// Optional object bounds for range filter pushdown object_bounds: Option, + wildcard_direction: Option, + wildcard_stream: Option, /// Whether this join is eligible for batched subject join batched_eligible: bool, /// Whether this join is eligible for batched object join @@ -848,9 +853,12 @@ impl NestedLoopJoinOperator { let has_bounds = object_bounds.is_some(); - let batched_eligible = is_batched_eligible(&bind_instructions, &right_pattern); + let wildcard_direction = wildcard::eligible(&bind_instructions, &right_pattern, has_bounds); + let batched_eligible = is_batched_eligible(&bind_instructions, &right_pattern) + || wildcard_direction == Some(WildcardDirection::Outgoing); let batched_object_eligible = !batched_eligible - && is_batched_object_eligible(&bind_instructions, &right_pattern, has_bounds); + && (is_batched_object_eligible(&bind_instructions, &right_pattern, has_bounds) + || wildcard_direction == Some(WildcardDirection::IncomingRefs)); let batched_exists_eligible = !batched_eligible && !batched_object_eligible && is_batched_subject_exists_eligible(&bind_instructions, &right_pattern); @@ -898,6 +906,8 @@ impl NestedLoopJoinOperator { active_right_batch_ref: None, active_right_left_row: 0, object_bounds, + wildcard_direction, + wildcard_stream: None, batched_eligible, batched_object_eligible, batched_exists_eligible, @@ -1457,6 +1467,12 @@ impl Operator for NestedLoopJoinOperator { // Process until we have output or exhaust input loop { ctx.check_cancelled()?; + if let Some(mut stream) = self.wildcard_stream.take() { + if let Some(batch) = stream.next_batch(self, ctx)? { + self.wildcard_stream = Some(stream); + return Ok(trim_batch(&self.out_schema, batch)); + } + } // 1. Pre-built output from batched flush if let Some(batch) = self.batched_output.pop_front() { return Ok(trim_batch(&self.out_schema, batch)); @@ -1592,8 +1608,10 @@ impl Operator for NestedLoopJoinOperator { } // A term handle keys OPST like a ref; per row it would // decode the term only for the scan to re-derive it. + // The wildcard cursor reads refs only. Binding::EncodedLit { o_kind, o_key, .. } if self.batched_object_eligible + && self.wildcard_direction.is_none() && *o_kind == ObjKind::TRIPLE_TERM.as_u8() => { Some((OType::TRIPLE_TERM.as_u16(), *o_key)) @@ -1781,6 +1799,7 @@ impl Operator for NestedLoopJoinOperator { fn close(&mut self) { self.left.close(); + self.wildcard_stream = None; self.current_left_batch = None; self.current_left_batch_stored_idx = None; self.pending_output.clear(); @@ -1939,6 +1958,11 @@ impl NestedLoopJoinOperator { } let accum_len = self.batched_accumulator.len(); + if let Some(direction) = self.wildcard_direction { + self.wildcard_stream = Some(wildcard::WildcardJoin::new(self, ctx)); + tracing::debug!(?direction, accum_len, "batched wildcard join engaged"); + return Ok(()); + } if self.batched_object_eligible { self.flush_batched_object_accumulator_binary(ctx) .instrument(tracing::debug_span!( @@ -1992,7 +2016,32 @@ impl NestedLoopJoinOperator { Some(pred) => &[pred], None => &[], }; - if let Some(plan) = crate::fast_path_common::probe_lane_admission(ctx, preds) { + let admitted = crate::fast_path_common::probe_lane_admission(ctx, preds); + if self.wildcard_direction.is_some() { + // The shared cursor reads persisted state only. Keep the first + // query lane restricted to HEAD, one ledger/graph, and no novelty. + return Ok( + if !matches!(admitted, Some(ProbeLanePlan::Decline)) + && !crate::fast_paths_disabled() + && self.mode.is_current() + && ctx.from_t.is_none() + && !ctx.is_multi_ledger() + && !ctx.eager_materialization + && !ctx.reasoning_active + && matches!(ctx.active_graphs(), ActiveGraphs::Single) + && ctx.overlay_free_single_graph() + && ctx + .binary_store + .as_ref() + .is_some_and(|s| ctx.to_t >= s.max_t()) + { + ProbeLanePlan::Clean + } else { + ProbeLanePlan::Decline + }, + ); + } + if let Some(plan) = admitted { return Ok(plan); } if !(self.batched_eligible || self.batched_object_eligible || self.batched_exists_eligible) diff --git a/fluree-db-query/src/join/wildcard.rs b/fluree-db-query/src/join/wildcard.rs new file mode 100644 index 0000000000..e0b1445c80 --- /dev/null +++ b/fluree-db-query/src/join/wildcard.rs @@ -0,0 +1,201 @@ +//! Wildcard-predicate bind joins over the shared streaming index cursor. +//! +//! The cursor sorts distinct probe keys; this layer restores driving-row +//! multiplicity, without collecting the (potentially enormous) expanded join. +//! As with other unordered physical plans, output order is not guaranteed. + +use super::*; +use fluree_db_binary_index::read::batched_lookup::{ + BatchedWildcardCursor, WildcardDirection, WildcardMatches, +}; + +pub(super) fn eligible( + binds: &[BindInstruction], + pattern: &TriplePattern, + has_bounds: bool, +) -> Option { + let (Ref::Var(s), Ref::Var(p), Term::Var(o)) = (&pattern.s, &pattern.p, &pattern.o) else { + return None; + }; + // Predicate must be a new variable. Repeated variables require cross-ID + // equality checks and remain on the ordinary scan path. + if s == p || s == o || p == o || has_bounds || pattern.dtc.is_some() || binds.len() != 1 { + return None; + } + match binds[0].position { + PatternPosition::Subject => Some(WildcardDirection::Outgoing), + PatternPosition::Object => Some(WildcardDirection::IncomingRefs), + PatternPosition::Predicate => None, + } +} + +pub(super) struct WildcardJoin { + cursor: BatchedWildcardCursor, + left_batches: Vec, + /// A lookup key can drive several rows, including exact duplicate rows. + left_rows: FxHashMap>, + matches: Option, + match_idx: usize, + left_idx: usize, + right: Option>, +} + +impl WildcardJoin { + pub(super) fn new(join: &mut NestedLoopJoinOperator, ctx: &ExecutionContext<'_>) -> Self { + let mut left_rows: FxHashMap> = FxHashMap::default(); + for (batch, row, key) in join.batched_accumulator.drain(..) { + left_rows.entry(key).or_default().push((batch, row)); + } + let keys: Vec = left_rows.keys().copied().collect(); + let cursor = BatchedWildcardCursor::new( + Arc::clone(ctx.binary_store.as_ref().unwrap()), + ctx.binary_g_id, + &keys, + ctx.to_t, + join.wildcard_direction.unwrap(), + fluree_db_binary_index::ColumnProjection::all(), + ); + let left_batches = std::mem::take(&mut join.stored_left_batches); + join.current_left_batch_stored_idx = None; + Self { + cursor, + left_batches, + left_rows, + matches: None, + match_idx: 0, + left_idx: 0, + right: None, + } + } + + pub(super) fn next_batch( + &mut self, + join: &NestedLoopJoinOperator, + ctx: &ExecutionContext<'_>, + ) -> Result> { + let store = ctx.binary_store.as_ref().unwrap(); + let mut columns: Vec> = (0..join.combined_schema.len()) + .map(|_| Vec::with_capacity(ctx.batch_size)) + .collect(); + let mut count = 0; + while count < ctx.batch_size { + ctx.check_cancelled()?; + if self + .matches + .as_ref() + .is_none_or(|m| self.match_idx == m.rows.len()) + { + self.matches = self + .cursor + .next_batch() + .map_err(|e| QueryError::from_io("batched wildcard probe", e))?; + self.match_idx = 0; + self.left_idx = 0; + self.right = None; + let Some(matches) = &self.matches else { + break; + }; + charge_probe_rows(ctx, matches.rows.len())?; + if matches.rows.is_empty() { + continue; + } + } + let matches = self.matches.as_ref().unwrap(); + let (key, row) = matches.rows[self.match_idx]; + if self.right.is_none() { + let batch = &matches.batch; + let p_id = batch.p_id.get_or(row, 0); + // Match BinaryScan's wildcard visibility rule, including the + // explicit inspection escape. Never hide the whole f: namespace. + if !ctx.include_system_facts + && store + .p_sid_table() + .get(p_id as usize) + .is_some_and(fluree_db_core::is_scan_hidden_predicate) + { + self.match_idx += 1; + continue; + } + let mut right = Vec::with_capacity(join.right_new_vars.len()); + for var in &join.right_new_vars { + let value = if Some(*var) == join.right_pattern.s.as_var() { + Binding::encoded_sid(batch.s_id.get(row)) + } else if Some(*var) == join.right_pattern.p.as_var() { + Binding::EncodedPid { p_id } + } else { + build_probe_object_binding( + ctx, + store, + None, + p_id, + batch.o_type.get_or(row, 0), + batch.o_key.get(row), + batch.o_i.get_or(row, u32::MAX), + batch.t.get_or(row, 0) as i64, + )? + }; + right.push(value); + } + if !join.apply_right_scan_inline_ops(ctx, &mut right)? { + self.match_idx += 1; + continue; + } + self.right = Some(right); + } + let left_rows = &self.left_rows[&key]; + let (batch_idx, row_idx) = left_rows[self.left_idx]; + self.left_idx += 1; + let left = &self.left_batches[batch_idx]; + let right = self.right.as_ref().unwrap(); + if join.inline_has_bind() { + let mut combined: Vec = (0..join.left_schema.len()) + .map(|col| left.get_by_col(row_idx, col).clone()) + .collect(); + combined.extend(right.iter().cloned()); + if apply_inline( + &join.inline_ops, + &join.combined_schema, + &mut combined, + Some(ctx), + )? { + for (col, val) in columns.iter_mut().zip(combined) { + col.push(val); + } + count += 1; + } + } else { + // Share the fixed-predicate join's row view: filters need no + // per-output-row allocation and left columns are copied once. + let view = CombinedRowView { + left_batch: left, + left_row: row_idx, + left_len: join.left_schema.len(), + right, + schema: &join.combined_schema, + }; + if apply_inline_filters_view(&join.inline_ops, &view, ctx)? { + for (i, col) in columns.iter_mut().enumerate() { + col.push(if i < join.left_schema.len() { + left.get_by_col(row_idx, i).clone() + } else { + right[i - join.left_schema.len()].clone() + }); + } + count += 1; + } + } + if self.left_idx == left_rows.len() { + self.left_idx = 0; + self.match_idx += 1; + self.right = None; + } + } + if count == 0 { + return Ok(None); + } + if columns.is_empty() { + return Ok(Some(Batch::empty_schema_with_len(count))); + } + Ok(Some(Batch::new(join.combined_schema.clone(), columns)?)) + } +} From d70ff69df45f9274d4b762172b4ed47a9598c376 Mon Sep 17 00:00:00 2001 From: bplatz Date: Sun, 4 Oct 2026 11:23:51 -0400 Subject: [PATCH 66/92] perf(query): a semijoin seeds its build from a small outer side The EXISTS semijoin built its key set by running the inner body once, unseeded. For an annotated edge's existence check that body is the whole edge predicate. Large outer sides gain from that (LDBC IC7, IC5), but a point query paid for a full predicate scan to check a handful of edges. The operator now reads its outer side first. If the outer side ends within 1024 distinct keys, the build is seeded by those keys; otherwise the whole body is built, as before. Seeding is limited to a conjunction of triple patterns, where a seeded solution is a solution of the unseeded body. A row with an unbound key seeds as a free variable, so the projected lookups for OPTIONAL-unbound keys still see every match. LDBC SF0.1, local: - IS3: 6.9ms -> 4.5ms (4.6ms before the semijoin) - IC1: 39ms -> 29ms - IC5, IC7: unchanged --- fluree-db-api/tests/it_query_negation.rs | 48 ++++++ fluree-db-query/src/semijoin.rs | 182 ++++++++++++++++++++--- 2 files changed, 211 insertions(+), 19 deletions(-) diff --git a/fluree-db-api/tests/it_query_negation.rs b/fluree-db-api/tests/it_query_negation.rs index 8c4b66841b..4023a7eb0e 100644 --- a/fluree-db-api/tests/it_query_negation.rs +++ b/fluree-db-api/tests/it_query_negation.rs @@ -1044,6 +1044,54 @@ SELECT ?p ?org WHERE { assert_eq!(normalize_rows(&rows), expected); } +/// A small outer side seeds the semijoin's build with its own keys. An +/// unbound OPTIONAL key seeds as a free variable, so its projected lookup +/// still sees every match. +#[tokio::test] +async fn semijoin_seeds_its_build_from_a_small_outer_side() { + use fluree_db_api::{QueryInput, ReindexOptions}; + + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "negation:seeded-semijoin"; + seed_knows_works_for(&fluree, ledger_id).await; + let not_exists = r"PREFIX ex: +SELECT ?p ?org WHERE { + ?p ex:knows ?f . + OPTIONAL { ?p ex:worksFor ?org } + FILTER NOT EXISTS { ?p ex:knows ?x . ?x ex:worksFor ?org } +}"; + let cases = [ + ( + not_exists.to_string(), + json!([["ex:carol", "ex:globex"], ["ex:dave", null]]), + ), + ( + not_exists.replace("NOT EXISTS", "EXISTS"), + json!([["ex:alice", null]]), + ), + ]; + for indexed in [false, true] { + if indexed { + fluree + .reindex(ledger_id, ReindexOptions::default()) + .await + .expect("reindex"); + } + let view = fluree.db(ledger_id).await.expect("view"); + for (query, expected) in &cases { + let result = fluree + .query(&view, QueryInput::Sparql(query)) + .await + .expect("query"); + assert_eq!( + normalize_rows(&result.to_jsonld(&view.snapshot).unwrap()), + normalize_rows(expected), + "indexed={indexed}: {query}" + ); + } + } +} + /// A missing OPTIONAL binding must use a reusable existence lookup, while /// bound values still constrain the inner match and outer duplicates survive. #[tokio::test] diff --git a/fluree-db-query/src/semijoin.rs b/fluree-db-query/src/semijoin.rs index 7379f5b938..a32ab7d8b7 100644 --- a/fluree-db-query/src/semijoin.rs +++ b/fluree-db-query/src/semijoin.rs @@ -2,8 +2,10 @@ //! //! Replaces per-row correlated subquery evaluation with a build-probe approach: //! -//! 1. **Build phase** (`open`): Execute inner patterns once (uncorrelated), collect -//! distinct key tuples (the correlation variables) into a `HashSet`. +//! 1. **Build phase** (`open`): Execute inner patterns once, collect distinct key +//! tuples (the correlation variables) into a `HashSet`. An outer side with +//! few distinct keys seeds the build with them, so a point query does not +//! pay for the whole inner relation. //! 2. **Probe phase** (`next_batch`): For each outer row, extract key var values and //! probe the set. EXISTS keeps matches; NOT EXISTS keeps non-matches. //! @@ -22,18 +24,28 @@ use crate::group_aggregate::{CompositeGroupKey, GroupKeyOwned}; use crate::ir::Pattern; use crate::object_binding::{equality_norm, EqualityNorm}; use crate::operator::{BoxedOperator, Operator, OperatorState}; -use crate::seed::{EmptyOperator, SeedOperator}; +use crate::seed::{BatchSeedOperator, EmptyOperator, SeedOperator}; use crate::temporal_mode::PlanningContext; use crate::var_registry::VarId; use async_trait::async_trait; use fluree_db_core::StatsView; use rustc_hash::{FxHashMap, FxHashSet}; +use std::collections::VecDeque; use std::sync::Arc; /// Avoid retaining every one of the exponentially many binding masks. Further /// masks use the existing seeded evaluation; cached masks remain reusable. const MAX_PARTIAL_KEY_SETS: usize = 4; +/// Outer sides with at most this many distinct keys build the inner side +/// seeded by those keys. Each seeded key costs an index lookup where the +/// unseeded build costs a scan row, so the unseeded build wins once the +/// outer side is a sizeable fraction of the inner relation. +const SEEDED_BUILD_MAX_KEYS: usize = 1024; + +/// Bound on the outer rows buffered while deciding, whatever their keys. +const SEEDED_BUILD_MAX_ROWS: usize = 64 * 1024; + /// Approximate retained key storage, shared by the base and projected sets. /// Counts the tuple and its cells; excludes table slack and shared payloads. fn key_entry_bytes(width: usize) -> usize { @@ -62,6 +74,12 @@ pub struct SemijoinOperator { partial_keys_safe: bool, /// Key positions (not batch columns) -> projected, normalized inner keys. partial_key_sets: FxHashMap, FxHashSet>, + /// Outer batches read during `open()` to size the build, replayed first. + buffered: VecDeque, + /// The child returned `None` while buffering. + child_exhausted: bool, + /// The build was seeded by the outer keys. + seeded: bool, /// Column indices of key_vars within child.schema(), computed in `open()`. key_col_indices: Vec, /// Stats for nested query building. @@ -98,6 +116,9 @@ impl SemijoinOperator { key_set: FxHashSet::default(), partial_keys_safe, partial_key_sets: FxHashMap::default(), + buffered: VecDeque::new(), + child_exhausted: false, + seeded: false, norm: None, key_col_indices: Vec::new(), stats, @@ -218,6 +239,19 @@ impl SemijoinOperator { } } +impl SemijoinOperator { + /// Outer batches buffered by `open()` first, then the child's. + async fn next_child_batch(&mut self, ctx: &ExecutionContext<'_>) -> Result> { + if let Some(batch) = self.buffered.pop_front() { + return Ok(Some(batch)); + } + if self.child_exhausted { + return Ok(None); + } + self.child.next_batch(ctx).await + } +} + /// Composite key over the columns `cols` of one row. fn row_key( batch: &Batch, @@ -247,9 +281,80 @@ impl Operator for SemijoinOperator { self.norm = equality_norm(ctx); } + // Compute key column indices for the child (outer) schema. + let child_schema = self.child.schema().to_vec(); + self.key_col_indices = self + .key_vars + .iter() + .map(|kv| { + child_schema.iter().position(|v| v == kv).ok_or_else(|| { + QueryError::Internal(format!("key var {kv:?} not found in child schema")) + }) + }) + .collect::>>()?; + self.child.open(ctx).await?; + + // Read the outer side until it ends or shows more distinct keys than a + // seeded build should look up. Only a conjunction of triples seeds: a + // seeded solution of it is a solution of the unseeded body. + let mut seen: FxHashSet = FxHashSet::default(); + let mut seed_rows: Vec> = Vec::new(); + let mut overflow = !self.partial_keys_safe; + let mut buffered_rows = 0usize; + while !overflow { + let Some(batch) = self.child.next_batch(ctx).await? else { + self.child_exhausted = true; + break; + }; + for row_idx in 0..batch.len() { + // An unbound key seeds as a free variable, so the build holds + // every inner solution the row could match. A poisoned row + // keeps its seeded per-row evaluation. + if self + .key_col_indices + .iter() + .any(|&ci| matches!(batch.get_by_col(row_idx, ci), Binding::Poisoned)) + { + continue; + } + let key = row_key(&batch, row_idx, &self.key_col_indices, &self.norm); + if seen.insert(key) { + seed_rows.push( + self.key_col_indices + .iter() + .map(|&ci| batch.get_by_col(row_idx, ci).clone()) + .collect(), + ); + if seen.len() > SEEDED_BUILD_MAX_KEYS { + overflow = true; + break; + } + } + } + buffered_rows += batch.len(); + overflow |= buffered_rows > SEEDED_BUILD_MAX_ROWS; + self.buffered.push_back(batch); + } + drop(seen); + self.seeded = !overflow; + // Build phase: execute inner patterns once, collect distinct key tuples. #[allow(clippy::box_default)] - let seed: BoxedOperator = Box::new(EmptyOperator::new()); + let seed: BoxedOperator = if overflow { + Box::new(EmptyOperator::new()) + } else { + let schema: Arc<[VarId]> = Arc::from(self.key_vars.clone().into_boxed_slice()); + let columns = (0..self.key_vars.len()) + .map(|col| seed_rows.iter().map(|r| r[col].clone()).collect()) + .collect(); + Box::new(BatchSeedOperator::from_batch(Batch::new(schema, columns)?)) + }; + tracing::debug!( + seeded = self.seeded, + seed_keys = seed_rows.len(), + "semijoin build" + ); + drop(seed_rows); let mut inner_op = build_where_operators_seeded( Some(seed), &self.inner_patterns, @@ -293,19 +398,6 @@ impl Operator for SemijoinOperator { inner_op.close(); build_result?; - // Compute key column indices for the child (outer) schema. - let child_schema = self.child.schema().to_vec(); - self.key_col_indices = self - .key_vars - .iter() - .map(|kv| { - child_schema.iter().position(|v| v == kv).ok_or_else(|| { - QueryError::Internal(format!("key var {kv:?} not found in child schema")) - }) - }) - .collect::>>()?; - - self.child.open(ctx).await?; self.state = OperatorState::Open; Ok(()) } @@ -316,7 +408,7 @@ impl Operator for SemijoinOperator { } loop { - let input_batch = match self.child.next_batch(ctx).await? { + let input_batch = match self.next_child_batch(ctx).await? { Some(b) if !b.is_empty() => b, Some(_) => continue, None => { @@ -336,6 +428,7 @@ impl Operator for SemijoinOperator { self.child.close(); self.key_set.clear(); self.partial_key_sets.clear(); + self.buffered.clear(); self.state = OperatorState::Closed; } @@ -345,7 +438,7 @@ impl Operator for SemijoinOperator { } let mut count: u64 = 0; loop { - match self.child.next_batch(ctx).await? { + match self.next_child_batch(ctx).await? { Some(batch) if !batch.is_empty() => { let keep = self.keep_mask(ctx, &batch).await?; let kept = keep.iter().filter(|&&k| k).count() as u64; @@ -571,6 +664,57 @@ mod tests { assert!(op.partial_key_sets.is_empty()); } + /// A conjunction of triples seeds its build from an outer side with few + /// distinct keys (unbound keys included); more keys, or another body + /// shape, build the whole body. + #[tokio::test] + async fn build_is_seeded_only_for_few_keys_over_triples() { + let snapshot = LedgerSnapshot::genesis("test:main"); + let vars = VarRegistry::new(); + let ctx = ExecutionContext::new(&snapshot, &vars); + let rows = |n: u64| { + (0..n) + .map(|i| { + vec![ + Binding::encoded_sid(i), + Binding::encoded_sid(1), + if i % 2 == 0 { + Binding::Unbound + } else { + Binding::encoded_sid(2) + }, + ] + }) + .collect::>() + }; + let values = Pattern::Values { + vars: vec![VarId(0), VarId(1), VarId(2)], + rows: vec![], + }; + for (n, body, seeded) in [ + (SEEDED_BUILD_MAX_KEYS as u64, triple(), true), + (SEEDED_BUILD_MAX_KEYS as u64 + 1, triple(), false), + (2, values, false), + ] { + let mut op = SemijoinOperator::new( + Box::new(BatchSeedOperator::from_batch(batch(rows(n)))), + vec![body], + vec![VarId(0), VarId(1), VarId(2)], + false, + None, + PlanningContext::current(), + ); + op.open(&ctx).await.unwrap(); + assert_eq!(op.seeded, seeded, "{n} keys"); + let mut replayed = 0; + while let Some(batch) = op.next_child_batch(&ctx).await.unwrap() { + replayed += batch.len(); + } + assert_eq!(replayed as u64, n, "every outer row is replayed"); + op.close(); + } + } + #[tokio::test] async fn base_lookup_charges_only_distinct_keys_and_enforces_budget() { let snapshot = LedgerSnapshot::genesis("test:main"); From b91e779af4a1b815f3c189a95498591898b06dd2 Mon Sep 17 00:00:00 2001 From: bplatz Date: Sun, 4 Oct 2026 12:38:22 -0400 Subject: [PATCH 67/92] perf(query): batched subject probes skip leaflets by their directory keys The nested-loop join's subject lane covered its whole leaf range: it loaded every leaflet in the probed subject span and scanned its predicate column before checking whether any probed subject fell inside. A scattered probe set, such as the reifiers of a few annotations, spans most of the predicate's partition, so a few hundred probes walked thousands of leaflets. Leaflets are now rejected on their PSOT directory keys before they load. This is the same check the standalone subject probe made for single-predicate leaflets, now shared and extended to mixed-predicate leaflets: an end key bounds the predicate's subjects only where it carries that predicate. Leaflets that replay history keep loading, since replayed rows can lie outside the current keys. Single-predicate leaflets also skip the linear scan for their predicate run. Annotation benchmark (61M triples), warm local server: - P5 12.3ms -> 1.6ms - P8 12.5ms -> 1.5ms - P13 11ms -> 1.5ms - P14 5.5ms -> 0.6ms - P21 8.2ms -> 0.85ms - P9 17.7ms -> 8.0ms --- fluree-db-query/src/join.rs | 172 +++++++++++++++++++++++++----------- 1 file changed, 120 insertions(+), 52 deletions(-) diff --git a/fluree-db-query/src/join.rs b/fluree-db-query/src/join.rs index 93a998df80..6d467ebb95 100644 --- a/fluree-db-query/src/join.rs +++ b/fluree-db-query/src/join.rs @@ -287,6 +287,57 @@ pub(crate) fn prepare_leaf_for_scan( /// Binary search for the first row in `batch.s_id[start..end]` where `s_id >= target`. #[inline] +/// Whether a PSOT leaflet's current rows can hold one of `s_ids` (sorted) +/// under `p_id`, read from its directory keys alone, so a scattered probe set +/// decodes only the leaflets it touches. Keys order by `(p_id, s_id)`: an +/// end key bounds the predicate's subjects only where it carries `p_id`. +/// Rows a history replay restores can lie outside these keys. +fn psot_leaflet_may_hold( + entry: &fluree_db_binary_index::format::leaf::LeafletDirEntryV3, + p_id: u32, + s_ids: &[u64], +) -> bool { + use fluree_db_binary_index::format::run_record_v2::read_ordered_key_v2; + use fluree_db_binary_index::RunSortOrder; + let first = read_ordered_key_v2(RunSortOrder::Psot, &entry.first_key); + let last = read_ordered_key_v2(RunSortOrder::Psot, &entry.last_key); + if first.p_id > p_id || last.p_id < p_id { + return false; + } + let lo = if first.p_id == p_id { + first.s_id.as_u64() + } else { + 0 + }; + let hi = if last.p_id == p_id { + last.s_id.as_u64() + } else { + u64::MAX + }; + s_ids.partition_point(|&x| x < lo) < s_ids.partition_point(|&x| x <= hi) +} + +/// The row range of `p_id` in a PSOT leaflet; the whole leaflet when it holds +/// one predicate. +fn psot_predicate_run( + batch: &fluree_db_binary_index::ColumnBatch, + entry: &fluree_db_binary_index::format::leaf::LeafletDirEntryV3, + p_id: u32, +) -> (usize, usize) { + let row_count = batch.row_count; + if entry.p_const == Some(p_id) { + return (0, row_count); + } + let p_start = (0..row_count) + .position(|i| batch.p_id.get_or(i, 0) >= p_id) + .unwrap_or(row_count); + let p_end = (p_start..row_count) + .position(|i| batch.p_id.get_or(p_start + i, 0) > p_id) + .map(|offset| p_start + offset) + .unwrap_or(row_count); + (p_start, p_end) +} + fn lower_bound_s_id( batch: &fluree_db_binary_index::ColumnBatch, start: usize, @@ -2199,21 +2250,14 @@ impl NestedLoopJoinOperator { if entry.p_const.is_some() && entry.p_const != Some(p_id) { continue; } + if !needs_history_replay && !psot_leaflet_may_hold(entry, p_id, unique_s_ids) { + continue; + } let batch = leaf.load_leaflet(store, leaflet_idx, &proj, replay_to)?; ctx.check_cancelled()?; - let row_count = batch.row_count; - - // For PSOT, leaflets are sorted by p_id then s_id. - // Find the contiguous segment for our p_id. - let p_start = (0..row_count) - .position(|i| batch.p_id.get_or(i, 0) >= p_id) - .unwrap_or(row_count); - let p_end = (p_start..row_count) - .position(|i| batch.p_id.get_or(p_start + i, 0) > p_id) - .map(|offset| p_start + offset) - .unwrap_or(row_count); + let (p_start, p_end) = psot_predicate_run(&batch, entry, p_id); if p_start == p_end { continue; } @@ -3384,9 +3428,7 @@ fn batched_subject_probe_binary_uncharged( params: &SubjectProbeParams<'_>, mut probe_ops: Option<&mut ProbeOps>, ) -> Result> { - use fluree_db_binary_index::format::run_record_v2::{ - cmp_v2_for_order, read_ordered_key_v2, RunRecordV2, - }; + use fluree_db_binary_index::format::run_record_v2::{cmp_v2_for_order, RunRecordV2}; use fluree_db_binary_index::{ColumnProjection, RunSortOrder}; if params.subject_ids.is_empty() { @@ -3459,49 +3501,14 @@ fn batched_subject_probe_binary_uncharged( continue; } - // Directory-level subject skip: for a predicate-homogeneous leaflet - // the stored keys ascend by subject, so first_key/last_key bound its - // subject range. A scattered probe set spans the whole predicate - // partition but only touches a few leaflets — decline the rest here, - // before the (expensive) column decode + p-run scan, rather than - // after it (the `subj_start >= subj_end` check below). Only sound on - // homogeneous leaflets — a mixed-predicate leaflet resets s_id at each - // predicate boundary, so its key range isn't a clean subject interval - // — and only when not replaying history (a current-state key range can - // omit subjects that existed at an earlier `t`). - if entry.p_const == Some(p_id) && !needs_history_replay { - let lo = read_ordered_key_v2(RunSortOrder::Psot, &entry.first_key) - .s_id - .as_u64(); - let hi = read_ordered_key_v2(RunSortOrder::Psot, &entry.last_key) - .s_id - .as_u64(); - let a = unique_s_ids.partition_point(|&x| x < lo); - let b = unique_s_ids.partition_point(|&x| x <= hi); - if a >= b { - continue; - } + if !needs_history_replay && !psot_leaflet_may_hold(entry, p_id, &unique_s_ids) { + continue; } let batch = leaf.load_leaflet(store, leaflet_idx, &proj, replay_to)?; ctx.check_cancelled()?; - let row_count = batch.row_count; - // A p-homogeneous leaflet is entirely this predicate — its p-run is - // the whole leaflet, so skip the two linear scans that would walk - // every row only to rediscover [0, row_count). - let (p_start, p_end) = if entry.p_const == Some(p_id) { - (0, row_count) - } else { - let p_start = (0..row_count) - .position(|i| batch.p_id.get_or(i, 0) >= p_id) - .unwrap_or(row_count); - let p_end = (p_start..row_count) - .position(|i| batch.p_id.get_or(p_start + i, 0) > p_id) - .map(|offset| p_start + offset) - .unwrap_or(row_count); - (p_start, p_end) - }; + let (p_start, p_end) = psot_predicate_run(&batch, entry, p_id); if p_start == p_end { continue; } @@ -4143,6 +4150,67 @@ mod tests { } } + /// A leaflet is skipped only when no probed subject of the predicate can + /// lie between its directory keys, including where the leaflet spans + /// other predicates on either side. + #[test] + fn psot_leaflet_skip_reads_subject_bounds_per_predicate() { + use fluree_db_binary_index::format::leaf::LeafletDirEntryV3; + use fluree_db_binary_index::format::run_record_v2::{ + write_ordered_key_v2, RunRecordV2, ORDERED_KEY_V2_SIZE, + }; + use fluree_db_binary_index::RunSortOrder; + let key = |p_id: u32, s_id: u64| { + let mut buf = [0u8; ORDERED_KEY_V2_SIZE]; + let rec = RunRecordV2 { + s_id: SubjectId(s_id), + o_key: 0, + p_id, + t: 0, + o_i: 0, + o_type: 0, + g_id: 0, + }; + write_ordered_key_v2(RunSortOrder::Psot, &rec, &mut buf); + buf + }; + let entry = |first: (u32, u64), last: (u32, u64)| LeafletDirEntryV3 { + row_count: 1, + lead_group_count: 0, + first_key: key(first.0, first.1), + last_key: key(last.0, last.1), + p_const: None, + o_type_const: None, + flags: 0, + payload_offset: 0, + payload_len: 0, + column_refs: Vec::new(), + history_offset: 0, + history_len: 0, + history_min_t: 0, + history_max_t: 0, + }; + // One predicate: subjects 100..=200. + let only = entry((7, 100), (7, 200)); + assert!(psot_leaflet_may_hold(&only, 7, &[150])); + assert!(psot_leaflet_may_hold(&only, 7, &[100, 900])); + assert!(!psot_leaflet_may_hold(&only, 7, &[50, 250])); + assert!(!psot_leaflet_may_hold(&only, 8, &[150])); + // Starts in an earlier predicate: p=7 rows run from subject 0 to 40. + let starts_before = entry((5, 900), (7, 40)); + assert!(psot_leaflet_may_hold(&starts_before, 7, &[3])); + assert!(!psot_leaflet_may_hold(&starts_before, 7, &[41])); + // Ends in a later predicate: p=7 rows run from subject 60 upward. + let ends_after = entry((7, 60), (9, 2)); + assert!(psot_leaflet_may_hold(&ends_after, 7, &[10_000])); + assert!(!psot_leaflet_may_hold(&ends_after, 7, &[59])); + // Spans p=7 entirely: any subject. + let spans = entry((5, 900), (9, 2)); + assert!(psot_leaflet_may_hold(&spans, 7, &[1])); + assert!(!psot_leaflet_may_hold(&spans, 4, &[1])); + assert!(!psot_leaflet_may_hold(&spans, 10, &[1])); + } + #[test] fn test_bind_instruction_creation() { // Left schema: [?s, ?name] From 42b519173828bbb43738e132835407d89bf32ae9 Mon Sep 17 00:00:00 2001 From: bplatz Date: Mon, 14 Sep 2026 19:30:04 -0400 Subject: [PATCH 68/92] perf(query): reuse independent scans and count Cartesian joins Ported from perf/batched-wildcard-joins. NestedLoopJoinOperator's count drain now tries the product of independent sides first, then falls back to the subject-probe count drain added on main since. --- fluree-db-api/Cargo.toml | 6 + .../tests/it_independent_scan_replay.rs | 537 ++++++++++++++++++ fluree-db-query/src/dataset_operator.rs | 29 +- fluree-db-query/src/execute/operator_tree.rs | 3 +- fluree-db-query/src/join.rs | 127 ++++- fluree-db-query/src/join/replay.rs | 166 ++++++ fluree-db-query/src/operator.rs | 23 +- 7 files changed, 865 insertions(+), 26 deletions(-) create mode 100644 fluree-db-api/tests/it_independent_scan_replay.rs create mode 100644 fluree-db-query/src/join/replay.rs diff --git a/fluree-db-api/Cargo.toml b/fluree-db-api/Cargo.toml index 0e2382156c..3d60023239 100644 --- a/fluree-db-api/Cargo.toml +++ b/fluree-db-api/Cargo.toml @@ -193,6 +193,12 @@ name = "it_wildcard_batched_join" path = "tests/it_wildcard_batched_join.rs" required-features = ["native"] +# Standalone: independent scan replay routing and the global fast-path switch. +[[test]] +name = "it_independent_scan_replay" +path = "tests/it_independent_scan_replay.rs" +required-features = ["native"] + # Standalone (NOT in grp_import): re-execs itself and lowers RLIMIT_NOFILE # hard+soft in the child, which must not share a process with other tests. [[test]] diff --git a/fluree-db-api/tests/it_independent_scan_replay.rs b/fluree-db-api/tests/it_independent_scan_replay.rs new file mode 100644 index 0000000000..430480450f --- /dev/null +++ b/fluree-db-api/tests/it_independent_scan_replay.rs @@ -0,0 +1,537 @@ +//! Standalone: independent scan routing capture and the process-global fast-path switch. +#![cfg(feature = "native")] +mod support; + +use fluree_db_api::{set_fast_paths_disabled, Fluree, FlureeBuilder, ReindexOptions}; +use fluree_db_ledger::LedgerState; +use serde_json::{json, Value}; + +const PREFIX: &str = "PREFIX ex: "; +static SERIAL: tokio::sync::Mutex<()> = tokio::sync::Mutex::const_new(()); + +const EVENT: &str = "independent scan replay engaged"; +const COUNT_EVENT: &str = "independent join count product engaged"; + +#[derive(Clone, Copy, Debug)] +enum Routing { + MustFire, + MustNotFire, +} + +struct Reset; +impl Drop for Reset { + fn drop(&mut self) { + set_fast_paths_disabled(false); + } +} + +async fn check( + fluree: &Fluree, + ledger: &LedgerState, + query: &str, + expected_len: usize, + routing: Routing, +) -> Value { + let query = format!("{PREFIX}{query}"); + let (spans, _guard) = support::span_capture::init_test_tracing(); + set_fast_paths_disabled(false); + let fast = support::query_sparql(fluree, ledger, &query) + .await + .expect(&query) + .to_sparql_json(&ledger.snapshot) + .unwrap(); + let fast_rows = normalized(&fast); + assert_eq!(fast_rows.len(), expected_len, "{query}: {fast}"); + let fired = spans.has_event(EVENT); + assert_eq!( + fired, + matches!(routing, Routing::MustFire), + "{routing:?}: {query}" + ); + drop(_guard); + let (spans, _guard) = support::span_capture::init_test_tracing(); + set_fast_paths_disabled(true); + let generic = support::query_sparql(fluree, ledger, &query) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + assert_eq!(fast_rows, normalized(&generic), "{query}"); + if query.contains("ORDER BY") { + assert_eq!( + fast["results"]["bindings"], generic["results"]["bindings"], + "ordered output" + ); + } + assert!(!spans.has_event(EVENT), "disabled lane fired: {query}"); + set_fast_paths_disabled(false); + fast +} + +async fn check_count( + fluree: &Fluree, + ledger: &LedgerState, + query: &str, + expected: u64, + product: bool, +) { + for disabled in [false, true] { + set_fast_paths_disabled(disabled); + let (spans, guard) = support::span_capture::init_test_tracing(); + let value = support::query_sparql(fluree, ledger, &format!("{PREFIX}{query}")) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + assert_eq!( + value["results"]["bindings"][0]["total"]["value"], + expected.to_string(), + "{query}" + ); + assert_eq!( + spans.has_event(COUNT_EVENT), + product && !disabled, + "count routing: {query}" + ); + drop(guard); + } + set_fast_paths_disabled(false); +} + +async fn check_count_lifecycle(ledger: &LedgerState) { + use fluree_db_query::{ + binary_scan::EmitMask, + binding::Batch, + context::ExecutionContext, + ir::{Ref, Term, TriplePattern}, + join::NestedLoopJoinOperator, + operator::Operator, + seed::BatchSeedOperator, + TemporalMode, VarRegistry, + }; + use std::sync::Arc; + let mut vars = VarRegistry::new(); + let subject = vars.get_or_insert("?s"); + let object = vars.get_or_insert("?o"); + let store = ledger + .binary_store + .as_ref() + .unwrap() + .0 + .clone() + .downcast::() + .unwrap(); + let ctx = ExecutionContext::new(&ledger.snapshot, &vars) + .with_binary_store(store, 0) + .with_batch_size(1); + let pattern = TriplePattern::new( + Ref::Var(subject), + Ref::Sid( + ledger + .snapshot + .encode_iri("http://example.org/items") + .unwrap(), + ), + Term::Var(object), + ); + let make = |rows| { + NestedLoopJoinOperator::new( + Box::new(BatchSeedOperator::from_batch(Batch::empty_schema_with_len( + rows, + ))), + Arc::from([]), + pattern.clone(), + None, + Vec::new(), + EmitMask::ALL, + TemporalMode::Current, + ) + }; + let mut fresh = make(3); + fresh.open(&ctx).await.unwrap(); + assert_eq!(fresh.drain_count(&ctx).await.unwrap(), Some(9)); + assert!(fresh.next_batch(&ctx).await.unwrap().is_none()); + fresh.close(); + let mut partial = make(3); + partial.open(&ctx).await.unwrap(); + let mut rows = partial.next_batch(&ctx).await.unwrap().unwrap().len(); + assert_eq!(rows, 1); + assert_eq!( + partial.drain_count(&ctx).await.unwrap(), + None, + "must decline after partial output" + ); + while let Some(batch) = partial.next_batch(&ctx).await.unwrap() { + rows += batch.len(); + } + assert_eq!(rows, 9); + partial.close(); + #[cfg(target_pointer_width = "64")] + { + // Empty-schema batches carry multiplicity without allocating the rows. + let mut overflow = make(usize::MAX); + overflow.open(&ctx).await.unwrap(); + assert!(overflow + .drain_count(&ctx) + .await + .unwrap_err() + .to_string() + .contains("overflow")); + overflow.close(); + } +} + +fn normalized(result: &Value) -> Vec { + let mut rows: Vec<_> = result["results"]["bindings"] + .as_array() + .unwrap() + .iter() + .map(|r| serde_json::to_string(r).unwrap()) + .collect(); + rows.sort(); + rows +} + +#[tokio::test(flavor = "current_thread")] +async fn independent_scans_preserve_multiplicity_filters_and_fallbacks() { + let _serial = SERIAL.lock().await; + assert!(std::env::var_os("FLUREE_DISABLE_QUERY_FAST_PATHS").is_none()); + let _reset = Reset; + let dir = tempfile::tempdir().unwrap(); + let fluree = FlureeBuilder::file(dir.path().to_string_lossy().to_string()) + .build() + .unwrap(); + let alias = "replay/test:main"; + let ledger = fluree.create_ledger(alias).await.unwrap(); + fluree + .insert( + ledger, + &json!({ + "@context":{"ex":"http://example.org/"}, + "@graph":[ + {"@id":"ex:a", "ex:driver":1, "ex:items":{"@list":[10,10,20]}, + "ex:edge":{"@id":"ex:b","@annotation":{"ex:source":"one"}}}, + {"@id":"ex:b", "ex:driver":2, + "ex:edge":{"@id":"ex:c","@annotation":{"ex:source":"two"}}}, + {"@id":"ex:c", "ex:driver":3, "ex:label":[ + {"@value":"bonjour","@language":"fr"}, {"@value":"hello","@language":"en"}]} + ] + }), + ) + .await + .unwrap(); + fluree + .reindex(alias, ReindexOptions::default()) + .await + .unwrap(); + let ledger = fluree.ledger(alias).await.unwrap(); + check_count_lifecycle(&ledger).await; + use Routing::*; + let body = "?l ex:driver ?n . ?s ex:items ?o"; + let q = format!("SELECT ?l ?s ?o WHERE {{ {body} }}"); + check_count( + &fluree, + &ledger, + &format!("SELECT (COUNT(*) AS ?total) WHERE {{ {body} }}"), + 9, + true, + ) + .await; + check_count( + &fluree, + &ledger, + &format!("SELECT (COUNT(*) AS ?total) WHERE {{ {body} FILTER(?n + ?o > 21) }}"), + 2, + false, + ) + .await; + check_count( + &fluree, + &ledger, + &format!("SELECT (COUNT(DISTINCT ?l) AS ?total) WHERE {{ {body} }}"), + 3, + false, + ) + .await; + check_count( + &fluree, + &ledger, + "SELECT (COUNT(*) AS ?total) WHERE { ?l ex:driver ?x . ?l ex:items ?o }", + 3, + false, + ) + .await; + check_count( + &fluree, + &ledger, + "SELECT (COUNT(*) AS ?total) WHERE { ?l ex:driver ?x . ?s ex:missing ?o }", + 0, + true, + ) + .await; + + let result = check(&fluree, &ledger, &q, 9, MustFire).await; + assert_eq!( + result["results"]["bindings"] + .as_array() + .unwrap() + .iter() + .filter(|r| r["o"]["value"] == "10") + .count(), + 6 + ); + check( + &fluree, + &ledger, + &format!("SELECT ?o WHERE {{ {body} }}"), + 9, + MustFire, + ) + .await; + check( + &fluree, + &ledger, + &format!("SELECT ?l ?o WHERE {{ {body} FILTER(?n + ?o > 21) }}"), + 2, + MustFire, + ) + .await; + check(&fluree, &ledger, &format!("SELECT ?l ?o ?v WHERE {{ {body} BIND(?n + ?o AS ?v) }} ORDER BY ?l ?o LIMIT 4 OFFSET 2"), 4, MustFire).await; + check( + &fluree, + &ledger, + "SELECT ?l ?v WHERE { ?l ex:driver ?n . ?s ex:label ?v }", + 6, + MustFire, + ) + .await; + check( + &fluree, + &ledger, + "SELECT ?l ?o WHERE { ?l ex:driver ?n . ?l ex:items ?o }", + 3, + MustNotFire, + ) + .await; + check( + &fluree, + &ledger, + "SELECT ?l ?o WHERE { ?l ex:driver ?n . ?s ex:missing ?o }", + 0, + MustNotFire, + ) + .await; + // The two annotation components are independent, as in S21/S22. Their + // source comparisons still run for every joined pair. + check(&fluree, &ledger, "SELECT ?x ?y WHERE { << ?a ex:edge ?b >> ex:source ?x . << ?c ex:edge ?d >> ex:source ?y . FILTER(STR(?x) > STR(?y)) }", 1, MustFire).await; + + let json_query = json!({"@context":{"ex":"http://example.org/"}, + "select":["?l","?s","?o"], "where":[ + {"@id":"?l","ex:driver":"?n"}, {"@id":"?s","ex:items":"?o"}]}); + let (spans, guard) = support::span_capture::init_test_tracing(); + let twin = support::query_jsonld(&fluree, &ledger, &json_query) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + assert_eq!(normalized(&twin), normalized(&result)); + assert!(spans.has_event(EVENT)); + drop(guard); + + let mut json_count = json_query.clone(); + json_count["select"] = json!(["(as (count *) ?total)"]); + let (spans, guard) = support::span_capture::init_test_tracing(); + let count = support::query_jsonld(&fluree, &ledger, &json_count) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + assert_eq!(count["results"]["bindings"][0]["total"]["value"], "9"); + assert!(spans.has_event(COUNT_EVENT)); + drop(guard); + + let mut restricted = json_query.clone(); + restricted["from"] = json!(alias); + restricted["opts"] = json!({"default-allow":true, "policy":[{ + "@id":"ex:hideItems", "f:required":true, "f:action":"f:view", + "f:onProperty":[{"@id":"http://example.org/items"}], "f:allow":false + }]}); + let (spans, guard) = support::span_capture::init_test_tracing(); + let hidden = fluree + .query_connection(&restricted) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + assert!(normalized(&hidden).is_empty()); + assert!( + !spans.has_event(EVENT), + "policy-filtered scans must decline" + ); + drop(guard); + // A post-join volatile filter still runs for every pair during replay. + check( + &fluree, + &ledger, + &format!("SELECT ?l ?o WHERE {{ {body} FILTER(RAND() >= 0) }}"), + 9, + MustFire, + ) + .await; + + // More than one output batch, followed by the cache capacity boundary. + // The planner drives these from the three-row predicate. + let wide_alias = "replay/wide:main"; + let wide = fluree.create_ledger(wide_alias).await.unwrap(); + fluree + .insert( + wide, + &json!({"@context":{"ex":"http://example.org/"}, + "@graph":[{"@id":"ex:a","ex:driver":1}, {"@id":"ex:b","ex:driver":2}, + {"@id":"ex:c","ex:driver":3}, {"@id":"ex:wide", + "ex:fits":{"@list":(0..8192).collect::>()}, + "ex:overflows":{"@list":(0..8193).collect::>()}}]}), + ) + .await + .unwrap(); + fluree + .reindex(wide_alias, ReindexOptions::default()) + .await + .unwrap(); + let wide = fluree.ledger(wide_alias).await.unwrap(); + check( + &fluree, + &wide, + "SELECT ?l ?v WHERE { ?l ex:driver ?n . ?s ex:fits ?v }", + 3 * 8192, + MustFire, + ) + .await; + check( + &fluree, + &wide, + "SELECT ?l ?v WHERE { ?l ex:driver ?n . ?s ex:overflows ?v }", + 3 * 8193, + MustNotFire, + ) + .await; + // A small LIMIT must not eagerly drain or fill the right cache. + check( + &fluree, + &wide, + "SELECT ?l ?v WHERE { ?l ex:driver ?n . ?s ex:fits ?v } LIMIT 1", + 1, + MustNotFire, + ) + .await; + + let receipt = fluree + .insert( + ledger, + &json!({"@context":{"ex":"http://example.org/"}, + "@id":"ex:d", "ex:driver":4}), + ) + .await + .unwrap(); + check(&fluree, &receipt.ledger, &q, 12, MustNotFire).await; + check_count( + &fluree, + &receipt.ledger, + &format!("SELECT (COUNT(*) AS ?total) WHERE {{ {body} }}"), + 12, + false, + ) + .await; +} + +/// Reproducible local A/B probe benchmark. Run separately in dev-fast/release; +/// the fixture and expected counts are independent of StarBench and its ledger. +#[tokio::test(flavor = "current_thread")] +#[ignore = "local performance experiment; run with --profile dev-fast --ignored --nocapture"] +async fn independent_scan_replay_benchmark() { + let _serial = SERIAL.lock().await; + use std::fmt::Write; + use std::time::Instant; + let _reset = Reset; + let dir = tempfile::tempdir().unwrap(); + let data = tempfile::tempdir().unwrap(); + let mut ttl = String::from("@prefix ex: .\n"); + let n: usize = std::env::var("FLUREE_REPLAY_BENCH_KEYS") + .map(|s| s.parse().expect("positive key count")) + .unwrap_or(6000); + assert!(n > 0); + for i in 0..n { + writeln!(ttl, "ex:l{i} ex:left {i} .").unwrap(); + } + for i in 0..619 { + writeln!(ttl, "ex:r{i} ex:right {i} .").unwrap(); + } + std::fs::write(data.path().join("data.ttl"), ttl).unwrap(); + let fluree = FlureeBuilder::file(dir.path().to_string_lossy().to_string()) + .build() + .unwrap(); + let alias = "replay/bench:main"; + fluree + .create(alias) + .import(data.path()) + .threads(1) + .memory_budget_mb(256) + .execute() + .await + .unwrap(); + let ledger = fluree.ledger(alias).await.unwrap(); + for (name, body, expected) in [ + ("cross-product", "?l ex:left ?x . ?r ex:right ?y", n * 619), + ( + "filtered-product", + "?l ex:left ?x . ?r ex:right ?y . FILTER(?x > ?y)", + (0..n).map(|i| i.min(619)).sum(), + ), + ] { + let query = format!("{PREFIX} SELECT (COUNT(*) AS ?n) WHERE {{ {body} }}"); + set_fast_paths_disabled(false); + let (spans, tracing_guard) = support::span_capture::init_test_tracing(); + support::query_sparql(&fluree, &ledger, &query) + .await + .unwrap(); + assert!( + spans.has_event(if name == "cross-product" { + COUNT_EVENT + } else { + EVENT + }), + "benchmark must engage: {name}" + ); + drop(tracing_guard); + let mut times = [Vec::new(), Vec::new()]; + for round in 0..8 { + // Alternate mode order; exclude the first warm-up pair. + for mode in [round % 2, 1 - round % 2] { + set_fast_paths_disabled(mode == 1); + let start = Instant::now(); + let result = support::query_sparql(&fluree, &ledger, &query) + .await + .unwrap() + .to_sparql_json(&ledger.snapshot) + .unwrap(); + let elapsed = start.elapsed().as_secs_f64(); + assert_eq!( + result["results"]["bindings"][0]["n"]["value"], + expected.to_string() + ); + if round != 0 { + times[mode].push(elapsed); + } + } + } + for t in &mut times { + t.sort_by(f64::total_cmp); + } + println!( + "{name}: optimized={:.6}s generic={:.6}s speedup={:.2}x (median of 7, {n} keys)", + times[0][3], + times[1][3], + times[1][3] / times[0][3] + ); + } +} diff --git a/fluree-db-query/src/dataset_operator.rs b/fluree-db-query/src/dataset_operator.rs index c201dc9d28..0218c47818 100644 --- a/fluree-db-query/src/dataset_operator.rs +++ b/fluree-db-query/src/dataset_operator.rs @@ -37,7 +37,7 @@ use crate::error::{QueryError, Result}; use crate::ir::triple::TriplePattern; use crate::object_binding::{equality_norm, normalize_for_key, EqualityNorm}; use crate::operator::inline::{extend_schema, InlineOperator}; -use crate::operator::{BoxedOperator, Operator, OperatorState}; +use crate::operator::{count_operator, BoxedOperator, Operator, OperatorState}; use crate::sort::SortSpec; use crate::temporal_mode::TemporalMode; use crate::var_registry::VarId; @@ -503,24 +503,6 @@ fn sid_to_iri_match( )) } -/// Count a single member to exhaustion, preferring its `drain_count` -/// (count-only, no binding materialization) and falling back to a streaming -/// `next_batch` row count when the member declines count-only mode. -async fn count_member(op: &mut BoxedOperator, ctx: &ExecutionContext<'_>) -> Result { - if let Some(n) = op.drain_count(ctx).await? { - return Ok(n); - } - let mut n: u64 = 0; - while let Some(batch) = op.next_batch(ctx).await? { - ctx.check_cancelled()?; - n = n - .checked_add(batch.len() as u64) - .ok_or_else(|| QueryError::execution("COUNT(*) overflow in dataset drain_count"))?; - } - ctx.check_cancelled()?; - Ok(n) -} - #[async_trait] impl Operator for DatasetOperator { fn schema(&self) -> &[VarId] { @@ -798,11 +780,14 @@ impl Operator for DatasetOperator { let n = match &graphs { ActiveGraphs::Many(g) => { let graph_ctx = ctx.with_graph_ref(g[self.current_member]); - count_member(&mut self.members[self.current_member].operator, &graph_ctx) - .await? + count_operator( + self.members[self.current_member].operator.as_mut(), + &graph_ctx, + ) + .await? } ActiveGraphs::Single => { - count_member(&mut self.members[self.current_member].operator, ctx).await? + count_operator(self.members[self.current_member].operator.as_mut(), ctx).await? } }; total = total diff --git a/fluree-db-query/src/execute/operator_tree.rs b/fluree-db-query/src/execute/operator_tree.rs index 2a34655a2e..faad96167e 100644 --- a/fluree-db-query/src/execute/operator_tree.rs +++ b/fluree-db-query/src/execute/operator_tree.rs @@ -2244,7 +2244,8 @@ fn detect_union_star_count_all( /// an operational escape hatch when triaging a suspected fast-path bug. /// It is NOT a tuning knob: runtime operator-internal optimizations /// (cursor selection, fixed-predicate batched joins) are unaffected. The -/// wildcard-predicate join lane also honors this switch for differential tests. +/// wildcard-predicate join, independent-scan replay and independent-join count +/// lanes also honor this switch for differential tests. static FAST_PATHS_DISABLED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); diff --git a/fluree-db-query/src/join.rs b/fluree-db-query/src/join.rs index 6d467ebb95..f5cb339208 100644 --- a/fluree-db-query/src/join.rs +++ b/fluree-db-query/src/join.rs @@ -4,6 +4,7 @@ //! where left results drive right scans. It enforces var unification - shared //! vars between left and right must match exactly. +mod replay; mod wildcard; use crate::binary_scan::EmitMask; @@ -633,10 +634,14 @@ pub struct NestedLoopJoinOperator { /// draining an entire left batch up front, which makes top-level LIMITs much /// more responsive on wide joins like `?s rdf:type ; ?p ?o`. active_right_scan: Option, - /// Left-row provenance for `active_right_scan`. + /// Left-row provenance for the active right scan or replay. active_right_batch_ref: Option, - /// Left-row index for `active_right_scan`. + /// Left-row index for the active right scan or replay. active_right_left_row: usize, + /// Small independent scans can be replayed after the first streamed pass. + right_replay: Option, + active_replay_batch: Option, + logged_replay: bool, /// Optional object bounds for range filter pushdown object_bounds: Option, wildcard_direction: Option, @@ -956,6 +961,9 @@ impl NestedLoopJoinOperator { active_right_scan: None, active_right_batch_ref: None, active_right_left_row: 0, + right_replay: None, + active_replay_batch: None, + logged_replay: false, object_bounds, wildcard_direction, wildcard_stream: None, @@ -1351,6 +1359,24 @@ impl NestedLoopJoinOperator { .collect() } + /// Ordinary indexed scans whose inputs cannot depend on a driving row. + fn independent_scan_eligible(&self, ctx: &ExecutionContext<'_>) -> bool { + self.bind_instructions.is_empty() + && self.unify_instructions.is_empty() + && self.right_scan_inline_ops.is_empty() + && !self.left.is_identity_seed() + && !crate::execute::fast_paths_disabled() + && self.mode.is_current() + && ctx.from_t.is_none() + && ctx.binary_store.is_some() + && ctx.dataset.is_none() + && !ctx.is_multi_ledger() + && !ctx.eager_materialization + && !ctx.reasoning_active + && ctx.policy_enforcer.as_ref().is_none_or(|p| p.is_root()) + && ctx.overlay().is_effectively_empty() + } + fn bounds_for_row( &self, _left_batch: &Batch, @@ -1419,6 +1445,15 @@ impl Operator for NestedLoopJoinOperator { self.active_right_batch_ref = None; self.active_right_left_row = 0; self.logged_runtime_mode = false; + self.active_replay_batch = None; + self.logged_replay = false; + // No shared variables means substitution cannot change this scan. + // Do not cache pushed FILTERs: even a variable-free expression may be + // volatile (RAND/UUID) or inspect correlated state through EXISTS. + // Keep the initial scope to ordinary indexed single-graph snapshots. + self.right_replay = self + .independent_scan_eligible(ctx) + .then(|| replay::ScanReplay::new(ctx)); tracing::trace!( left_schema_cols = self.left_schema.len(), @@ -1538,6 +1573,27 @@ impl Operator for NestedLoopJoinOperator { continue; } + // Replay one batch at a time through the existing combination and + // inline-expression path. Duplicate rows and output backpressure + // have exactly the same handling as a freshly opened scan. + if let Some(index) = self.active_replay_batch { + let batch = self.right_replay.as_ref().and_then(|r| r.batch(index)); + if let Some(batch) = batch { + self.active_replay_batch = Some(index + 1); + self.pending_output.push_back(( + self.active_right_batch_ref + .clone() + .expect("replay left batch"), + self.active_right_left_row, + batch, + )); + } else { + self.active_replay_batch = None; + self.active_right_batch_ref = None; + } + continue; + } + // 3. Resume an in-flight right scan for the current left row. if let Some(scan) = &mut self.active_right_scan { ctx.check_cancelled()?; @@ -1546,6 +1602,14 @@ impl Operator for NestedLoopJoinOperator { match next { Some(batch) if !batch.is_empty() => { + if let Some(replay) = &mut self.right_replay { + if !replay.record(&batch) { + self.right_replay = None; + tracing::debug!( + "independent scan replay declined at capacity or binding guard" + ); + } + } let batch_ref = self .active_right_batch_ref .clone() @@ -1559,6 +1623,9 @@ impl Operator for NestedLoopJoinOperator { } Some(_) => continue, None => { + if let Some(replay) = &mut self.right_replay { + replay.finish(); + } if let Some(mut scan) = self.active_right_scan.take() { scan.close(); } @@ -1722,6 +1789,20 @@ impl Operator for NestedLoopJoinOperator { // Non-batched path: existing per-row join let batch_idx = self.ensure_current_batch_stored(); let batch_ref = BatchRef::Stored(batch_idx); + if self + .right_replay + .as_ref() + .is_some_and(replay::ScanReplay::is_complete) + { + self.active_replay_batch = Some(0); + self.active_right_batch_ref = Some(batch_ref); + self.active_right_left_row = left_row; + if !self.logged_replay { + tracing::debug!("independent scan replay engaged"); + self.logged_replay = true; + } + continue; + } let left_batch = self.stored_left_batches.last().unwrap(); let bound_pattern = self.substitute_pattern_with_store(left_batch, left_row, cached_gv.as_ref())?; @@ -1744,6 +1825,46 @@ impl Operator for NestedLoopJoinOperator { } async fn drain_count(&mut self, ctx: &ExecutionContext<'_>) -> Result> { + // Only a fresh stream can be counted as a product. Once next_batch has + // run, the remaining rows may start in the middle of a driving row. + // Pair-dependent FILTER/BIND expressions must use normal row evaluation. + if self.state == OperatorState::Open + && !self.logged_runtime_mode + && self.inline_ops.is_empty() + && self.independent_scan_eligible(ctx) + { + ctx.check_cancelled()?; + let left_count = crate::operator::count_operator(self.left.as_mut(), ctx).await?; + let right_count = if left_count == 0 { + 0 + } else { + let mut right = make_right_scan( + self.right_pattern.clone(), + self.object_bounds.clone(), + self.right_emit, + Vec::new(), + self.right_index_hint, + ctx, + self.mode, + ); + right.open(ctx).await?; + let count = crate::operator::count_operator(right.as_mut(), ctx).await; + right.close(); + count? + }; + let count = left_count + .checked_mul(right_count) + .ok_or_else(|| QueryError::execution("COUNT(*) overflow in independent join"))?; + ctx.check_cancelled()?; + self.right_replay = None; + self.state = OperatorState::Exhausted; + tracing::debug!( + left_count, + right_count, + "independent join count product engaged" + ); + return Ok(Some(count)); + } // Reuse the regular driver and all its runtime admission/fallback // rules. Only subject probes without BIND on joined rows avoid emission. if !self.state.can_next() @@ -1850,6 +1971,8 @@ impl Operator for NestedLoopJoinOperator { fn close(&mut self) { self.left.close(); + self.right_replay = None; + self.active_replay_batch = None; self.wildcard_stream = None; self.current_left_batch = None; self.current_left_batch_stored_idx = None; diff --git a/fluree-db-query/src/join/replay.rs b/fluree-db-query/src/join/replay.rs new file mode 100644 index 0000000000..f19432d03c --- /dev/null +++ b/fluree-db-query/src/join/replay.rs @@ -0,0 +1,166 @@ +//! Bounded, query-local replay of an independent triple scan. +//! +//! Record while the first scan streams; only replay after EOF. Overflow or a +//! heap-backed binding abandons the cache, leaving the original scan untouched. + +use crate::binding::{Batch, Binding}; +use crate::context::ExecutionContext; +use fluree_db_core::QueryCancellation; + +const MAX_ROWS: usize = 8192; +const MAX_BYTES: usize = 4 * 1024 * 1024; + +pub(super) struct ScanReplay { + batches: Vec, + rows: usize, + bytes: usize, + complete: bool, + cancellation: QueryCancellation, +} + +impl ScanReplay { + pub(super) fn new(ctx: &ExecutionContext<'_>) -> Self { + Self { + batches: Vec::new(), + rows: 0, + bytes: 0, + complete: false, + cancellation: ctx.cancellation.clone(), + } + } + + /// False means the caller must drop the cache and keep streaming normally. + pub(super) fn record(&mut self, batch: &Batch) -> bool { + let bytes = batch + .len() + .saturating_mul(batch.schema().len()) + .saturating_mul(std::mem::size_of::()) + // Allow for the outer Vec's spare capacity as it grows. + .saturating_add(2 * std::mem::size_of::()) + .saturating_add(batch.schema().len() * std::mem::size_of::>()); + let budget = self + .cancellation + .memory_limit() + .unwrap_or_else(crate::context::query_memory_budget_bytes); + if self.rows.saturating_add(batch.len()) > MAX_ROWS + || self.bytes.saturating_add(bytes) > MAX_BYTES + || (budget != 0 && self.cancellation.allocated_bytes().saturating_add(bytes) > budget) + { + return false; + } + // Keep the memory ceiling meaningful even for very large strings, + // vectors and decimals. Those scans retain the ordinary per-row path. + for col in 0..batch.schema().len() { + for row in 0..batch.len() { + if !matches!( + batch.get_by_col(row, col), + Binding::EncodedSid { .. } + | Binding::EncodedPid { .. } + | Binding::EncodedLit { .. } + | Binding::Unbound + | Binding::Poisoned + ) { + return false; + } + } + } + self.batches.push(batch.clone()); + self.rows += batch.len(); + self.bytes += bytes; + self.cancellation.record_alloc(bytes); + true + } + + pub(super) fn finish(&mut self) { + self.complete = true; + } + + pub(super) fn is_complete(&self) -> bool { + self.complete + } + + pub(super) fn batch(&self, index: usize) -> Option { + self.batches.get(index).cloned() + } +} + +impl Drop for ScanReplay { + fn drop(&mut self) { + self.cancellation.release(self.bytes); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::var_registry::{VarId, VarRegistry}; + use fluree_db_core::LedgerSnapshot; + use std::sync::Arc; + + fn batch(rows: usize) -> Batch { + Batch::new( + Arc::from([VarId(0)]), + vec![vec![Binding::EncodedPid { p_id: 17 }; rows]], + ) + .unwrap() + } + + #[test] + fn replay_requires_eof_and_releases_memory_on_drop() { + let snapshot = LedgerSnapshot::genesis("replay:main"); + let vars = VarRegistry::new(); + let ctx = + ExecutionContext::new(&snapshot, &vars).with_cancellation(QueryCancellation::new()); + let mut replay = ScanReplay::new(&ctx); + assert!(replay.record(&batch(2))); + assert!(replay.record(&batch(3))); + assert!(!replay.is_complete()); + assert!(ctx.mem_used() > 0); + replay.finish(); + assert!(replay.is_complete()); + assert_eq!(replay.batch(0).unwrap().len(), 2); + assert_eq!(replay.batch(1).unwrap().len(), 3); + assert!(replay.batch(2).is_none()); + drop(replay); + assert_eq!(ctx.mem_used(), 0); + } + + #[test] + fn capacity_and_memory_budget_decline_without_retaining_new_rows() { + let snapshot = LedgerSnapshot::genesis("replay:main"); + let vars = VarRegistry::new(); + let ctx = + ExecutionContext::new(&snapshot, &vars).with_cancellation(QueryCancellation::new()); + let mut replay = ScanReplay::new(&ctx); + assert!(replay.record(&batch(MAX_ROWS))); + let used = ctx.mem_used(); + assert!(!replay.record(&batch(1))); + assert_eq!(ctx.mem_used(), used); + drop(replay); + assert_eq!(ctx.mem_used(), 0); + ctx.cancellation.set_memory_limit(1); + assert!(!ScanReplay::new(&ctx).record(&batch(1))); + assert_eq!(ctx.mem_used(), 0); + } + + #[test] + fn empty_schema_multiplicity_and_materialized_binding_guard() { + let snapshot = LedgerSnapshot::genesis("replay:main"); + let vars = VarRegistry::new(); + let ctx = ExecutionContext::new(&snapshot, &vars); + let mut replay = ScanReplay::new(&ctx); + assert!(replay.record(&Batch::empty_schema_with_len(7))); + replay.finish(); + assert_eq!(replay.batch(0).unwrap().len(), 7); + let materialized = Batch::single_row( + Arc::from([VarId(0)]), + vec![Binding::Iri(Arc::from("http://example.org/large"))], + ) + .unwrap(); + assert!(!ScanReplay::new(&ctx).record(&materialized)); + let mut empty = ScanReplay::new(&ctx); + empty.finish(); + assert!(empty.is_complete()); + assert!(empty.batch(0).is_none()); + } +} diff --git a/fluree-db-query/src/operator.rs b/fluree-db-query/src/operator.rs index dfd8440f8c..4e44eb97b0 100644 --- a/fluree-db-query/src/operator.rs +++ b/fluree-db-query/src/operator.rs @@ -8,7 +8,7 @@ pub mod inline; use crate::binding::{Batch, Binding}; use crate::context::ExecutionContext; -use crate::error::Result; +use crate::error::{QueryError, Result}; use crate::sort::SortSpec; use crate::var_registry::VarId; use async_trait::async_trait; @@ -287,3 +287,24 @@ pub fn trim_batch(out_schema: &Option>, batch: Batch) -> Option Some(batch), } } + +/// Count an operator to exhaustion, preferring its `drain_count` +/// (count-only, no binding materialization) and falling back to a streaming +/// `next_batch` row count when the operator declines count-only mode. +pub(crate) async fn count_operator( + op: &mut dyn Operator, + ctx: &ExecutionContext<'_>, +) -> Result { + if let Some(n) = op.drain_count(ctx).await? { + return Ok(n); + } + let mut n: u64 = 0; + while let Some(batch) = op.next_batch(ctx).await? { + ctx.check_cancelled()?; + n = n + .checked_add(batch.len() as u64) + .ok_or_else(|| QueryError::execution("COUNT(*) overflow while counting operator"))?; + } + ctx.check_cancelled()?; + Ok(n) +} From 088da2c5f5ab6f05a801291836c5a5803d9b5869 Mon Sep 17 00:00:00 2001 From: bplatz Date: Sun, 4 Oct 2026 18:28:35 -0400 Subject: [PATCH 69/92] fix: TriG default-graph blocks, escaped local names and canonical N-Triples MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit TriG ingest (insert, upsert, import) refused an unlabeled `{ … }` block, the default graph's wrapped form that most W3C TriG documents use. Its statements now read as default-graph Turtle, including a last statement written without a `.`. Directives inside a block are refused. The named-graph block parser kept `\x` escapes in a prefixed name's local part, so `ex:a\-b` became the IRI `…a\-b`. It now unescapes them as the Turtle parser does, through the same function. It also refused an empty `{| |}` annotation late, if at all; that is now a syntax error, as in Turtle. The N-Triples and N-Quads writers now emit the canonical form: - language tags are lowercased; - U+FFFE and U+FFFF are written as `￾` / `ï¿¿`. --- docs/transactions/turtle.md | 2 + fluree-db-api/tests/it_trig_insert.rs | 32 +++++ fluree-db-transact/src/parse/trig_meta.rs | 142 ++++++++++++++++------ fluree-graph-format/src/rdf_text.rs | 2 + fluree-graph-ir/src/syntax.rs | 51 +++++--- fluree-graph-turtle/src/parser.rs | 2 +- 6 files changed, 176 insertions(+), 55 deletions(-) diff --git a/docs/transactions/turtle.md b/docs/transactions/turtle.md index 77529f078e..c59f81b553 100644 --- a/docs/transactions/turtle.md +++ b/docs/transactions/turtle.md @@ -602,6 +602,8 @@ GRAPH { } ``` +Default-graph triples can also be wrapped in an unlabeled `{ ... }` block. Graph labels must be IRIs; a blank-node label (`_:g { ... }`, `[] { ... }`) is refused. Inside a labeled block, blank-node property lists (`[ ... ]`, `[]`) and collections (`( ... )`) are not supported yet; use labeled blank nodes (`_:b`) there instead. + ### Submitting TriG Data TriG is supported on the **insert** and **upsert** endpoints, and on **sync** for replacing one named graph's contents ([Sync](sync.md#payload-formats)). Use the `application/trig` content type: diff --git a/fluree-db-api/tests/it_trig_insert.rs b/fluree-db-api/tests/it_trig_insert.rs index e252b077ab..97cf286fee 100644 --- a/fluree-db-api/tests/it_trig_insert.rs +++ b/fluree-db-api/tests/it_trig_insert.rs @@ -313,6 +313,38 @@ async fn trig_with_only_graph_blocks_is_accepted() { ); } +/// A `{ … }` default-graph block (its last statement without a `.`) lands +/// in the default graph, and a named graph block's escaped local name is +/// unescaped. +#[tokio::test] +async fn trig_insert_takes_default_blocks_and_escaped_local_names() { + let fluree = memory(); + let trig = "@prefix ex: .\n\ + { ex:alice ex:name \"Alice\" }\n\ + { ex:alice ex:knows ex:b\\-ob . }\n"; + let result = fluree + .insert_turtle( + genesis_ledger(&fluree, "it/trig-insert-default-block:main"), + trig, + ) + .await + .expect("default block and escaped local name"); + let db = GraphDb::from_ledger_state(&result.ledger); + assert_eq!( + select( + &fluree, + &db, + "PREFIX ex: \nSELECT ?s ?p ?o WHERE { ?s ?p ?o }", + ) + .await, + vec![row(&["ex:alice", "ex:name", "Alice"])] + ); + assert_eq!( + knows_in(&fluree, &db, G1).await, + vec![row(&["ex:alice", "ex:b-ob"])] + ); +} + /// Insert adds to a named graph; upsert replaces. Same document, so this pins /// that the TriG path runs with the caller's transaction type. #[tokio::test] diff --git a/fluree-db-transact/src/parse/trig_meta.rs b/fluree-db-transact/src/parse/trig_meta.rs index ffe45f3b22..4b8813fc12 100644 --- a/fluree-db-transact/src/parse/trig_meta.rs +++ b/fluree-db-transact/src/parse/trig_meta.rs @@ -271,11 +271,15 @@ pub fn unwrap_trig_graph_blocks(input: &str) -> Result { *b = b' '; } }; + // A block's last triple may omit its `.`; Turtle's may not. for block in &parser.unwrapped { blank(&mut turtle[block.header.0..block.header.1]); - // A block's last triple may omit its `.`; Turtle's may not. turtle[block.close] = if block.needs_dot { b'.' } else { b' ' }; } + for &(open, close, needs_dot) in &parser.default_blocks { + turtle[open] = b' '; + turtle[close] = if needs_dot { b'.' } else { b' ' }; + } for &(start, end) in &parser.excised { blank(&mut turtle[start..end]); } @@ -477,6 +481,11 @@ struct TrigMetaParser<'a> { directives: Vec<(usize, usize)>, // (start, end) byte ranges /// Default graph triple ranges default_triples: Vec<(usize, usize)>, + /// Ends of default-graph statements closed by a `}` instead of a `.`. + dotless_ends: Vec, + /// `{ … }` default-graph blocks: `{` offset, `}` offset, and whether the + /// last statement omits its `.` (see [`unwrap_trig_graph_blocks`]). + default_blocks: Vec<(usize, usize, bool)>, /// All GRAPH blocks (supports multiple named graphs) graph_blocks: Vec, /// Triples of the statement currently being parsed inside a GRAPH @@ -593,6 +602,8 @@ impl<'a> TrigMetaParser<'a> { base: None, directives: Vec::new(), default_triples: Vec::new(), + dotless_ends: Vec::new(), + default_blocks: Vec::new(), graph_blocks: Vec::new(), stmt_triples: Vec::new(), reified: Vec::new(), @@ -690,19 +701,8 @@ impl<'a> TrigMetaParser<'a> { { self.parse_graph_block(start_pos)?; } - // Anonymous default-graph wrapped block `{ ... }`. Valid W3C TriG - // (it denotes the default graph), but unsupported here: this parser - // separates default-graph Turtle from labeled graph blocks. Reject - // cleanly rather than letting the brace leak into the reconstructed - // Turtle and surface as a misleading low-level parse error. - TokenKind::LBrace => { - return Err(TransactError::Parse( - "anonymous default-graph block `{ ... }` is not supported; \ - write default-graph triples directly (outside any block), \ - or use a labeled graph block ` { ... }`" - .to_string(), - )); - } + // `{ … }`: the default graph, its statements default-graph Turtle. + TokenKind::LBrace => self.parse_default_block()?, // Blank-node graph label (`_:b { ... }`). Valid W3C TriG, but not // supported here in either form (the keyword form rejects it too). // Emit a clear error instead of a silent mis-parse. @@ -863,7 +863,7 @@ impl<'a> TrigMetaParser<'a> { let span = self.span_text(s, e); let (prefix, local) = split_prefixed_name(span); self.advance(); - self.expand_prefixed_name(prefix, local)? + self.expand_prefixed_name(prefix, &local)? } _ => { return Err(TransactError::Parse(format!( @@ -1144,12 +1144,11 @@ impl<'a> TrigMetaParser<'a> { r } }; - if !self.check(&TokenKind::AnnotationClose) { - self.annotation_depth += 1; - let body = self.parse_predicate_object_list(&reifier); - self.annotation_depth -= 1; - body?; - } + // `{| |}` needs a predicate-object list, as in Turtle. + self.annotation_depth += 1; + let body = self.parse_predicate_object_list(&reifier); + self.annotation_depth -= 1; + body?; if !self.check(&TokenKind::AnnotationClose) { return Err(TransactError::Parse(format!( "expected '|}}' to close annotation block, found {}", @@ -1460,7 +1459,7 @@ impl<'a> TrigMetaParser<'a> { let span = self.span_text(s, e); let (prefix, local) = split_prefixed_name(span); self.advance(); - self.expand_prefixed_name(prefix, local)? + self.expand_prefixed_name(prefix, &local)? } _ => { return Err(TransactError::Parse(format!( @@ -1475,6 +1474,54 @@ impl<'a> TrigMetaParser<'a> { } } + /// A `{ … }` default-graph block, positioned at its `{`. Its last + /// statement may omit the `.` a Turtle statement needs. + fn parse_default_block(&mut self) -> Result<()> { + let open = self.current().start as usize; + self.advance(); + let mut needs_dot = false; + while !self.check(&TokenKind::RBrace) && !self.is_at_end() { + if matches!( + self.current().kind, + TokenKind::KwPrefix + | TokenKind::KwSparqlPrefix + | TokenKind::KwBase + | TokenKind::KwSparqlBase + | TokenKind::KwVersion + | TokenKind::KwSparqlVersion + ) { + return Err(TransactError::Parse( + "directives are not allowed inside a graph block".to_string(), + )); + } + let start = self.current().start as usize; + while !self.check(&TokenKind::Dot) + && !self.check(&TokenKind::RBrace) + && !self.is_at_end() + { + self.advance(); + } + needs_dot = !self.check(&TokenKind::Dot); + if !needs_dot { + self.advance(); + } + let end = self.tokens[self.pos.saturating_sub(1)].end as usize; + self.default_triples.push((start, end)); + if needs_dot { + self.dotless_ends.push(end); + } + } + if !self.check(&TokenKind::RBrace) { + return Err(TransactError::Parse( + "expected '}' to close the default graph block".to_string(), + )); + } + let close = self.current().start as usize; + self.advance(); + self.default_blocks.push((open, close, needs_dot)); + Ok(()) + } + fn parse_default_triple(&mut self, start_pos: usize) -> Result<()> { // Skip to end of triple (dot terminator) while !self.check(&TokenKind::Dot) && !self.is_at_end() { @@ -1578,6 +1625,9 @@ impl<'a> TrigMetaParser<'a> { let mut turtle = String::new(); for (start, end) in spans { turtle.push_str(&self.input[start..end]); + if self.dotless_ends.contains(&end) { + turtle.push_str(" ."); + } turtle.push('\n'); } turtle @@ -1887,10 +1937,19 @@ impl<'a> TrigMetaParser<'a> { } /// Split a prefixed name into prefix and local parts. -fn split_prefixed_name(span: &str) -> (&str, &str) { - match span.find(':') { +/// A prefixed name's prefix and its local part with `\x` escapes resolved. +fn split_prefixed_name(span: &str) -> (&str, Cow<'_, str>) { + let (prefix, local) = match span.find(':') { Some(pos) => (&span[..pos], &span[pos + 1..]), None => (span, ""), + }; + if local.contains('\\') { + ( + prefix, + Cow::Owned(fluree_graph_turtle::parser::unescape_pn_local(local)), + ) + } else { + (prefix, Cow::Borrowed(local)) } } @@ -2610,20 +2669,27 @@ GRAPH { } #[test] - fn test_anonymous_default_graph_block_clean_error() { - // Anonymous `{ ... }` is valid W3C TriG but unsupported here; it must - // produce a clear error, not a silent mis-parse / misleading downstream - // Turtle error. - let mut ns = test_registry(); - let input = "@prefix ex: .\n{\n ex:a ex:b ex:c .\n}\n"; - - let err = extract_trig_txn_meta(input, &mut ns) - .unwrap_err() - .to_string(); - assert!( - err.contains("anonymous default-graph block"), - "expected a clear anonymous-block error, got: {err}" - ); + fn anonymous_default_graph_block_is_default_graph_turtle() { + // `{ … }` is the default graph; its last statement may omit the `.`. + for input in [ + "@prefix ex: .\n{ ex:a ex:b ex:c . ex:d ex:e ex:f }\n { ex:x ex:y ex:z }\n", + "@prefix ex: .\n{ ex:a ex:b ex:c . ex:d ex:e ex:f . }\n { ex:x ex:y ex:z }\n", + ] { + let phase1 = parse_trig_phase1(input).unwrap(); + assert_eq!(phase1.named_graphs.len(), 1, "{input}"); + assert!(!phase1.turtle.contains('{'), "{}", phase1.turtle); + let mut sink = fluree_graph_ir::GraphCollectorSink::new(); + fluree_graph_turtle::parse(&phase1.turtle, &mut sink) + .unwrap_or_else(|e| panic!("{e}: {}", phase1.turtle)); + assert_eq!(sink.into_graph().len(), 2, "{}", phase1.turtle); + + let unwrapped = unwrap_trig_graph_blocks(input).unwrap(); + let mut sink = fluree_graph_ir::GraphCollectorSink::new(); + fluree_graph_turtle::parse(&unwrapped.turtle, &mut sink) + .unwrap_or_else(|e| panic!("{e}: {}", unwrapped.turtle)); + assert_eq!(sink.into_graph().len(), 3, "{}", unwrapped.turtle); + assert!(unwrapped.mixes_default_and_named); + } } #[test] diff --git a/fluree-graph-format/src/rdf_text.rs b/fluree-graph-format/src/rdf_text.rs index d62297a50e..31df645c7c 100644 --- a/fluree-graph-format/src/rdf_text.rs +++ b/fluree-graph-format/src/rdf_text.rs @@ -245,6 +245,8 @@ fn push_nt_term(out: &mut String, term: &Term) { language, } => { push_quoted(out, value); + // Tags are case-insensitive; the canonical form is lowercase. + let language = language.as_deref().map(str::to_ascii_lowercase); push_literal_suffix(out, datatype, language.as_deref(), |out, iri| { syntax::push_iri_ref(out, iri); }); diff --git a/fluree-graph-ir/src/syntax.rs b/fluree-graph-ir/src/syntax.rs index e4419ed8f0..4142ddf8b4 100644 --- a/fluree-graph-ir/src/syntax.rs +++ b/fluree-graph-ir/src/syntax.rs @@ -15,32 +15,46 @@ const HEX: &[u8; 16] = b"0123456789ABCDEF"; /// Escape `s` as the body of a `"…"` string literal, in canonical N-Triples /// form (also valid Turtle, N-Quads and TriG): `"` `\` and the control /// characters with a short escape use it (`\t \b \n \r \f`), every other C0 -/// control and DEL is `\uXXXX`, and everything else is written as is. +/// control, DEL and the noncharacters U+FFFE / U+FFFF are `\uXXXX`, and +/// everything else is written as is. pub fn escape_string(s: &str, mut put: impl FnMut(&str) -> Result<(), E>) -> Result<(), E> { let bytes = s.as_bytes(); let mut start = 0; - for (i, &b) in bytes.iter().enumerate() { - let short: &str = match b { - b'"' => "\\\"", - b'\\' => "\\\\", - b'\t' => "\\t", - 0x08 => "\\b", - b'\n' => "\\n", - b'\r' => "\\r", - 0x0C => "\\f", - 0x00..=0x1F | 0x7F => "", - _ => continue, + let mut i = 0; + while i < bytes.len() { + let b = bytes[i]; + // U+FFFE / U+FFFF are `EF BF BE` / `EF BF BF`. + let (escape, len): (&str, usize) = match b { + b'"' => ("\\\"", 1), + b'\\' => ("\\\\", 1), + b'\t' => ("\\t", 1), + 0x08 => ("\\b", 1), + b'\n' => ("\\n", 1), + b'\r' => ("\\r", 1), + 0x0C => ("\\f", 1), + 0x00..=0x1F | 0x7F => ("", 1), + 0xEF if bytes.get(i + 1) == Some(&0xBF) && bytes.get(i + 2) == Some(&0xBE) => { + ("\\uFFFE", 3) + } + 0xEF if bytes.get(i + 1) == Some(&0xBF) && bytes.get(i + 2) == Some(&0xBF) => { + ("\\uFFFF", 3) + } + _ => { + i += 1; + continue; + } }; - // Every escaped byte is ASCII, so `i` is a char boundary. + // Every escaped sequence starts a char, so `i` is a char boundary. if start < i { put(&s[start..i])?; } - if short.is_empty() { + if escape.is_empty() { put(ascii(&uchar(b)))?; } else { - put(short)?; + put(escape)?; } - start = i + 1; + i += len; + start = i; } if start < bytes.len() { put(&s[start..])?; @@ -310,6 +324,11 @@ mod tests { assert_eq!(string("\u{0}\u{1f}\u{7f}"), r"\u0000\u001F\u007F"); // C1 controls and non-ASCII are legal in a string literal. assert_eq!(string("\u{85}é😀"), "\u{85}é😀"); + // The two noncharacters are escaped; their neighbours are not. + assert_eq!( + string("a\u{FFFE}b\u{FFFF}\u{FFFD}\u{EFBF}"), + "a\\uFFFEb\\uFFFF\u{FFFD}\u{EFBF}" + ); } #[test] diff --git a/fluree-graph-turtle/src/parser.rs b/fluree-graph-turtle/src/parser.rs index 1c500ee833..8c0a84826f 100644 --- a/fluree-graph-turtle/src/parser.rs +++ b/fluree-graph-turtle/src/parser.rs @@ -1680,7 +1680,7 @@ impl<'a, 'input, S: GraphSink> Parser<'a, 'input, S> { /// Unescape local name escape sequences (`\x` → `x`). /// /// Only called when `\` is detected in the local part (extremely rare). -fn unescape_pn_local(local: &str) -> String { +pub fn unescape_pn_local(local: &str) -> String { let mut result = String::with_capacity(local.len()); let mut chars = local.chars(); while let Some(c) = chars.next() { From 196d7c49793bf5cee3ea16c2282193bda7398c6f Mon Sep 17 00:00:00 2001 From: bplatz Date: Sun, 4 Oct 2026 18:28:35 -0400 Subject: [PATCH 70/92] test(w3c): run the RDF 1.2 N-Triples, N-Quads and TriG suites MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The vendored suites, each including its RDF 1.1 suite, now run through the parsers their ingest paths use: - N-Triples through the Turtle parser; - N-Quads regrouped by `nquads_to_trig`; - TriG through `parse_trig_phase1`, whose dataset builder the SPARQL update harness now shares. Canonicalization tests write the parsed document back with the N-Triples and N-Quads writers. The registers record what still fails, by cause: - Turtle-grammar acceptance of N-Triples-invalid input (N-Triples is read as Turtle); - blank-node graph labels; - numeric lexical forms; - annotation-of-annotation; - the named-graph block parser's missing `[ … ]`, `[]`, collections, repeated `;` and bare `<< … >> .`. Also drops a fixed limitation from the compatibility doc's planned list. --- docs/reference/compatibility.md | 7 +- testsuite-sparql/Cargo.lock | 1 + testsuite-sparql/Cargo.toml | 1 + testsuite-sparql/src/rdf_handlers.rs | 155 +++++++++++++++++++++++- testsuite-sparql/src/result_format.rs | 105 ++++++++-------- testsuite-sparql/src/vocab.rs | 14 +++ testsuite-sparql/tests/registers/mod.rs | 120 ++++++++++++++++++ testsuite-sparql/tests/w3c_rdf.rs | 33 ++++- 8 files changed, 376 insertions(+), 60 deletions(-) diff --git a/docs/reference/compatibility.md b/docs/reference/compatibility.md index 2254dedb0d..dcab80dc04 100644 --- a/docs/reference/compatibility.md +++ b/docs/reference/compatibility.md @@ -39,8 +39,9 @@ N-Quads files. As in RDF 1.2, only the annotation syntax asserts the triple; `<< s p o >>` and `rdf:reifies <<( s p o )>>` reify it without asserting it. The `VERSION "1.2"` / `@version` directive and `--ltr` / `--rtl` base-direction language tags are accepted. -The vendored W3C RDF 1.1 and RDF 1.2 Turtle suites run in CI -(`testsuite-sparql/tests/w3c_rdf.rs`), with known gaps in the skip register. +The vendored W3C RDF 1.1 and RDF 1.2 Turtle, N-Triples, N-Quads and TriG +suites run in CI (`testsuite-sparql/tests/w3c_rdf.rs`), with known gaps in the +skip register. Not yet supported: - Annotation-of-annotation (a `{| ... |}` or `<< ... >>` inside an annotation body) @@ -485,7 +486,7 @@ Export Fluree data to: - SPARQL 1.1 Federation: remote `SERVICE` endpoints (local-ledger `SERVICE` is supported) - Remote `LOAD` in SPARQL UPDATE - GeoSPARQL: remaining OGC functions (only `geof:distance` is implemented today) -- RDF 1.2 / SPARQL 1.2: annotation-of-annotation; a `GRAPH ?g` name written as an INSERT template object +- RDF 1.2 / SPARQL 1.2: annotation-of-annotation **Storage:** - Additional cloud providers (GCP, Azure) diff --git a/testsuite-sparql/Cargo.lock b/testsuite-sparql/Cargo.lock index 3c43d5a5e1..ef278c7f78 100644 --- a/testsuite-sparql/Cargo.lock +++ b/testsuite-sparql/Cargo.lock @@ -2435,6 +2435,7 @@ dependencies = [ "fluree-db-api", "fluree-db-sparql", "fluree-db-transact", + "fluree-graph-format", "fluree-graph-ir", "fluree-graph-turtle", "quick-xml", diff --git a/testsuite-sparql/Cargo.toml b/testsuite-sparql/Cargo.toml index d397bacb65..1b913cf761 100644 --- a/testsuite-sparql/Cargo.toml +++ b/testsuite-sparql/Cargo.toml @@ -17,6 +17,7 @@ fluree-db-sparql = { path = "../fluree-db-sparql" } fluree-db-transact = { path = "../fluree-db-transact" } fluree-graph-turtle = { path = "../fluree-graph-turtle" } fluree-graph-ir = { path = "../fluree-graph-ir" } +fluree-graph-format = { path = "../fluree-graph-format" } # Test infrastructure anyhow = "1" diff --git a/testsuite-sparql/src/rdf_handlers.rs b/testsuite-sparql/src/rdf_handlers.rs index 3ecec9f735..ec8f9ae212 100644 --- a/testsuite-sparql/src/rdf_handlers.rs +++ b/testsuite-sparql/src/rdf_handlers.rs @@ -15,6 +15,12 @@ //! valid Turtle document. Negative N-Triples tests can legitimately be valid //! Turtle (prefixed names, `a`, numeric shorthands); such entries belong in //! the suite's skip register with that reason. +//! +//! TriG documents parse the way TriG ingest parses them ([`trig_dataset`]), +//! and N-Quads documents are first regrouped into TriG by +//! `nquads_to_trig`, as bulk import does. Canonicalization (C14N) tests +//! write the parsed document back with the N-Triples / N-Quads writer that +//! serves CONSTRUCT results and compare it with the canonical form. use std::collections::{BTreeMap, HashSet}; @@ -27,9 +33,10 @@ use crate::files::read_file_to_string; use crate::manifest::Test; use crate::result_comparison::{are_results_isomorphic, format_results_diff}; use crate::result_format::{ - ir_term_to_rdf_term, reification_triples, RdfTerm, SparqlResults, Triple, + ir_term_to_rdf_term, reification_triples, trig_dataset, RdfTerm, SparqlResults, Triple, }; use crate::vocab::rdft; +use fluree_graph_ir::Dataset; const RDF_FIRST: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#first"; const RDF_REST: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#rest"; @@ -49,6 +56,152 @@ pub fn register_rdf_tests(evaluator: &mut TestEvaluator) { rdft::TEST_NTRIPLES_NEGATIVE_SYNTAX, evaluate_negative_syntax, ); + evaluator.register(rdft::TEST_NTRIPLES_POSITIVE_C14N, evaluate_ntriples_c14n); + evaluator.register(rdft::TEST_NQUADS_POSITIVE_SYNTAX, |t| { + positive_dataset_syntax(t, nquads_dataset) + }); + evaluator.register(rdft::TEST_NQUADS_NEGATIVE_SYNTAX, |t| { + negative_dataset_syntax(t, nquads_dataset) + }); + evaluator.register(rdft::TEST_NQUADS_POSITIVE_C14N, evaluate_nquads_c14n); + evaluator.register(rdft::TEST_TRIG_POSITIVE_SYNTAX, |t| { + positive_dataset_syntax(t, trig_action) + }); + evaluator.register(rdft::TEST_TRIG_NEGATIVE_SYNTAX, |t| { + negative_dataset_syntax(t, trig_action) + }); + evaluator.register(rdft::TEST_TRIG_NEGATIVE_EVAL, |t| { + negative_dataset_syntax(t, trig_action) + }); + evaluator.register(rdft::TEST_TRIG_EVAL, evaluate_trig_eval); +} + +/// A TriG action, relative IRIs resolved against its manifest URL. +fn trig_action(url: &str) -> Result { + let content = read_file_to_string(url).with_context(|| format!("Reading {url}"))?; + trig_dataset(&format!("@base <{url}> .\n{content}")) +} + +/// An N-Quads document, regrouped into TriG as bulk import does. +fn nquads_dataset(url: &str) -> Result { + let content = read_file_to_string(url).with_context(|| format!("Reading {url}"))?; + let trig = + fluree_db_transact::parse::nquads_to_trig(&content).map_err(|e| anyhow::anyhow!("{e}"))?; + trig_dataset(&trig) +} + +fn positive_dataset_syntax(test: &Test, parse: fn(&str) -> Result) -> Result<()> { + let url = action_url(test)?; + parse(url).map(|_| ()).with_context(|| { + format!( + "Positive syntax test failed — parser rejected a valid document.\n\ + Test: {}\nFile: {url}", + test.id + ) + }) +} + +fn negative_dataset_syntax(test: &Test, parse: fn(&str) -> Result) -> Result<()> { + let url = action_url(test)?; + ensure!( + parse(url).is_err(), + "Negative syntax test failed — parser accepted an invalid document.\n\ + Test: {}\nFile: {url}", + test.id + ); + Ok(()) +} + +/// `rdft:TestTrigEval`: the action's dataset against the expected N-Quads, +/// graph by graph up to blank-node isomorphism. +fn evaluate_trig_eval(test: &Test) -> Result<()> { + let url = action_url(test)?; + let result_url = test + .result + .as_deref() + .with_context(|| format!("{}: evaluation test has no mf:result", test.id))?; + let actual = trig_action(url).with_context(|| { + format!( + "Evaluation test failed — parser rejected the action document.\n\ + Test: {}\nFile: {url}", + test.id + ) + })?; + let expected = nquads_dataset(result_url).with_context(|| { + format!( + "Evaluation test failed — could not parse the expected dataset.\n\ + Test: {}\nFile: {result_url}", + test.id + ) + })?; + let names = |d: &Dataset| d.named.keys().map(ir_term_to_rdf_term).collect::>(); + let as_graphs = |d: &Dataset| { + std::iter::once(graph_to_rdf_triples(&d.default)) + .chain(d.named.values().map(graph_to_rdf_triples)) + .collect::>() + }; + let (expected_names, actual_names) = (names(&expected), names(&actual)); + ensure!( + expected_names.len() == actual_names.len() + && expected_names + .iter() + .zip(&actual_names) + .all(|(e, a)| e == a + || matches!((e, a), (RdfTerm::BlankNode(_), RdfTerm::BlankNode(_)))), + "Evaluation test failed — graph names differ.\nTest: {}\nFile: {url}\n\ + expected {expected_names:?}\nactual {actual_names:?}", + test.id + ); + for (expected, actual) in as_graphs(&expected).into_iter().zip(as_graphs(&actual)) { + let expected = SparqlResults::Graph(expected); + let actual = SparqlResults::Graph(actual); + if !are_results_isomorphic(&expected, &actual) { + bail!( + "Evaluation test failed — graphs differ.\nTest: {}\nFile: {url}\n{}", + test.id, + format_results_diff(&expected, &actual) + ); + } + } + Ok(()) +} + +/// `rdft:TestNTriplesPositiveC14N`: the parsed document, written by the +/// N-Triples writer, is the expected canonical form. +fn evaluate_ntriples_c14n(test: &Test) -> Result<()> { + let url = action_url(test)?; + let graph = parse_action(url)?.into_graph(); + let written = + fluree_graph_format::format_ntriples(&graph).map_err(|e| anyhow::anyhow!("{e}"))?; + compare_c14n(test, &written) +} + +/// `rdft:TestNQuadsPositiveC14N`: as for N-Triples, with the N-Quads writer. +fn evaluate_nquads_c14n(test: &Test) -> Result<()> { + let dataset = nquads_dataset(action_url(test)?)?; + let written = + fluree_graph_format::format_nquads(&dataset).map_err(|e| anyhow::anyhow!("{e}"))?; + compare_c14n(test, &written) +} + +/// Canonical documents are sets of lines; their order is unspecified. +fn compare_c14n(test: &Test, written: &str) -> Result<()> { + let result_url = test + .result + .as_deref() + .with_context(|| format!("{}: C14N test has no mf:result", test.id))?; + let expected = read_file_to_string(result_url)?; + let lines = |s: &str| { + let mut v: Vec = s.lines().map(str::to_string).collect(); + v.sort(); + v + }; + ensure!( + lines(&expected) == lines(written), + "C14N test failed — written form differs.\nTest: {}\nexpected:\n{expected}\nwritten:\n{written}", + test.id + ); + Ok(()) } /// Parse an action document from its manifest URL, resolving relative IRIs diff --git a/testsuite-sparql/src/result_format.rs b/testsuite-sparql/src/result_format.rs index 94d44eb410..d7b49f3538 100644 --- a/testsuite-sparql/src/result_format.rs +++ b/testsuite-sparql/src/result_format.rs @@ -11,7 +11,7 @@ use std::collections::HashMap; use anyhow::{bail, Context, Result}; -use fluree_graph_ir::{Graph as IrGraph, GraphCollectorSink, Term as IrTerm}; +use fluree_graph_ir::{Dataset as IrDataset, Graph as IrGraph, GraphCollectorSink, Term as IrTerm}; use fluree_graph_turtle::parse as parse_turtle; use quick_xml::events::Event; use quick_xml::Reader; @@ -766,79 +766,81 @@ pub type ExpectedDataset = (Vec, Vec<(String, Vec)>); /// An expected-state file's default graph and named graphs. A `.trig` file /// names its graphs in `GRAPH` blocks; any other file is one default graph. pub fn parse_expected_dataset(url: &str) -> Result { - use fluree_db_transact::{RawObject, RawTerm}; if !url.ends_with(".trig") { return Ok((parse_expected_graph(url)?, Vec::new())); } let content = read_file_to_string(url).with_context(|| format!("Reading expected graph file: {url}"))?; - let with_base = format!("@base <{url}> .\n{content}"); - let phase1 = fluree_db_transact::parse_trig_phase1(&with_base) - .map_err(|e| anyhow::anyhow!("Parsing expected TriG {url}: {e}"))?; + let dataset = trig_dataset(&format!("@base <{url}> .\n{content}")) + .with_context(|| format!("Parsing expected TriG {url}"))?; + let named = dataset + .named + .iter() + .map(|(name, graph)| (term_label(name), graph_triples(graph))) + .collect(); + Ok((graph_triples(&dataset.default), named)) +} + +fn term_label(term: &IrTerm) -> String { + match term { + IrTerm::Iri(iri) => iri.to_string(), + IrTerm::BlankNode(id) => format!("_:{}", id.as_str()), + other => format!("{other:?}"), + } +} + +/// A TriG document as a dataset, parsed the way TriG ingest parses it: the +/// default graph through the Turtle parser, `GRAPH` blocks through +/// [`fluree_db_transact::parse_trig_phase1`]. +pub fn trig_dataset(input: &str) -> Result { + use fluree_db_transact::{RawObject, RawTerm}; + use fluree_graph_ir::Datatype; + let phase1 = + fluree_db_transact::parse_trig_phase1(input).map_err(|e| anyhow::anyhow!("{e}"))?; let mut sink = GraphCollectorSink::new(); - parse_turtle(&phase1.turtle, &mut sink) - .with_context(|| format!("Parsing expected graph: {url}"))?; - let default = graph_triples(&sink.into_graph()); + parse_turtle(&phase1.turtle, &mut sink)?; + let mut dataset = IrDataset::new(); + dataset.default = sink.into_graph(); - let mut named = Vec::new(); for block in &phase1.named_graphs { - let node = |term: &RawTerm| -> Result { + let node = |term: &RawTerm| -> Result { Ok(match term { RawTerm::Iri(iri) => match iri.strip_prefix("_:") { - Some(label) => RdfTerm::BlankNode(label.to_string()), - None => RdfTerm::Iri(iri.clone()), + Some(label) => IrTerm::blank(label), + None => IrTerm::iri(iri), }, RawTerm::PrefixedName { prefix, local } => { let ns = block .prefixes .get(prefix.as_str()) - .with_context(|| format!("undefined prefix {prefix}: in {url}"))?; - RdfTerm::Iri(format!("{ns}{local}")) + .with_context(|| format!("undefined prefix {prefix}:"))?; + IrTerm::iri(format!("{ns}{local}")) } }) }; - fn object(o: &RawObject, node: &dyn Fn(&RawTerm) -> Result) -> Result { - let typed = |value: String, dt: &str| RdfTerm::Literal { - value, - datatype: Some(format!("http://www.w3.org/2001/XMLSchema#{dt}")), - language: None, - }; + fn object(o: &RawObject, node: &dyn Fn(&RawTerm) -> Result) -> Result { Ok(match o { RawObject::Iri(iri) => node(&RawTerm::Iri(iri.clone()))?, RawObject::PrefixedName { prefix, local } => node(&RawTerm::PrefixedName { prefix: prefix.clone(), local: local.clone(), })?, - RawObject::String(s) => RdfTerm::Literal { - value: s.clone(), - datatype: None, - language: None, - }, - RawObject::Integer(n) => typed(n.to_string(), "integer"), - RawObject::Double(d) => typed(d.to_string(), "double"), - RawObject::Boolean(b) => typed(b.to_string(), "boolean"), - RawObject::TypedLiteral { value, datatype } => RdfTerm::Literal { - value: value.clone(), - datatype: Some(datatype.clone()), - language: None, - }, - RawObject::LangString { value, lang } => RdfTerm::Literal { - value: value.clone(), - datatype: None, - language: Some(lang.clone()), - }, + RawObject::String(s) => IrTerm::string(s), + RawObject::Integer(n) => IrTerm::integer(*n), + RawObject::Double(d) => IrTerm::double(*d), + RawObject::Boolean(b) => IrTerm::boolean(*b), + RawObject::TypedLiteral { value, datatype } => { + IrTerm::typed(value, Datatype::from_iri(datatype)) + } + RawObject::LangString { value, lang } => IrTerm::lang_string(value, lang), RawObject::TripleTerm { subject, predicate, object: o, - } => RdfTerm::Triple(Box::new(Triple { - subject: node(subject)?, - predicate: node(predicate)?, - object: object(o, node)?, - })), + } => IrTerm::triple(node(subject)?, node(predicate)?, object(o, node)?), }) } - let mut triples = Vec::new(); + let graph = dataset.graph_mut(Some(&node(&RawTerm::Iri(block.iri.clone()))?)); for t in &block.triples { let subject = node( t.subject @@ -847,24 +849,19 @@ pub fn parse_expected_dataset(url: &str) -> Result { )?; let predicate = node(&t.predicate)?; for o in &t.objects { - triples.push(Triple { - subject: subject.clone(), - predicate: predicate.clone(), - object: object(o, &node)?, - }); + graph.add_triple(subject.clone(), predicate.clone(), object(o, &node)?); } } for r in &block.reified { - triples.extend(reification_triples( - node(&r.reifier)?, + graph.add_reification( node(&r.subject)?, node(&r.predicate)?, object(&r.object, &node)?, - )); + node(&r.reifier)?, + ); } - named.push((block.iri.clone(), triples)); } - Ok((default, named)) + Ok(dataset) } /// `reifier`'s attachment to `(s, p, o)`: `reifier rdf:reifies <<( s p o )>>`. diff --git a/testsuite-sparql/src/vocab.rs b/testsuite-sparql/src/vocab.rs index 3d9d9aa312..e842a3b9d2 100644 --- a/testsuite-sparql/src/vocab.rs +++ b/testsuite-sparql/src/vocab.rs @@ -97,6 +97,20 @@ pub mod rdft { "http://www.w3.org/ns/rdftest#TestNTriplesPositiveSyntax"; pub const TEST_NTRIPLES_NEGATIVE_SYNTAX: &str = "http://www.w3.org/ns/rdftest#TestNTriplesNegativeSyntax"; + pub const TEST_NTRIPLES_POSITIVE_C14N: &str = + "http://www.w3.org/ns/rdftest#TestNTriplesPositiveC14N"; + pub const TEST_NQUADS_POSITIVE_SYNTAX: &str = + "http://www.w3.org/ns/rdftest#TestNQuadsPositiveSyntax"; + pub const TEST_NQUADS_NEGATIVE_SYNTAX: &str = + "http://www.w3.org/ns/rdftest#TestNQuadsNegativeSyntax"; + pub const TEST_NQUADS_POSITIVE_C14N: &str = + "http://www.w3.org/ns/rdftest#TestNQuadsPositiveC14N"; + pub const TEST_TRIG_POSITIVE_SYNTAX: &str = + "http://www.w3.org/ns/rdftest#TestTrigPositiveSyntax"; + pub const TEST_TRIG_NEGATIVE_SYNTAX: &str = + "http://www.w3.org/ns/rdftest#TestTrigNegativeSyntax"; + pub const TEST_TRIG_EVAL: &str = "http://www.w3.org/ns/rdftest#TestTrigEval"; + pub const TEST_TRIG_NEGATIVE_EVAL: &str = "http://www.w3.org/ns/rdftest#TestTrigNegativeEval"; } /// Standard RDF vocabulary diff --git a/testsuite-sparql/tests/registers/mod.rs b/testsuite-sparql/tests/registers/mod.rs index eb8ed40c37..ddfa008096 100644 --- a/testsuite-sparql/tests/registers/mod.rs +++ b/testsuite-sparql/tests/registers/mod.rs @@ -557,3 +557,123 @@ pub const RDF12_TURTLE_EVAL: &[&str] = &[ "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-turtle/eval#turtle12-annotation-04", "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-turtle/eval#turtle12-reified-triples-annotation-01", ]; + +pub const RDF12_NTRIPLES: &[&str] = &[ + // N-Triples is read by the Turtle parser (the `.nt` ingest path), which + // accepts these as Turtle: relative IRIs, directives, `,` object lists, + // numeric and long-string shorthands, `<< >>` reified triples and + // annotations (21) + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-uri-06", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-uri-07", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-uri-08", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-uri-09", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-prefix-01", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-base-01", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-struct-01", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-string-02", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-string-03", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-string-04", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-string-05", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-num-01", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-num-02", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-triples/manifest.ttl#nt-syntax-bad-num-03", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-n-triples/syntax#ntriples12-bad-09", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-n-triples/syntax#ntriples12-bad-iri-1", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-n-triples/syntax#ntriples12-bad-reified-1", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-n-triples/syntax#ntriples12-bad-reified-2", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-n-triples/syntax#ntriples12-bad-reified-3", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-n-triples/syntax#ntriples12-bnode-bad-annotated-syntax-1", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-n-triples/syntax#ntriples12-bnode-bad-annotated-syntax-2", + // ill-formed literals the Turtle grammar admits: an explicit + // `^^rdf:langString` / `^^rdf:dirLangString`, and an over-long BCP 47 + // primary subtag (3) + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-n-triples/syntax#ntriples-langdir-bad-3", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-n-triples/syntax#ntriples-langdir-bad-4", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-n-triples/syntax#ntriples-langdir-bad-5", +]; + +pub const RDF12_NQUADS: &[&str] = &[ + // blank-node graph labels are refused on import (6) + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nq-syntax-bnode-01", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nq-syntax-bnode-02", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nq-syntax-bnode-03", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nq-syntax-bnode-04", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nq-syntax-bnode-05", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nq-syntax-bnode-06", + // `nquads_to_trig` hands each line to the Turtle parser, which accepts + // these as Turtle: a relative graph IRI, a directive, numeric and + // long-string shorthands (9) + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nq-syntax-bad-uri-01", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nt-syntax-bad-prefix-01", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nt-syntax-bad-string-02", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nt-syntax-bad-string-03", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nt-syntax-bad-string-04", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nt-syntax-bad-string-05", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nt-syntax-bad-num-01", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nt-syntax-bad-num-02", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-n-quads/manifest.ttl#nt-syntax-bad-num-03", +]; + +pub const RDF12_TRIG: &[&str] = &[ + // the named-graph block parser (`parse_trig_phase1`, behind TriG insert + // and import) has no `[ … ]` property lists, `[]`, collections, repeated + // `;` or bare `<< … >> .` statements; default-graph content goes through + // the full Turtle parser and has all of them (32) + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#sole_blankNodePropertyList", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#blankNodePropertyList_as_subject", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#blankNodePropertyList_with_multiple_triples", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#nested_blankNodePropertyLists", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#blankNodePropertyList_containing_collection", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#blankNodePropertyList_as_object", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#blankNodePropertyList_as_object_containing_objectList", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#blankNodePropertyList_as_object_containing_objectList_of_two_objects", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#predicateObjectList_with_blankNodePropertyList_as_object", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#anonymous_blank_node_subject", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#anonymous_blank_node_object", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#collection_subject", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#collection_object", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#nested_collection", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#empty_collection", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#first", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#last", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#repeated_semis_at_end", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#repeated_semis_not_at_end", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-subm-01", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-subm-05", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-subm-06", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-subm-08", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-subm-09", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-subm-10", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-subm-14", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-trig/eval#trig12-annotation-02", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-trig/syntax#trig12-4", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-trig/syntax#trig12-ann-2", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-trig/syntax#trig12-bnode-3", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-trig/syntax#trig12-inside-1", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-trig/syntax#trig12-inside-2", + // blank-node graph labels (`_:g { }`, `[] { }`) are refused (6) + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#labeled_blank_node_graph", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#alternating_bnode_graphs", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#anonymous_blank_node_graph", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-syntax-minimal-whitespace-01", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-kw-graph-06", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-kw-graph-07", + // numeric literals are canonicalized on ingest; the expected graphs keep + // the source lexical form (as in RDF11_TURTLE) (7) + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#bareword_double", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#double_lower_case_e", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#positive_numeric", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#numeric_with_leading_0", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-subm-11", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-subm-19", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-subm-20", + // lexer, as in RDF11_TURTLE: `\u` escapes in IRIREF that decode to + // forbidden characters are accepted (3); PN_LOCAL with interior dots (1) + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-eval-bad-01", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-eval-bad-02", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-eval-bad-03", + "https://w3c.github.io/rdf-tests/rdf/rdf11/rdf-trig/manifest.ttl#trig-syntax-ln-dots", + // annotation-of-annotation, deferred as in RDF12_TURTLE_EVAL (2) + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-trig/eval#trig12-annotation-04", + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-trig/eval#trig12-reified-triples-annotation-01", +]; diff --git a/testsuite-sparql/tests/w3c_rdf.rs b/testsuite-sparql/tests/w3c_rdf.rs index 640d28a301..15a6c252ba 100644 --- a/testsuite-sparql/tests/w3c_rdf.rs +++ b/testsuite-sparql/tests/w3c_rdf.rs @@ -1,9 +1,11 @@ -//! W3C RDF (Turtle) test suite registration. +//! W3C RDF test suite registration. //! //! Runs the vendored `rdf-tests` RDF 1.1 and RDF 1.2 Turtle manifests //! through the Turtle parser on the `GraphCollectorSink` path — the parser //! behind `parse_to_json`, i.e. the conversion every JSON-LD write path -//! (upsert, graph sync, memory import) uses for Turtle input. Same contract +//! (upsert, graph sync, memory import) uses for Turtle input — and the RDF +//! 1.2 N-Triples, N-Quads and TriG manifests (each including its RDF 1.1 +//! suite) through the parsers their ingest paths use. Same contract //! as `w3c_sparql.rs`: a suite is green when every test passes or appears in //! its register, and `check_testsuite` polices the register both ways. @@ -40,3 +42,30 @@ fn rdf12_turtle_eval() -> Result<()> { reg::RDF12_TURTLE_EVAL, ) } + +/// RDF 1.2 N-Triples (with RDF 1.1): syntax and canonical form. +#[test] +fn rdf12_ntriples() -> Result<()> { + check_testsuite( + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-n-triples/manifest.ttl", + reg::RDF12_NTRIPLES, + ) +} + +/// RDF 1.2 N-Quads (with RDF 1.1): syntax and canonical form. +#[test] +fn rdf12_nquads() -> Result<()> { + check_testsuite( + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-n-quads/manifest.ttl", + reg::RDF12_NQUADS, + ) +} + +/// RDF 1.2 TriG (with RDF 1.1): syntax and evaluation. +#[test] +fn rdf12_trig() -> Result<()> { + check_testsuite( + "https://w3c.github.io/rdf-tests/rdf/rdf12/rdf-trig/manifest.ttl", + reg::RDF12_TRIG, + ) +} From 0ac086e3a9130fc0da73952ef7f2c3be555e038b Mon Sep 17 00:00:00 2001 From: bplatz Date: Sun, 4 Oct 2026 19:24:14 -0400 Subject: [PATCH 71/92] test(api): serialize the tests that flip the fast-path switch `it_wildcard_batched_join` compares each query with the process-global fast-path switch on and off. With two tests in the binary, one could flip the switch while the other checked its routing. --- fluree-db-api/tests/it_wildcard_batched_join.rs | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/fluree-db-api/tests/it_wildcard_batched_join.rs b/fluree-db-api/tests/it_wildcard_batched_join.rs index 028d0ccb86..8c6f10e68c 100644 --- a/fluree-db-api/tests/it_wildcard_batched_join.rs +++ b/fluree-db-api/tests/it_wildcard_batched_join.rs @@ -9,6 +9,9 @@ use serde_json::{json, Value}; const PREFIX: &str = "PREFIX ex: "; const EVENT: &str = "batched wildcard join engaged"; +/// The fast-path switch is process-global: tests that flip it run one at a time. +static SERIAL: tokio::sync::Mutex<()> = tokio::sync::Mutex::const_new(()); + #[derive(Clone, Copy, Debug)] enum Routing { MustFire, @@ -79,6 +82,7 @@ fn normalized(result: &Value) -> Vec { #[tokio::test(flavor = "current_thread")] async fn wildcard_joins_preserve_facts_multiplicity_and_fallbacks() { assert!(std::env::var_os("FLUREE_DISABLE_QUERY_FAST_PATHS").is_none()); + let _serial = SERIAL.lock().await; let _reset = Reset; let dir = tempfile::tempdir().unwrap(); let fluree = FlureeBuilder::file(dir.path().to_string_lossy().to_string()) @@ -449,6 +453,7 @@ async fn wildcard_joins_preserve_facts_multiplicity_and_fallbacks() { async fn wildcard_join_benchmark() { use std::fmt::Write; use std::time::Instant; + let _serial = SERIAL.lock().await; let _reset = Reset; let dir = tempfile::tempdir().unwrap(); let data = tempfile::tempdir().unwrap(); @@ -524,6 +529,7 @@ async fn wildcard_join_benchmark() { /// scan; the ref keys beside them still batch. #[tokio::test(flavor = "current_thread")] async fn wildcard_incoming_probe_leaves_term_keys_to_the_scan() { + let _serial = SERIAL.lock().await; let _reset = Reset; let dir = tempfile::tempdir().unwrap(); let fluree = FlureeBuilder::file(dir.path().to_string_lossy().to_string()) From f277ae481cb72014098cc186fa878fbf16f9d625 Mon Sep 17 00:00:00 2001 From: bplatz Date: Sun, 4 Oct 2026 19:24:14 -0400 Subject: [PATCH 72/92] feat: CONSTRUCT templates write triple terms under any predicate Triple terms are values under any predicate everywhere else, but a CONSTRUCT template took one only as the object of `rdf:reifies`, and not nested even there. The JSON-LD template parser dropped them silently. A template's triple term is now a term template, built from each solution: `?d ex:mentions <<( ?s ex:p ?o )>>`, constant, nested, or as the reified triple of `rdf:reifies`. A term with an unbound component writes nothing. A blank node inside a term is fresh per solution, like any template blank node. SPARQL and JSON-LD templates both lower to it. A datalog rule head refuses one, as it refuses blank nodes. --- docs/concepts/edge-annotations.md | 3 +- docs/query/construct.md | 7 +- docs/query/sparql.md | 2 +- docs/reference/compatibility.md | 1 - fluree-db-api/src/format/construct.rs | 76 ++++++++--- .../tests/it_datalog_rules_sparql.rs | 35 ++++++ fluree-db-api/tests/it_triple_term_links.rs | 118 ++++++++++++++++++ fluree-db-query/src/datalog_rules/parse.rs | 6 + fluree-db-query/src/ir/query.rs | 27 +++- fluree-db-query/src/parse/lower.rs | 51 +++++++- fluree-db-query/src/parse/mod.rs | 8 +- fluree-db-sparql/src/lower/construct.rs | 50 ++++---- 12 files changed, 327 insertions(+), 57 deletions(-) diff --git a/docs/concepts/edge-annotations.md b/docs/concepts/edge-annotations.md index 32b5385ff5..4fd5904f65 100644 --- a/docs/concepts/edge-annotations.md +++ b/docs/concepts/edge-annotations.md @@ -537,7 +537,7 @@ A triple term's object may itself be a triple term — `<<( ex:alice ex:says <<( These produce a clear error with a span pointing at the offending construct: - **Annotation on a property-path triple.** `?s ex:p1/ex:p2 ?o {| ... |}` is rejected — the grammar only attaches annotations to simple-predicate triples. -- **Property paths and nested triple terms in a `CONSTRUCT` template's annotation.** A template annotation block (`{| ... |}`) takes simple predicates only, and a template triple term (`?r rdf:reifies <<( ... )>>`) cannot nest. +- **Property paths in a `CONSTRUCT` template's annotation.** A template annotation block (`{| ... |}`) takes simple predicates only. Annotations on literal-valued objects (plain, typed, and language-tagged) are supported on **both** the JSON-LD and SPARQL UPDATE write surfaces — the SPARQL path records the language tag for language-tagged objects so the stored annotation matches the base edge. @@ -564,7 +564,6 @@ The bare-quoted-triple form combined with an annotation tail (`<< :s :p :o >> :p Today's surface covers the common LPG / RDF-star use cases. The following are not yet supported and produce a clear validation error rather than silent partial behavior: - **Annotations on list-occurrence triples.** `@list` membership is in scope as a future extension; the on-disk format already reserves space for it. Today, annotating a list element is rejected at parse time. -- **Triple-term constants in a `CONSTRUCT` template** other than the object of `rdf:reifies`. A template variable bound to a term (`CONSTRUCT { ?d :mentions ?t }`) writes it. The mandated SPARQL 1.2 `VERSION "1.2"` prologue declaration is **accepted** (lex-and-skipped): the RDF 1.2 surface runs ungated, so a conformant 1.2 client that emits the declaration parses normally. diff --git a/docs/query/construct.md b/docs/query/construct.md index 93182e8e71..d4381f9e78 100644 --- a/docs/query/construct.md +++ b/docs/query/construct.md @@ -376,9 +376,10 @@ details of each. - `GRAPH` blocks cannot nest, and the `CONSTRUCT WHERE` shorthand has no `GRAPH` form (its template is a basic graph pattern, per SPARQL 1.1). -- A triple-term constant in a template is accepted only as the object of `rdf:reifies`; nested - triple terms and property paths inside a template annotation block are rejected. A template - variable bound to a stored triple term (`CONSTRUCT { ?d ex:mentions ?t }`) writes the term. +- A template writes a triple term under any predicate, nested ones included, built from each + solution (`CONSTRUCT { ?d ex:mentions <<( ?s ex:p ?o )>> }`); a term with an unbound component + writes nothing. A template variable bound to a stored triple term (`CONSTRUCT { ?d ex:mentions + ?t }`) writes the term. Property paths inside a template annotation block are rejected. - `?r rdf:reifies <<( s p o )>>` in a template writes the reification without `s p o`, as RDF 1.2 defines it; the annotation tail `s p o ~ ?r` writes both. `?r rdf:reifies ?t`, with `?t` bound to a triple term, writes the same as the first. A JSON-LD result writes a reification diff --git a/docs/query/sparql.md b/docs/query/sparql.md index fbd3289ec9..f64281fd32 100644 --- a/docs/query/sparql.md +++ b/docs/query/sparql.md @@ -1097,7 +1097,7 @@ Annotation tails are supported in `INSERT DATA`, `DELETE DATA`, and `INSERT { } - **Simple-predicate triples only.** `?s ex:p1/ex:p2 ?o {| ... |}` (property-path) is rejected. - **Triple terms in object position only** (a triple term's object may be another one). - **No reserved predicates by hand.** The [system predicates](../reference/vocabulary.md#edge-annotation-predicates-reserved) that back annotations are rejected on every UPDATE clause; mint annotations only through the `~` / `{| |}` surface. -- **`CONSTRUCT` template annotation blocks take simple predicates only**, and a template triple term cannot nest. +- **`CONSTRUCT` template annotation blocks take simple predicates only.** ## SPARQL UPDATE diff --git a/docs/reference/compatibility.md b/docs/reference/compatibility.md index dcab80dc04..409231ce5e 100644 --- a/docs/reference/compatibility.md +++ b/docs/reference/compatibility.md @@ -198,7 +198,6 @@ triple terms; and the triple-term functions `TRIPLE()`, `SUBJECT()`, `PREDICATE( `OBJECT()` and `isTRIPLE()`. Not yet supported: -- Triple-term constants in a `CONSTRUCT` template other than as the object of `rdf:reifies` - Nested annotations (annotation-of-annotation) **Specification:** https://www.w3.org/TR/sparql12-query/ diff --git a/fluree-db-api/src/format/construct.rs b/fluree-db-api/src/format/construct.rs index f89946b83f..c761922bf0 100644 --- a/fluree-db-api/src/format/construct.rs +++ b/fluree-db-api/src/format/construct.rs @@ -100,16 +100,27 @@ pub(super) fn instantiate_construct_graph( Ref::Var(v) => Ok(terms_slot(template, *v)), constant => terms.constant_ref(constant, position).map(Slot::Const), }; + let object_slot = |terms: &mut TermResolver<'_>, pattern: &fluree_db_query::TriplePattern| { + Ok::<_, FormatError>(match &pattern.o { + Term::Var(v) => terms_slot(template, *v), + constant => Slot::Const(terms.constant_object(constant, pattern.dtc.as_ref())?), + }) + }; + let mut term_slots = Vec::with_capacity(template.term_templates().len()); + for pattern in template.term_templates() { + term_slots.push([ + slot_of(&mut terms, &pattern.s, Position::Subject)?, + slot_of(&mut terms, &pattern.p, Position::Predicate)?, + object_slot(&mut terms, pattern)?, + ]); + } // Each pattern's slots, graph, the reifiers attached to it, and the // components of a constant triple-term object. let mut patterns = Vec::with_capacity(template.patterns().len()); for (i, pattern) in template.patterns().iter().enumerate() { let s = slot_of(&mut terms, &pattern.s, Position::Subject)?; let p = slot_of(&mut terms, &pattern.p, Position::Predicate)?; - let o = match &pattern.o { - Term::Var(v) => terms_slot(template, *v), - constant => Slot::Const(terms.constant_object(constant, pattern.dtc.as_ref())?), - }; + let o = object_slot(&mut terms, pattern)?; let const_term = match &pattern.o { Term::Value(FlakeValue::TripleTerm(term)) => terms.term_components(term)?, _ => None, @@ -145,25 +156,46 @@ pub(super) fn instantiate_construct_graph( // row's blanks. let mut row_bnodes: HashMap = HashMap::new(); let mut reifiers: Vec = Vec::new(); + // This row's triple terms, by term template; `None` where a component + // is unbound. + let mut row_terms: Vec> = Vec::with_capacity(term_slots.len()); for batch in &result.batches { for row in 0..batch.len() { row_bnodes.clear(); - let mut resolve = - |terms: &mut TermResolver<'_>, slot: &Slot, position| -> Result> { - Ok(match slot { - Slot::Const(term) => term.clone(), - Slot::Blank(v) => Some(IrTerm::BlankNode(row_blank( - *v, - &mut row_bnodes, - &mut bnode_counter, - ))), - Slot::Var(v) => match batch.get(row, *v) { - Some(binding) => terms.binding(binding, position)?, - None => None, - }, - }) - }; + let mut resolve_in = |terms: &mut TermResolver<'_>, + row_terms: &[Option], + slot: &Slot, + position| + -> Result> { + Ok(match slot { + Slot::Const(term) => term.clone(), + Slot::Blank(v) => Some(IrTerm::BlankNode(row_blank( + *v, + &mut row_bnodes, + &mut bnode_counter, + ))), + Slot::Var(v) => match batch.get(row, *v) { + Some(binding) => terms.binding(binding, position)?, + None => None, + }, + Slot::Term(i) => row_terms[*i].clone(), + }) + }; + row_terms.clear(); + for slots in &term_slots { + let mut parts: [Option; 3] = [None, None, None]; + for (part, (slot, position)) in parts.iter_mut().zip(slots.iter().zip(POSITIONS)) { + *part = resolve_in(&mut terms, &row_terms, slot, position)?; + } + row_terms.push(match parts { + [Some(s), Some(p), Some(o)] => Some(IrTerm::triple(s, p, o)), + _ => None, + }); + } + let mut resolve = |terms: &mut TermResolver<'_>, slot: &Slot, position| { + resolve_in(terms, &row_terms, slot, position) + }; 'pattern: for (slots, graph_slot, reifier_slots, const_term, asserted) in &patterns { // `?r rdf:reifies ?t` with a triple term bound to ?t (or a // constant one) writes what `?r rdf:reifies <<( s p o )>>` @@ -246,16 +278,20 @@ pub(super) fn instantiate_construct_graph( } /// A template position: a constant resolved up front, a variable bound per -/// row, or a template blank node minted per row. +/// row, a template blank node minted per row, or a triple term built per row +/// from a term template (an index into `term_templates`). enum Slot { Const(Option), Var(VarId), Blank(VarId), + Term(usize), } fn terms_slot(template: &ConstructTemplate, v: VarId) -> Slot { if template.bnode_vars.contains(&v) { Slot::Blank(v) + } else if let Some(i) = template.term_template_of(v) { + Slot::Term(i) } else { Slot::Var(v) } diff --git a/fluree-db-api/tests/it_datalog_rules_sparql.rs b/fluree-db-api/tests/it_datalog_rules_sparql.rs index 881a9a8060..e57fbfb94d 100644 --- a/fluree-db-api/tests/it_datalog_rules_sparql.rs +++ b/fluree-db-api/tests/it_datalog_rules_sparql.rs @@ -285,3 +285,38 @@ async fn sparql_rule_head_with_annotation_or_graph_rejected() { ); } } + +#[tokio::test] +async fn sparql_rule_head_with_triple_term_rejected() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger0 = genesis_ledger(&fluree, "datalog/sparql-head-term"); + let rule_data = json!({ + "@context": { "f": "https://ns.flur.ee/db#" }, + "@id": "http://example.org/termHead", + "f:rule": { + "@type": "https://ns.flur.ee/db#sparql", + "@value": "PREFIX ex: \ + CONSTRUCT { ?x ex:derived <<( ?x ex:a ?y )>> } WHERE { ?x ex:a ?y }" + } + }); + let ledger = fluree.insert(ledger0, &rule_data).await.unwrap().ledger; + let data = json!({ + "@context": { "ex": "http://example.org/" }, + "@graph": [ {"@id": "ex:thing", "ex:a": 1} ] + }); + let ledger = fluree.insert(ledger, &data).await.unwrap().ledger; + let q = json!({ + "@context": { "ex": "http://example.org/" }, + "select": "?x", + "where": {"@id": "?x", "ex:derived": "?y"}, + "reasoning": "datalog" + }); + let err = support::query_jsonld(&fluree, &ledger, &q) + .await + .expect_err("a rule head cannot write a triple term"); + let message = err.to_string(); + assert!( + message.contains("termHead") && message.contains("triple term"), + "{message}" + ); +} diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 52c41ae1ac..6ed652bac2 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -3399,3 +3399,121 @@ async fn term_bound_links_probe_in_bulk() { let ledger = fluree.ledger(ledger_id).await.expect("reload"); assert_eq!(answers(ledger, false).await, expected, "after reindex"); } + +/// A CONSTRUCT template writes a triple term under any predicate: constant, +/// built from each solution, nested, and as `rdf:reifies`' nested object. +/// A term with an unbound component writes nothing; a blank node inside +/// one is fresh per solution. JSON-LD templates build the same terms. +#[tokio::test] +async fn construct_templates_write_triple_terms_as_values() { + use fluree_db_api::format::{format_results_string, FormatterConfig}; + let fluree = FlureeBuilder::memory().build_memory(); + let ledger = fluree + .insert_turtle( + support::genesis_ledger(&fluree, "it/triple-term-links:construct-values"), + CLAIMS, + ) + .await + .expect("insert") + .ledger; + let lines = |result: &fluree_db_api::QueryResult| { + let text = format_results_string( + result, + &result.context, + &ledger.snapshot, + &FormatterConfig::ntriples(), + ) + .expect("N-Triples"); + let mut lines: Vec = text.lines().map(str::to_string).collect(); + lines.sort(); + lines + }; + let construct = |body: &'static str| { + let fluree = &fluree; + let ledger = &ledger; + async move { + let sparql = format!( + "PREFIX ex: \n\ + PREFIX rdf: \n{body}" + ); + support::query_sparql(fluree, ledger, &sparql) + .await + .unwrap_or_else(|e| panic!("{sparql}: {e}")) + } + }; + let ex = |l: &str| format!(""); + let term = |s: &str, p: &str, o: &str| format!("<<( {s} {p} {o} )>>"); + let says = |o: &str| format!("{} {} {o} .", ex("doc"), ex("says")); + + let result = construct("CONSTRUCT { ex:doc ex:says <<( ex:a ex:b ex:c )>> } WHERE {}").await; + assert_eq!(lines(&result), [says(&term(&ex("a"), &ex("b"), &ex("c")))]); + + let knows = |o: &str| term(&ex("alice"), &ex("knows"), &ex(o)); + let result = construct( + "CONSTRUCT { ex:doc ex:says <<( ex:alice ex:knows ?o )>> } \ + WHERE { ex:alice ex:knows ?o }", + ) + .await; + assert_eq!(lines(&result), [says(&knows("bob")), says(&knows("carol"))]); + + let heard = |o: &str| term(&ex("me"), &ex("heard"), &knows(o)); + let nested = [says(&heard("bob")), says(&heard("carol"))]; + let result = construct( + "CONSTRUCT { ex:doc ex:says <<( ex:me ex:heard <<( ex:alice ex:knows ?o )>> )>> } \ + WHERE { ex:alice ex:knows ?o }", + ) + .await; + assert_eq!(lines(&result), nested); + + let result = construct( + "CONSTRUCT { ex:r rdf:reifies <<( ex:me ex:heard <<( ex:alice ex:knows ?o )>> )>> } \ + WHERE { ex:alice ex:knows ?o }", + ) + .await; + let reifies = ""; + assert_eq!( + lines(&result), + [ + format!("{} {reifies} {} .", ex("r"), heard("bob")), + format!("{} {reifies} {} .", ex("r"), heard("carol")), + ] + ); + + let result = + construct("CONSTRUCT { ex:doc ex:says <<( ex:alice ex:knows ?nope )>> } WHERE {}").await; + assert!(lines(&result).is_empty(), "{:?}", lines(&result)); + + let result = construct( + "CONSTRUCT { ex:doc ex:says <<( _:b ex:knows ?o )>> } WHERE { ex:alice ex:knows ?o }", + ) + .await; + let blanks: std::collections::HashSet = lines(&result) + .iter() + .map(|l| { + let start = l.find("<<( ").expect("a term") + 4; + l[start..].split(' ').next().unwrap().to_string() + }) + .collect(); + assert_eq!( + blanks.len(), + 2, + "a fresh blank node per solution: {blanks:?}" + ); + assert!(blanks.iter().all(|b| b.starts_with("_:")), "{blanks:?}"); + + let query = json!({ + "@context": {"ex": "http://example.org/"}, + "where": {"@id": "ex:alice", "ex:knows": "?o"}, + "construct": [{ + "@id": "ex:doc", + "ex:says": {"@id": { + "@id": "ex:me", + "ex:heard": {"@id": {"@id": "ex:alice", "ex:knows": "?o"}} + }} + }] + }); + let result = support::query_jsonld(&fluree, &ledger, &query) + .await + .expect("JSON-LD construct"); + assert_eq!(lines(&result), nested); +} diff --git a/fluree-db-query/src/datalog_rules/parse.rs b/fluree-db-query/src/datalog_rules/parse.rs index 109c25621a..69f808529b 100644 --- a/fluree-db-query/src/datalog_rules/parse.rs +++ b/fluree-db-query/src/datalog_rules/parse.rs @@ -209,6 +209,12 @@ pub(super) fn parse_sparql_rule( nodes (every round would mint new ones and the fixpoint would never end)", )); } + if !template.term_templates().is_empty() { + return Err(invalid( + &label, + "the CONSTRUCT template writes a triple term; a rule head cannot", + )); + } if template.patterns().is_empty() { return Err(invalid( &label, diff --git a/fluree-db-query/src/ir/query.rs b/fluree-db-query/src/ir/query.rs index dac48cb2ce..d659c29ae6 100644 --- a/fluree-db-query/src/ir/query.rs +++ b/fluree-db-query/src/ir/query.rs @@ -10,7 +10,7 @@ //! an optional pre-resolved schema bundle). Hydration formatting lives //! inside the `Column::Hydration` variant on the SELECT projection. -use std::collections::HashSet; +use std::collections::{HashMap, HashSet}; use fluree_graph_json_ld::ParsedContext; @@ -56,6 +56,11 @@ pub struct ConstructTemplate { /// Patterns that are only reified (`r rdf:reifies <<( s p o )>>`), so /// their triple is not written. Empty when every pattern is asserted. reified_only: HashSet, + /// Triple terms the template writes as values: variable `v` stands for + /// the term `term_templates[term_vars[v]]` instantiates to on each row. + /// A term template's object may be another term variable, pushed first. + term_templates: Vec, + term_vars: HashMap, } /// A reifier attachment in a CONSTRUCT template (see @@ -84,9 +89,28 @@ impl ConstructTemplate { graphs: Vec::new(), reifications: Vec::new(), reified_only: HashSet::new(), + term_templates: Vec::new(), + term_vars: HashMap::new(), } } + /// Make `var` stand for the triple term `pattern` instantiates to. + pub fn push_term_template(&mut self, var: VarId, pattern: TriplePattern) { + self.term_vars.insert(var, self.term_templates.len()); + self.term_templates.push(pattern); + } + + /// The term templates, each after any it nests. + pub fn term_templates(&self) -> &[TriplePattern] { + &self.term_templates + } + + /// The index into [`term_templates`](Self::term_templates) of the term + /// `var` stands for. + pub fn term_template_of(&self, var: VarId) -> Option { + self.term_vars.get(&var).copied() + } + /// Append a pattern that writes into `graph` (`None`: the default graph) /// and return its index. `graphs` stays empty until the first named /// graph appears, then is kept aligned with `patterns`. @@ -155,6 +179,7 @@ impl ConstructTemplate { .chain(self.reifications.iter().map(|r| &r.reifier)); self.patterns .iter() + .chain(&self.term_templates) .flat_map(TriplePattern::referenced_vars) .chain(refs.filter_map(Ref::as_var)) } diff --git a/fluree-db-query/src/parse/lower.rs b/fluree-db-query/src/parse/lower.rs index 19caac4ef1..0b733fe1dd 100644 --- a/fluree-db-query/src/parse/lower.rs +++ b/fluree-db-query/src/parse/lower.rs @@ -1520,10 +1520,20 @@ fn lower_construct_patterns( }; lower_construct_patterns(patterns, Some(&name), encoder, vars, out)?; } - UnresolvedPattern::TripleTermValue { .. } => { - return Err(ParseError::InvalidConstruct( - "a triple-term value in a CONSTRUCT template is not supported".to_string(), - )) + UnresolvedPattern::TripleTermValue { + subject, + predicate, + term, + } => { + let o = construct_term_var(term, encoder, vars, out)?; + out.push_pattern( + TriplePattern::new( + lower_ref_term(subject, encoder, vars)?, + lower_ref_term(predicate, encoder, vars)?, + Term::Var(o), + ), + graph.cloned(), + ); } // Filters, optionals and binds have no meaning in a template. _ => {} @@ -1532,6 +1542,39 @@ fn lower_construct_patterns( Ok(()) } +/// A template variable standing for the triple term `term` instantiates to, +/// its nested term registered first. +fn construct_term_var( + term: &UnresolvedTermPattern, + encoder: &E, + vars: &mut VarRegistry, + out: &mut ConstructTemplate, +) -> Result { + let pattern = match &term.o { + UnresolvedTermObject::Value { o, dtc } => lower_triple_pattern( + &UnresolvedTriplePattern { + s: term.s.clone(), + p: term.p.clone(), + o: o.clone(), + dtc: dtc.clone(), + }, + encoder, + vars, + )?, + UnresolvedTermObject::Term(inner) => { + let o = construct_term_var(inner, encoder, vars, out)?; + TriplePattern::new( + lower_ref_term(&term.s, encoder, vars)?, + lower_ref_term(&term.p, encoder, vars)?, + Term::Var(o), + ) + } + }; + let var = vars.get_or_insert(&format!("?__tt{}", vars.len())); + out.push_term_template(var, pattern); + Ok(var) +} + // ============================================================================ // Hydration lowering // ============================================================================ diff --git a/fluree-db-query/src/parse/mod.rs b/fluree-db-query/src/parse/mod.rs index 5738084c0d..71cc189b62 100644 --- a/fluree-db-query/src/parse/mod.rs +++ b/fluree-db-query/src/parse/mod.rs @@ -551,7 +551,9 @@ fn parse_construct_query( .filter(|p| { matches!( p, - UnresolvedPattern::Triple(_) | UnresolvedPattern::EdgeAnnotation { .. } + UnresolvedPattern::Triple(_) + | UnresolvedPattern::EdgeAnnotation { .. } + | UnresolvedPattern::TripleTermValue { .. } ) }) .cloned() @@ -632,7 +634,9 @@ fn parse_construct_items( patterns.extend(temp_query.patterns.into_iter().filter(|p| { matches!( p, - UnresolvedPattern::Triple(_) | UnresolvedPattern::EdgeAnnotation { .. } + UnresolvedPattern::Triple(_) + | UnresolvedPattern::EdgeAnnotation { .. } + | UnresolvedPattern::TripleTermValue { .. } ) })); } diff --git a/fluree-db-sparql/src/lower/construct.rs b/fluree-db-sparql/src/lower/construct.rs index 652f2768c6..edd4178530 100644 --- a/fluree-db-sparql/src/lower/construct.rs +++ b/fluree-db-sparql/src/lower/construct.rs @@ -8,7 +8,7 @@ use std::sync::Arc; use crate::ast::annotation::{AnnotationVerb, ReifierId}; use crate::ast::query::{ConstructQuery, ConstructTemplate}; use crate::ast::term::QuotedTriple; -use crate::ast::{GraphName, SubjectTerm, Term}; +use crate::ast::{GraphName, SubjectTerm, Term, TripleTerm}; use fluree_db_query::ir::triple::{Ref, Term as IrTerm, TriplePattern}; use fluree_db_query::ir::{ @@ -116,24 +116,12 @@ impl LoweringContext<'_, E> { // `?r rdf:reifies <<( s p o )>>`: the triple term is the reified // triple, not asserted, and `?r` its reifier. if let Term::TripleTerm(term) = &tp.object { - if p != reifies || tp.annotation.is_some() { - return Err(LowerError::not_implemented( - "a triple term in a CONSTRUCT template is only supported as the \ - object of rdf:reifies", - tp.span, - )); + if p == reifies && tp.annotation.is_none() { + let reified = self.construct_term_pattern(term, &mut out)?; + let triple = out.push_reified_pattern(reified, graph.clone()); + out.push_reification(triple, s); + continue; } - let mut siblings = Vec::new(); - let reified = self.lower_triple_term(term, &mut siblings)?; - if !siblings.is_empty() { - return Err(LowerError::not_implemented( - "nested triple terms in a CONSTRUCT template", - tp.span, - )); - } - let triple = out.push_reified_pattern(reified, graph.clone()); - out.push_reification(triple, s); - continue; } // Carry the declared datatype into the template. @@ -224,15 +212,31 @@ impl LoweringContext<'_, E> { IrTerm::from(self.construct_reified_triple(qt, graph, out)?), None, )), - Term::TripleTerm(tt) => Err(LowerError::not_implemented( - "a triple term in a CONSTRUCT template is only supported as the object of \ - rdf:reifies", - tt.span, - )), + Term::TripleTerm(tt) => { + let pattern = self.construct_term_pattern(tt, out)?; + let var = self + .vars + .get_or_insert(&format!("?#__tt_{}", self.vars.len())); + out.push_term_template(var, pattern); + Ok((IrTerm::Var(var), None)) + } other => self.lower_object_with_constraint(other), } } + /// A template triple term's triple; a nested term becomes a term + /// template first. + fn construct_term_pattern( + &mut self, + term: &TripleTerm, + out: &mut QueryConstructTemplate, + ) -> Result { + let s = self.lower_subject(&term.subject)?; + let p = self.lower_predicate(&term.predicate)?; + let (o, dtc) = self.construct_object(&term.object, &None, out)?; + Ok(TriplePattern { s, p, o, dtc }) + } + /// Extract the CONSTRUCT WHERE shorthand's template from the lowered WHERE /// patterns: every triple, and each edge annotation's link and body. fn extract_template_from_patterns(&self, patterns: &[Pattern]) -> QueryConstructTemplate { From 870d431cfcbdbe9531bbc7bd504b876cfbe0a08a Mon Sep 17 00:00:00 2001 From: bplatz Date: Sun, 4 Oct 2026 20:40:20 -0400 Subject: [PATCH 73/92] test(api): the include-system-facts pragma twin reads the rdf:reifies link The SPARQL twin added on main expected the pragma to surface stored f:reifies* bundles. Annotations now store only the rdf:reifies link, as the JSON-LD original already asserts. --- fluree-db-api/tests/it_edge_annotations.rs | 15 ++------------- 1 file changed, 2 insertions(+), 13 deletions(-) diff --git a/fluree-db-api/tests/it_edge_annotations.rs b/fluree-db-api/tests/it_edge_annotations.rs index 6f5200a460..b54a64ad85 100644 --- a/fluree-db-api/tests/it_edge_annotations.rs +++ b/fluree-db-api/tests/it_edge_annotations.rs @@ -1988,17 +1988,6 @@ async fn pragma_include_system_facts_surfaces_f_reifies() { .expect("annotated insert"); let query = "SELECT ?p WHERE { ?p ?o }"; - let reifies = |rows: &JsonValue| { - rows.to_string() - .matches("https://ns.flur.ee/db#reifies") - .count() - }; - - let hidden = support::query_sparql_formatted(&fluree, &committed.ledger, query) - .await - .expect("query without the pragma"); - assert_eq!(reifies(&hidden), 0, "hidden without the pragma: {hidden}"); - let shown = support::query_sparql_formatted( &fluree, &committed.ledger, @@ -2007,8 +1996,8 @@ async fn pragma_include_system_facts_surfaces_f_reifies() { .await .expect("query with the pragma"); assert!( - reifies(&shown) >= 3, - "the pragma must surface the f:reifies* bundle: {shown}" + shown.to_string().contains(fluree_vocab::rdf::REIFIES), + "the pragma must surface the rdf:reifies link: {shown}" ); } From 9acf941c1e353bedf97725d41f683491a722adc6 Mon Sep 17 00:00:00 2001 From: bplatz Date: Mon, 5 Oct 2026 08:40:50 -0400 Subject: [PATCH 74/92] chore(bench): drop the budgets of the benches the arena retirement removed The annotation_hydration and annotation_planner benches went with the annotation arena; their regression-budget entries stayed behind. --- regression-budget.json | 10 ---------- 1 file changed, 10 deletions(-) diff --git a/regression-budget.json b/regression-budget.json index 17b9e86965..e2610d8800 100644 --- a/regression-budget.json +++ b/regression-budget.json @@ -63,16 +63,6 @@ "small": 5.0, "medium": 3.0 }, - "annotation_hydration": { - "tiny": 10.0, - "small": 5.0, - "medium": 5.0 - }, - "annotation_planner": { - "tiny": 10.0, - "small": 5.0, - "medium": 5.0 - }, "annotation_varlength_probe": { "tiny": 10.0, "small": 5.0, From 54a68570f6c859a2dda82b3d153a797a5093272b Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 07:52:47 -0400 Subject: [PATCH 75/92] fix(policy): lanes decline predicates that may hold triple terms A triple-term object is checked as the triple it names, but the scan and probe lanes skip per-flake filtering whenever the scanned predicate is uncovered by the view policy. On an indexed ledger a policy hiding ex:salary therefore still returned the hidden triple through `?r rdf:reifies ?t`, `<< ?s ex:salary ?o >> ...` and any other predicate holding a term of it. Under a view policy that restricts anything, the shared predicate gate now declines `rdf:reifies` and any predicate whose observed datatypes include the untagged kind triple terms carry, are unknown, or miss novelty the stats cannot see. The policy test now runs from novelty, from an index and from novelty over one, and covers a term held under another predicate. --- fluree-db-api/tests/it_edge_annotations.rs | 180 ++++++++++++++------- fluree-db-query/src/fast_path_common.rs | 41 +++++ fluree-db-query/src/policy/enforcer.rs | 11 ++ 3 files changed, 176 insertions(+), 56 deletions(-) diff --git a/fluree-db-api/tests/it_edge_annotations.rs b/fluree-db-api/tests/it_edge_annotations.rs index b54a64ad85..76e939b192 100644 --- a/fluree-db-api/tests/it_edge_annotations.rs +++ b/fluree-db-api/tests/it_edge_annotations.rs @@ -4958,8 +4958,9 @@ async fn policy_hiding_base_edge_blocks_annotation_rooted_query() { } /// A link `r rdf:reifies <<( s p o )>>` names its triple, so a policy hiding -/// the triple hides the link on every route to it, not only `@reifies`: -/// SPARQL's quoted pattern and a plain `rdf:reifies` scan. +/// the triple hides the link on every route to it, not only `@reifies`, and so +/// does a triple term held under any other predicate. Checked from novelty, from +/// an index (where the scan and probe lanes run), and from novelty over one. #[tokio::test] async fn policy_hiding_base_edge_hides_its_link() { let fluree = FlureeBuilder::memory().build_memory(); @@ -4967,66 +4968,133 @@ async fn policy_hiding_base_edge_hides_its_link() { let ledger0 = genesis_ledger(&fluree, ledger_id); let insert = json!({ "@context": ctx(), - "@id": "ex:alice", - "ex:worksFor": { - "@id": "ex:acme", - "@annotation": { "@id": "ex:emp-A", "ex:role": "Engineer" } - } + "@graph": [ + { + "@id": "ex:alice", + "ex:worksFor": { + "@id": "ex:acme", + "@annotation": { "@id": "ex:emp-A", "ex:role": "Engineer" } + } + }, + { + "@id": "ex:doc", + "ex:mentions": {"@id": {"@id": "ex:bob", "ex:worksFor": {"@id": "ex:initech"}}} + } + ] }); fluree.insert(ledger0, &insert).await.expect("insert"); - let ledger = fluree.ledger(ledger_id).await.expect("reload"); - let opts = fluree_db_api::GovernanceOptions { - policy: Some(json!([{ - "@id": "ex:hide-worksFor", - "f:required": true, - "f:onProperty": [{"@id": "http://example.org/worksFor"}], - "f:action": "f:view", - "f:query": serde_json::to_string(&json!({ - "where": {"@id": "?$identity", "@type": "http://example.org/NeverMatches"} - })).unwrap() - }])), - default_allow: Some(true), - ..Default::default() + let check = |ledger: MemoryLedger, label: &'static str, links: usize, mentions: usize| { + let fluree = &fluree; + async move { + let opts = fluree_db_api::GovernanceOptions { + policy: Some(json!([{ + "@id": "ex:hide-worksFor", + "f:required": true, + "f:onProperty": [{"@id": "http://example.org/worksFor"}], + "f:action": "f:view", + "f:query": serde_json::to_string(&json!({ + "where": {"@id": "?$identity", "@type": "http://example.org/NeverMatches"} + })).unwrap() + }])), + default_allow: Some(true), + ..Default::default() + }; + let policy = fluree_db_api::policy_builder::build_policy_context_from_opts( + &ledger.snapshot, + ledger.novelty.as_ref(), + Some(ledger.novelty.as_ref()), + ledger.t(), + &opts, + &[0], + ) + .await + .expect("policy context"); + + let reifies = ""; + let works_for = ""; + let mentions_iri = ""; + for (sparql, open) in [ + ( + format!( + "SELECT ?person ?org WHERE {{ ?r {reifies} \ + <<( ?person {works_for} ?org )>> }}" + ), + links, + ), + (format!("SELECT ?r ?t WHERE {{ ?r {reifies} ?t }}"), links), + ( + format!( + "SELECT ?s ?o WHERE {{ << ?s {works_for} ?o >> \ + ?x }}" + ), + 1, + ), + ( + format!("SELECT ?s WHERE {{ ?r {reifies} ?t BIND(SUBJECT(?t) AS ?s) }}"), + links, + ), + ( + format!( + "SELECT ?s WHERE {{ {mentions_iri} ?t \ + BIND(SUBJECT(?t) AS ?s) }}" + ), + mentions, + ), + ( + format!("SELECT ?s ?o WHERE {{ ?d {mentions_iri} <<( ?s {works_for} ?o )>> }}"), + mentions, + ), + ] { + let count = |db: fluree_db_api::GraphDb| { + let sparql = sparql.clone(); + let ledger = &ledger; + async move { + let result = fluree.query(&db, sparql.as_str()).await.expect("query"); + let rows = result.to_jsonld(&ledger.snapshot).expect("to_jsonld"); + rows.as_array().map_or(0, Vec::len) + } + }; + assert_eq!( + count(support::graphdb_from_ledger(&ledger)).await, + open, + "[{label}] {sparql}" + ); + let policed = support::graphdb_from_ledger(&ledger) + .with_policy(std::sync::Arc::new(policy.clone())); + assert_eq!(count(policed).await, 0, "[{label}] hidden: {sparql}"); + } + + let counted = format!("SELECT (COUNT(*) AS ?n) WHERE {{ ?d {mentions_iri} ?t }}"); + let policed = + support::graphdb_from_ledger(&ledger).with_policy(std::sync::Arc::new(policy)); + let result = fluree + .query(&policed, counted.as_str()) + .await + .expect("count"); + let rows = result.to_jsonld(&ledger.snapshot).expect("to_jsonld"); + assert_eq!(rows, json!([[0]]), "[{label}] {counted}"); + } }; - let policy = fluree_db_api::policy_builder::build_policy_context_from_opts( - &ledger.snapshot, - ledger.novelty.as_ref(), - Some(ledger.novelty.as_ref()), - ledger.t(), - &opts, - &[0], + + check( + fluree.ledger(ledger_id).await.expect("load"), + "novelty", + 1, + 1, ) - .await - .expect("policy context"); + .await; + support::rebuild_and_publish_index(&fluree, ledger_id).await; + let indexed = fluree.ledger(ledger_id).await.expect("load"); + check(indexed.clone(), "indexed", 1, 1).await; - let reifies = ""; - for sparql in [ - format!( - "SELECT ?person ?org WHERE {{ ?r {reifies} \ - <<( ?person ?org )>> }}" - ), - format!("SELECT ?r ?t WHERE {{ ?r {reifies} ?t }}"), - ] { - let count = |db: fluree_db_api::GraphDb| { - let fluree = &fluree; - let ledger = &ledger; - let sparql = sparql.clone(); - async move { - let result = fluree.query(&db, sparql.as_str()).await.expect("query"); - let rows = result.to_jsonld(&ledger.snapshot).expect("to_jsonld"); - rows.as_array().map_or(0, Vec::len) - } - }; - assert_eq!( - count(support::graphdb_from_ledger(&ledger)).await, - 1, - "{sparql}" - ); - let policed = - support::graphdb_from_ledger(&ledger).with_policy(std::sync::Arc::new(policy.clone())); - assert_eq!(count(policed).await, 0, "the hidden edge's link: {sparql}"); - } + let more = json!({ + "@context": ctx(), + "@id": "ex:doc", + "ex:mentions": {"@id": {"@id": "ex:carol", "ex:worksFor": {"@id": "ex:hooli"}}} + }); + let ledger = fluree.insert(indexed, &more).await.expect("insert").ledger; + check(ledger, "novelty over an index", 1, 2).await; } // ===================================================================== diff --git a/fluree-db-query/src/fast_path_common.rs b/fluree-db-query/src/fast_path_common.rs index f33b7ae72e..ecfaf62204 100644 --- a/fluree-db-query/src/fast_path_common.rs +++ b/fluree-db-query/src/fast_path_common.rs @@ -3919,6 +3919,14 @@ pub fn cursor_fast_path_for_predicate( match ctx.policy_enforcer.as_ref() { Some(enforcer) => match enforcer.classify_view_predicate(pred_sid) { PredicateCoverage::Covered => PredicateFastPath::Decline, + // A triple-term object is checked as the triple it names, whose + // predicate the view may cover even when this one is not. + PredicateCoverage::UncoveredAllow + if enforcer.view_restricts_anything() + && predicate_may_hold_triple_terms(ctx, pred_sid) => + { + PredicateFastPath::Decline + } PredicateCoverage::UncoveredAllow => PredicateFastPath::Allow, PredicateCoverage::UncoveredDeny => PredicateFastPath::Empty, }, @@ -3926,6 +3934,39 @@ pub fn cursor_fast_path_for_predicate( } } +/// Whether `pred_sid`'s objects in the active graph may include triple terms. +/// Triple terms carry no datatype tag of their own, so an `UNKNOWN` tag, an +/// unknown set, or novelty the stats do not see answers yes. +fn predicate_may_hold_triple_terms(ctx: &ExecutionContext<'_>, pred_sid: &Sid) -> bool { + if fluree_db_core::is_rdf_reifies(pred_sid) { + return true; + } + let Some(store) = ctx.binary_store.as_ref() else { + return true; + }; + let Some(p_id) = store.sid_to_p_id(pred_sid) else { + return true; + }; + let overlay = ctx.overlay(); + let stats_see_novelty = overlay + .as_any() + .downcast_ref::() + .is_some() + || overlay.epoch() == 0 + || overlay.is_effectively_empty(); + if !stats_see_novelty { + return true; + } + let stats_view = crate::stats_cache::cached_stats_view_for_db( + fluree_db_core::GraphDbRef::new(ctx.active_snapshot, ctx.binary_g_id, overlay, ctx.to_t) + .with_runtime_small_dicts_opt(ctx.runtime_small_dicts), + Some(store), + false, + ); + crate::binary_scan::observed_datatypes(stats_view.as_deref(), ctx.binary_g_id, p_id) + .is_none_or(|tags| tags.contains(&fluree_db_core::ValueTypeTag::UNKNOWN)) +} + /// Shared single-predicate fast-path gate: normalize `predicate` against /// `store` and decide whether the cursor fast path may run. /// diff --git a/fluree-db-query/src/policy/enforcer.rs b/fluree-db-query/src/policy/enforcer.rs index 5d099409ff..7b31871776 100644 --- a/fluree-db-query/src/policy/enforcer.rs +++ b/fluree-db-query/src/policy/enforcer.rs @@ -84,6 +84,17 @@ impl QueryPolicyEnforcer { } } + /// Whether the view policy can hide any flake at all. Gates the lanes' + /// triple-term check: an uncovered predicate's term rows are hidden only + /// through the restrictions on the triples they name. + pub fn view_restricts_anything(&self) -> bool { + let view = self.policy.wrapper().view(); + !(view.by_property.is_empty() + && view.by_class.is_empty() + && view.by_subject.is_empty() + && view.defaults.is_empty()) + } + /// Filter a batch of flakes by policy using explicit graph parameters. /// /// This is the **correct** method for dataset mode - it uses the graph's From c7a597759bf74851aa8de865f84b759d4d333df4 Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 07:52:47 -0400 Subject: [PATCH 76/92] fix(api): export and hydration refuse a pre-link index like queries do On an annotated ledger indexed by an earlier release, link queries refused with the reindex error, but export returned Ok with every reifier body orphaned and no markers, and a crawl dropped @annotation: both read links the index does not have. The snapshot now carries `needs_link_reindex` (annotations, no term dictionary), which also gives embedding hosts a machine-readable signal to schedule the rebuild. Queries, export and hydration's annotation reads all refuse through `require_link_index`, whose message now names the API route as well as the CLI one. A ledger written and indexed by 4.2.3 is checked in as a fixture to pin it. --- fluree-db-api/src/export_annotations.rs | 2 + fluree-db-api/src/format/hydration.rs | 2 + fluree-db-api/src/format/mod.rs | 4 + ...84f107695ae8bc40c13952a69dc02e4ac3d58.dict | Bin 0 -> 26 bytes ...a367a63132ea28107d1ba098cd8d5ac67ecb4.dict | Bin 0 -> 138 bytes ...2d39d7e47e7b1f2f93cd915194aa490a02c9e.dict | Bin 0 -> 166 bytes ...bbfab26ce536919445e99cce47ab5f11ddd69.dict | Bin 0 -> 136 bytes ...afa86e62c1408212b9396cf3cf6fcc05df367.dict | Bin 0 -> 112 bytes ...394b9f333fc561ee7c7e9dc235a6eba250930.dict | Bin 0 -> 314 bytes ...4254673b4689c3f4798583b18b2dd56afd078.dict | Bin 0 -> 96 bytes ...e01048607f07115a101deec1c1e5b2be016ca.dict | Bin 0 -> 148 bytes ...f0efe0826218f3cbfb116c188ba94598c3d79.dict | Bin 0 -> 162 bytes ...c3e1057a7f1a3aac71ff993ff22226aa391ee.fcv2 | Bin 0 -> 441 bytes ...128579ab3142eb7e4e009087e44188749cec0.json | 1 + ...101e583c4422a4010966d0a9882ba3bd71e627.fbr | Bin 0 -> 126 bytes ...124ab1094c3ab03c1a5eacd540969f1419eff4.fbr | Bin 0 -> 126 bytes ...70df6865addb18b0e9301436f5c8a8423257a0.fbr | Bin 0 -> 126 bytes ...40dd8e3eb02449a36a54bb378a04ec00e5c641.fbr | Bin 0 -> 126 bytes ...d203a3b3dc2dc3407d1e24c6e20ee6e8e317cb.fli | Bin 0 -> 1692 bytes ...8a386e79e9b3d9dc8bf64a7dfa9b34f1a4aace.fli | Bin 0 -> 1242 bytes ...4a4bb578af01599dfc928c57d5e6469f9b47f3.fli | Bin 0 -> 1692 bytes ...312ba1699633a66a2fdf7c8e706748873160bd.fli | Bin 0 -> 449 bytes ...f1624f9ab21f97a8d686634247fb88bc90e6d1.fli | Bin 0 -> 845 bytes ...1df075492c2733ab563d9e53dd29fb7394ae90.fli | Bin 0 -> 413 bytes ...78092f7d413ddcd9b86424732b47b86c8182e4.fli | Bin 0 -> 813 bytes ...cc24617785802606d1d2a5d4454996e2c46ec9.fli | Bin 0 -> 1242 bytes ...c5c10b4c3fc6dc5f72bcfcbfb13a8e74cf9da.fir6 | Bin 0 -> 6334 bytes ...d7478fd770cdfca83943b2f495b87d987ea8de.hll | 1 + .../ns@v2/ann/main.commits.jsonl | 1 + .../ns@v2/ann/main.index.json | 9 ++ .../prelink-annotations/ns@v2/ann/main.json | 18 +++ fluree-db-api/tests/it_indexing_stats.rs | 1 + fluree-db-api/tests/it_triple_term_links.rs | 104 ++++++++++++++++++ .../src/format/index_root.rs | 1 + fluree-db-core/src/db.rs | 13 +++ fluree-db-query/src/term_components.rs | 24 ++-- 36 files changed, 169 insertions(+), 12 deletions(-) create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/0af0833b7745a49796a4e6ada4f84f107695ae8bc40c13952a69dc02e4ac3d58.dict create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/0afc10f276823245798a1bf40bca367a63132ea28107d1ba098cd8d5ac67ecb4.dict create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/923811f9ab369ec31fc9b60ef9b2d39d7e47e7b1f2f93cd915194aa490a02c9e.dict create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/ac53b8d9fa13b4a7a0ea75ba004bbfab26ce536919445e99cce47ab5f11ddd69.dict create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/ac5cea69d3bc8d7b9a385ef2ef2afa86e62c1408212b9396cf3cf6fcc05df367.dict create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/bc767f51d8018ba2eabe822f6b3394b9f333fc561ee7c7e9dc235a6eba250930.dict create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/cd8210876f1fdd6c94f1c3505db4254673b4689c3f4798583b18b2dd56afd078.dict create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/d5313020fff6f26e8b562dc63c6e01048607f07115a101deec1c1e5b2be016ca.dict create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/f7a29e434d94fc3cc5982f5d882f0efe0826218f3cbfb116c188ba94598c3d79.dict create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/commit/bf4ed1f97e8d6148436ee4738d4c3e1057a7f1a3aac71ff993ff22226aa391ee.fcv2 create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/garbage/de005b16a0236b1766054836918128579ab3142eb7e4e009087e44188749cec0.json create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/branches/3958a96f0aedaf642585bc9a05101e583c4422a4010966d0a9882ba3bd71e627.fbr create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/branches/47ad903bb5f649c5ded01f7b45124ab1094c3ab03c1a5eacd540969f1419eff4.fbr create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/branches/8038c04b716c475f8d54aabd6970df6865addb18b0e9301436f5c8a8423257a0.fbr create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/branches/9e9f8f012365ed439fe28bb1bd40dd8e3eb02449a36a54bb378a04ec00e5c641.fbr create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/0912eb77eaab94811929918917d203a3b3dc2dc3407d1e24c6e20ee6e8e317cb.fli create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/42fef3fbafb414b81a36d6b9a58a386e79e9b3d9dc8bf64a7dfa9b34f1a4aace.fli create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/479c534d43eab1878737247fa54a4bb578af01599dfc928c57d5e6469f9b47f3.fli create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/672ca6e1a10470c719b97865d0312ba1699633a66a2fdf7c8e706748873160bd.fli create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/7f0f32277708d596d4210a6158f1624f9ab21f97a8d686634247fb88bc90e6d1.fli create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/a818ba5f13f820e6e2ba26ea4c1df075492c2733ab563d9e53dd29fb7394ae90.fli create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/aa954caeb82011878906cf02d278092f7d413ddcd9b86424732b47b86c8182e4.fli create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/e8d562eb81fc02cc15ba0e4e25cc24617785802606d1d2a5d4454996e2c46ec9.fli create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/roots/f72ebe9b26fd2f6acfb59502f47c5c10b4c3fc6dc5f72bcfcbfb13a8e74cf9da.fir6 create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/stats/4942defc8b3d21c699b6a26962d7478fd770cdfca83943b2f495b87d987ea8de.hll create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ns@v2/ann/main.commits.jsonl create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ns@v2/ann/main.index.json create mode 100644 fluree-db-api/tests/fixtures/prelink-annotations/ns@v2/ann/main.json diff --git a/fluree-db-api/src/export_annotations.rs b/fluree-db-api/src/export_annotations.rs index 80965ce098..28bc6bf960 100644 --- a/fluree-db-api/src/export_annotations.rs +++ b/fluree-db-api/src/export_annotations.rs @@ -105,6 +105,8 @@ impl<'a> AnnotationProbe<'a> { if !ledger.snapshot.has_annotations && !ledger.novelty.has_annotations() { return Ok(None); } + fluree_db_query::term_components::require_link_index(&ledger.snapshot)?; + let graphs = std::iter::once(0).chain( ledger .snapshot diff --git a/fluree-db-api/src/format/hydration.rs b/fluree-db-api/src/format/hydration.rs index a6fd0f4515..9589b9504f 100644 --- a/fluree-db-api/src/format/hydration.rs +++ b/fluree-db-api/src/format/hydration.rs @@ -1764,6 +1764,7 @@ impl<'a> HydrationFormatter<'a> { if !self.may_hold_annotations() { return Ok(Vec::new()); } + fluree_db_query::term_components::require_link_index(self.db.snapshot)?; // A list element is never reified. if flake.m.as_ref().is_some_and(|m| m.i.is_some()) { return Ok(Vec::new()); @@ -1899,6 +1900,7 @@ impl<'a> HydrationFormatter<'a> { if sid.namespace_code != BLANK_NODE || !self.may_hold_annotations() { return Ok(false); } + fluree_db_query::term_components::require_link_index(self.db.snapshot)?; let links = self .db .range( diff --git a/fluree-db-api/src/format/mod.rs b/fluree-db-api/src/format/mod.rs index f96da05175..fc82164f98 100644 --- a/fluree-db-api/src/format/mod.rs +++ b/fluree-db-api/src/format/mod.rs @@ -267,6 +267,10 @@ pub enum FormatError { /// Fuel limit exceeded during formatting (expansion) #[error(transparent)] FuelExceeded(#[from] FuelExceededError), + + /// A read the ledger's index cannot answer + #[error(transparent)] + Query(#[from] fluree_db_query::QueryError), } /// Result type for formatting operations diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/0af0833b7745a49796a4e6ada4f84f107695ae8bc40c13952a69dc02e4ac3d58.dict b/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/0af0833b7745a49796a4e6ada4f84f107695ae8bc40c13952a69dc02e4ac3d58.dict new file mode 100644 index 0000000000000000000000000000000000000000..56fa70ec1c3e30727cdb49f45092003c9db87d51 GIT binary patch literal 26 YcmeZsax`RQU|?Y61rb2Z2_irM02*Qe9RL6T literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/0afc10f276823245798a1bf40bca367a63132ea28107d1ba098cd8d5ac67ecb4.dict b/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/0afc10f276823245798a1bf40bca367a63132ea28107d1ba098cd8d5ac67ecb4.dict new file mode 100644 index 0000000000000000000000000000000000000000..a7dea1b24369857e132b138315cf35fd5efc30ba GIT binary patch literal 138 zcmZ<@@HS**Ps(Pb@_+B%PIw7JdN}c!B>Pp>y1sEpCh3uLDvVhd;@~^NF#8&ApAZ qgH_hjGf*MSs>E12HB^{a7B&zRX$)qyqS@kP5v-sZKS6v>?&-devJ^T1 literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/ac5cea69d3bc8d7b9a385ef2ef2afa86e62c1408212b9396cf3cf6fcc05df367.dict b/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/ac5cea69d3bc8d7b9a385ef2ef2afa86e62c1408212b9396cf3cf6fcc05df367.dict new file mode 100644 index 0000000000000000000000000000000000000000..6d55ff42bc11043fcb48f38da33f4a3a8bd4e1e7 GIT binary patch literal 112 zcmXYpF%Ez*3#J$j#&-{^`n|P8JitS?kTbRPzYw62&n(F7l&sSkpKVy literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/bc767f51d8018ba2eabe822f6b3394b9f333fc561ee7c7e9dc235a6eba250930.dict b/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/bc767f51d8018ba2eabe822f6b3394b9f333fc561ee7c7e9dc235a6eba250930.dict new file mode 100644 index 0000000000000000000000000000000000000000..8176baa61b08cdcbb79fc12a826b5aaf83c614bb GIT binary patch literal 314 zcmY+A-HAdm5QVcp?vIcL?BRm5i8smXB0h-^b}&gMta#nPb-@m{hi$$fs~w6DShy15tRd?Qo=Q2RAiWmWY!EP0a9AV!_@|ADer1re)Ugx*Vf;8x2>;mr2dcG N<$k@L<^J&+JOi7oF#G@j literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/cd8210876f1fdd6c94f1c3505db4254673b4689c3f4798583b18b2dd56afd078.dict b/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/cd8210876f1fdd6c94f1c3505db4254673b4689c3f4798583b18b2dd56afd078.dict new file mode 100644 index 0000000000000000000000000000000000000000..4c450703b1e5fd5ea41c6828034a0136459cfb45 GIT binary patch literal 96 zcmZ<@@HS**|%%HN_&)GReZw)I7yPS2xKJtN=)W L)M@~+3J?PTYNZIZ literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/d5313020fff6f26e8b562dc63c6e01048607f07115a101deec1c1e5b2be016ca.dict b/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/d5313020fff6f26e8b562dc63c6e01048607f07115a101deec1c1e5b2be016ca.dict new file mode 100644 index 0000000000000000000000000000000000000000..fe98c0e369352199b6a96742e4cdd813db53718e GIT binary patch literal 148 zcmZ<@@HS**KmZsUL^mLW8JM7K4j}CT#IAYinR%(HMM-HUsVRnOmgcDzDQ1Qy7AD4K zsi`LB#uh0i$;PRM2BzkT=4pnB#)*l^=7wo$mX^k8X+}U`mY8U4X_%S{vjSwk1`tO8 FF#yGw7l;4= literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/f7a29e434d94fc3cc5982f5d882f0efe0826218f3cbfb116c188ba94598c3d79.dict b/fluree-db-api/tests/fixtures/prelink-annotations/ann/@shared/dicts/f7a29e434d94fc3cc5982f5d882f0efe0826218f3cbfb116c188ba94598c3d79.dict new file mode 100644 index 0000000000000000000000000000000000000000..730f7723819c8231f1e1674a6094a226b76b6c40 GIT binary patch literal 162 zcmXAhK@!3s3;?SqpCcxea_czy13wTTX|WY;<r@cL{>d^c-Ro&AN6Z6{F zc7hy~Si@at6?OTgdsiO#3!Y jX?#KnGvtV(ar?-uV=C**FYNt<`3qw%dc)nl%~19q|H3b4 literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/commit/bf4ed1f97e8d6148436ee4738d4c3e1057a7f1a3aac71ff993ff22226aa391ee.fcv2 b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/commit/bf4ed1f97e8d6148436ee4738d4c3e1057a7f1a3aac71ff993ff22226aa391ee.fcv2 new file mode 100644 index 0000000000000000000000000000000000000000..324b3372184654691ef3352c2f911d9abcd90f36 GIT binary patch literal 441 zcmZ>B4l`n5WMp7q;09tFAVvi&B86^Xe8IjMU2Md|tqMg~S^x`qb2 z24*3KhE~QVR)%JJ#ummF76#e|237_J42%pKTlN1kL>y&e2s*+mGGXh>0LRt4*X_9a z?f;~Y*L+Xhvpy-#oY;7ErDk|hQkLreOAJn&1uBmtL|>+dJ)N%E_~+AmRc1a0ciskW z{o)75vST!T7!Nuo_|0vy+aTe<`0Zc7ECv?V#GK6JREe~dBwa&`l%&)Yi$u#L3qw=$ z6boJ5Btzz;{3N#IoW#srLk^De{G#k)xBMc$qSVZ^%+%uG(xj}^4vElN&h&Cbg&FXl?l&r8cpNzF@6WoF7uVJj{v%FIh=Pf1PA z%uUQ;XGu)XO@+7%ZY5Jj5o>ZH&?Jz9Ag1hvhBGMW_d(g(K>8(=9SWrFK$1Wp48(y@ Kwg!-$2*d!eH-AF_ literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/garbage/de005b16a0236b1766054836918128579ab3142eb7e4e009087e44188749cec0.json b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/garbage/de005b16a0236b1766054836918128579ab3142eb7e4e009087e44188749cec0.json new file mode 100644 index 0000000000..572d32b126 --- /dev/null +++ b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/garbage/de005b16a0236b1766054836918128579ab3142eb7e4e009087e44188749cec0.json @@ -0,0 +1 @@ +{"ledger_id":"ann:main","t":1,"garbage":[],"created_at_ms":1791286460435} \ No newline at end of file diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/branches/3958a96f0aedaf642585bc9a05101e583c4422a4010966d0a9882ba3bd71e627.fbr b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/branches/3958a96f0aedaf642585bc9a05101e583c4422a4010966d0a9882ba3bd71e627.fbr new file mode 100644 index 0000000000000000000000000000000000000000..4fa247a2cabdb433ae34dc3aa3e1490f3d27e96d GIT binary patch literal 126 zcmZ>B3NmJ7Vq{=sKn9#}J||EZB=;W(7#bjIKw@Cfzz?Kgvdj!@P+@fj#=eFFj6w=8 mt|q;1{KIrcbQhnW>KT>9^4115wu_gRUUBuD_UK67Nd^FZbR8uC literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/branches/47ad903bb5f649c5ded01f7b45124ab1094c3ab03c1a5eacd540969f1419eff4.fbr b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/branches/47ad903bb5f649c5ded01f7b45124ab1094c3ab03c1a5eacd540969f1419eff4.fbr new file mode 100644 index 0000000000000000000000000000000000000000..b51b73bd32834b6cfa50ce03f11708182878449b GIT binary patch literal 126 zcmZ>B3NmJ7W@KPwKn9#JP6I!X1(N#@1k6xTPPl52I+!d&0~=Icoq@5h;Q*tM!m6o0 j>vkvzws*3fXS!6usbA}8d*|kk6qRCa_Z>NnO-~pAQN0@0 literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/branches/8038c04b716c475f8d54aabd6970df6865addb18b0e9301436f5c8a8423257a0.fbr b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/branches/8038c04b716c475f8d54aabd6970df6865addb18b0e9301436f5c8a8423257a0.fbr new file mode 100644 index 0000000000000000000000000000000000000000..6a5bd5ec1d6a5c9d18ceab12439abdc448dda202 GIT binary patch literal 126 zcmZ>B3NmJ706}EH3FmVHg+X%vfqB3NmJ7WMp7uKn9#}J||EZB=;W(7#bjIKw@Cfzz?Kgvdj!@P+@fj#=eFFj6w=d n|33d-zeQw+l-aeNOS>%cDqn8Cd8hlESM9IaCLfoqI>!J2u`VED literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/0912eb77eaab94811929918917d203a3b3dc2dc3407d1e24c6e20ee6e8e317cb.fli b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/0912eb77eaab94811929918917d203a3b3dc2dc3407d1e24c6e20ee6e8e317cb.fli new file mode 100644 index 0000000000000000000000000000000000000000..1d54923f194b5366ecb6f9d5634e27e8f535cf04 GIT binary patch literal 1692 zcmcJPO-{ow5QWFhkE5VSWyub@fn6my0YXATV$DG~fmv{*-i-|^?|EX!4zUDMGS#%U z-^}FmIQ8@E%Y)g8bRu#AAI&%`gCCnSY#ei;b1^JYS_6sBpJHx7VV-qNr5)iFK88qg z3HJlcIS%tiOntzdkTrbCuq~nUCo+kr?sYj+>U2NMQzr5}!6&9G%=U0>wc|L<$#jo- zU}VJH5~eCL?Pg4qVpFCH#gyqCqfMCV$fV~!^=VR?l!?7prqilS$@|neR92hysh~6r zcKrH+ulGJBOl|B_HDf9$El9Law@qQ{lG2gA0nkha7bI zT$SnWbo{>V9>XOyNL=8vc#rsn`bH!CWdETWRlQ@zs7dIB2OHDF1Awme7^zp#AZ&Mx u{9g&yW2%3%1MA;Q5G&Yh-~GbA8D9DLUu`OdXJE|F`jvHc5hm+m2>$^`j-XTk literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/42fef3fbafb414b81a36d6b9a58a386e79e9b3d9dc8bf64a7dfa9b34f1a4aace.fli b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/42fef3fbafb414b81a36d6b9a58a386e79e9b3d9dc8bf64a7dfa9b34f1a4aace.fli new file mode 100644 index 0000000000000000000000000000000000000000..b5c08f64dd6a7d7f7cc34130d21546ebd8e0a864 GIT binary patch literal 1242 zcmcJO%?-jZ422yvAj&@*04Hv(!~lp35)v~oOBP^=4#A-k8!-cg=Qs%ni0Fa1a?|9! zIGBElsUN7dAyrMAa>LZv6;b+-q%QL$VHyS|-$~!&h%y8TcZTSQ zOu{q@OxoZK{Ds=}DWW|4e}X#+(>O5sJ`EgGL>Yrb`gBRf6uAXBE}~2V)Ag@Rt8;Oi v_qM2MBd4(7Zlf&dg-cSV)!W)@@wc@P1h1`K+;8j;Xa9g6 literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/479c534d43eab1878737247fa54a4bb578af01599dfc928c57d5e6469f9b47f3.fli b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/479c534d43eab1878737247fa54a4bb578af01599dfc928c57d5e6469f9b47f3.fli new file mode 100644 index 0000000000000000000000000000000000000000..daa720d45a5f2e45ec09c4103b37d85b0b27016c GIT binary patch literal 1692 zcmcJPOHRZv42GR1eW<9?uw(~pU^fz+03jhEvF0G0ptImeb2m1l;s5h&n~7l2T5{{e zvHdxT^7irmtdvM6B3Do^N(N_*t+&(}g1OF*n=vk@2PuSVf2G($R>rj>Ln(WxhAJY- zHI#-85#b+6kJ6td*zp1#T2RUKJ{%Q@v)q7h-*mI=_e{$_8NhRE% zR`qS`R84y$8L>jygiMveQpJa4OdQPJ((jQWNvMCIZ;136^{?~;BK?21KWn~6(;LG) z?)KPx95wG!!=u1~$Kbh`?^KodNk5cb&G^iaIyIaSU0V(7RrHKGQSXM+lc!EUeBo-) yKsU!Y4`qlg92*{4K&#;_LB{j*Ed+ct@3y)>N3|%IU!QkrJpCFN#XoCyw|@YQ3qR)o literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/7f0f32277708d596d4210a6158f1624f9ab21f97a8d686634247fb88bc90e6d1.fli b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/7f0f32277708d596d4210a6158f1624f9ab21f97a8d686634247fb88bc90e6d1.fli new file mode 100644 index 0000000000000000000000000000000000000000..94ebf2a790df6ab8a3b7b56304b397bc78e74e5a GIT binary patch literal 845 zcmcIi!Ab)$5S?_ho4DHrMZ})U7J3jw3JRWj@gh~cioN&={zJZC@80z05BLv)-yjHn zhv<8gW{cH5I_;ax%*#w>l0Cn=I5C+B{!#&ypqlCnf+d$ShCR_;|TIR4?v0yIaUR6GK1xvG_RMp+kP|}4vVqu$0M#q8rC=R T(RvO2IUGO3j!hLckcRsPNz7qc literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/a818ba5f13f820e6e2ba26ea4c1df075492c2733ab563d9e53dd29fb7394ae90.fli b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/a818ba5f13f820e6e2ba26ea4c1df075492c2733ab563d9e53dd29fb7394ae90.fli new file mode 100644 index 0000000000000000000000000000000000000000..c4ab1b4c2f63a059cac693f691b5a9f986c28d82 GIT binary patch literal 413 zcma)%Jqp4=5QX35M~uG#6;f)ajRt>eJ3+Cv6I8s&;sJ7p9Kym*&)@~Th{iWrf)*Ce z=3{2|z1iI?*2{?iP)a0^pePWSqLhq*kJ5x&XsYO}p7G`QdsgSO%+YB^N_-+^?FJ;) z4&rIuBN1^R&WgE1YQ&{~$2=nH_vufVTSWc8qJJJeyRk30S~cL~?J9b&4L3Rtb7OR~ zKUblsH`YmmG)Zo#9atHUPf=XE8P5a{!90G2slf2NkSYwl8paDTK5eIav6pOsu5e#) GPs(q_)HV$O literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/aa954caeb82011878906cf02d278092f7d413ddcd9b86424732b47b86c8182e4.fli b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/objects/leaves/aa954caeb82011878906cf02d278092f7d413ddcd9b86424732b47b86c8182e4.fli new file mode 100644 index 0000000000000000000000000000000000000000..f3d655692ef30899d3bef4a15bd847773a8bc86a GIT binary patch literal 813 zcmZ`%yKVw85L|;n$NN%5B25%JL<16SBoZMZMM{|;NJ+;hIG@4?L`su_A_Wb91!QLJ zy$ctyV(i_W9eZSM_4+b*brJle324+qO3nzR18K#E^~EKoWeoL%T-6p#Cu#epWvyE3 zWes?or zF>Pa5dy3Q_Zz?_Bp~?YzWPSsvd3c;4H=Z7s7d(neIMhV2{=b*L&ZuXlf1rK@N{{A= z9+~%)#R89fSL5k%in`=Ea~b3B>G5IsDYAi2C!rJRP?M)a<4f7@JHMLvSM{++$==RM zpOZZ<_CjS3z+Id9hscfg>; zesYxk@vs#Wq8YJcQ5IC)f}eQ6D9|Cz$l+FbJc(%+fKN9F)+ mU;!V<6*CN47Hw7ws?5Q6`s-8NRc-un!)tJjDlu;v*4Zc{C`9Ek%^zBidjUn`%1jUY3z%W}TU3S<~oh zPsIdZNs$<9eA3iZtYS2cRvUV3Oq({TX-bovDoImIDm5qZQCm|^?SE%>h6Q%JLr;6p z*}XIOe*feC_rL%D&z)cONRe)(9#Lp$Ga;H~OB!j0MJQgHq5xB-Lz$>v6x}5@o7d~j z^%mxGywg^YmzQs=S-hy)UXL5;EIrB&0m?5fF1F51;A2I)IfoUp3}{LS zIPlD~0b21jV3T$~ZZJgJYWke|wLICwyA-B%u;d3#z zIcl>^$b=k>qoe`*O3Wx3ktvKz>|A3bBbEUFf(MIAEJ%kQD~TDZF6YLqlM!-REZR6$ zv^g;+xM_Z^D=QYoAP>)$1bD@vL`6N^OgFk+ShAfJjH^cmB>LPq4z%!ce1kBb@tg}-MI(Xp@OF5LLJchE72=Qu z**Uh3abVUCltx5596S~TBN5Lu!jK8MdECTs9vGt#NHidUStnx!0rMh^n?Rt*({@qN zp^3o@R2(KtY-V#Bu}F^(IV-8)90Q9kV;s{IlZ5vsLF5_Md4FVJEeGqSSq(^r6|ocZ zVcag*m0g@B(GF~98flk?)i^j$tqV_$yqY=}M~e?c#yP+Xmciy}SPX$`fDu=CFofW+ z2iQfk&T46wY@;DtP`%_~S)GyR5jWOihXeL~IXFJY%LjWTtj^D5UlsELqJR$@lq>^`5 zzw-X)iKl*Wpx<*AEqwLp)VHdcw2IZw{rc*<(>EqvVJ!cN_{~k$7B1t8tL?^xrK?U& znJ=8G-}~jt-8HkvoxHOA_@(B!AYx-63agSW1J;{On+oRno|7eU6%H&<|M>N5m$Sako}FUOQ-Oa!v~TX>wt>>ucH35MvBm$( zG&OTa>8o8SKihJ)ZROTEz9C#YztPXUb8g$?duFY^`RVk}H}}m*GG%6exBWMzcLusw z-EtBz1@R+D5zA-ap6#92@oalX-?5H=EU`B2`cdo8V-t5xXD;imcFw7h%^@cs`T76* zuh^fcdWQT#i;+9^KL?&58B!WfyBxc8NJnXiN`d&LM|v&jKnybgTEm$paV!d;lS3TP zVc8oPM*~trh!W~K9&&^x1!{dFHUTszvH~7fV4N&=ECQX)Kr%x+Nfj$vfFf5rMKxJN z%;*#=hqc54NnvC$EXO{KowUd_VbF~voj4%Oi%jU|1SW_oUbDERLkS_zY7xf{Rnm$K zTHISsW@7=Y!CRz+mH=1{M!rX2_fNk>*{NM#Ye=Jqy-ix?5}v?~LRMaZA1*Bqr0 zr43E9zOOw}g>-<`&HfEcx^n&?{2i zhhqFGIzy~KRTo8dC+h#psujL+0;wJ{yCsn4Kqm6cKe%DuvL z)FLbQRxh60-__DGbMl7oFIafG`8c)m`GIG)uITBTzyEucx1b97UyC#+{MFlkv~6SB z15ZDh`dj_smoDdAD0?D(^2Lwh`~LE$)L%jJQAH|-MJNi!lvSe98g|2&f@xk;Py6pO zEp*L|+9i8VW;}c3_nYnImH*swZfD=yz?p&#g^k!IZG?YT_rkTV%Ez`X`{8-U{XsoG z_J?sNuH_{a{p}Y=$_rK;fbbhY?%ThE%EF(_-TzT**ZHz5+rM;Ta@FBA%YHKRDZ_R2 z=Zjy_smS7*7dgu(#~o9?b8b(RHNB>^;tL&=x$e@7Te1(IU)whgf=+;E<*}W!Pv5C} zt@l#KM9 z3-TLa74QeWywB2rw14_x`*Ls4-qt}o2H6KjgVrsb5G6p$iGi|`v_M%&xTmag7?;8` z=tM>&G;Hr+RFtBfUW#HmDLUz;sH2mj3CW)=QcmtgkER^RYDle%dZ_5>jrXtE^HNgo zkFS1cHm$E1*HZOXHvpfMfK}za1Fds1FFyD3A*S}-${p{z-yAqnJon`7U9UW`d;O8# Fe*=!pt2+Py literal 0 HcmV?d00001 diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/stats/4942defc8b3d21c699b6a26962d7478fd770cdfca83943b2f495b87d987ea8de.hll b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/stats/4942defc8b3d21c699b6a26962d7478fd770cdfca83943b2f495b87d987ea8de.hll new file mode 100644 index 0000000000..460448d42a --- /dev/null +++ b/fluree-db-api/tests/fixtures/prelink-annotations/ann/main/index/stats/4942defc8b3d21c699b6a26962d7478fd770cdfca83943b2f495b87d987ea8de.hll @@ -0,0 +1 @@ +{"version":1,"index_t":1,"entries":[{"g_id":0,"p_id":1,"count":1,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000030000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000100000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[16,1]]},{"g_id":0,"p_id":2,"count":2,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000030000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000030000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000010000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000001000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[16,2]]},{"g_id":0,"p_id":3,"count":2,"values_hll":"00000000060000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000010000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000010000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000001000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[16,2]]},{"g_id":0,"p_id":4,"count":2,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000100000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000030000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000010000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000001000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[16,2]]},{"g_id":0,"p_id":5,"count":1,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000003000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000010000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[0,1]]},{"g_id":0,"p_id":6,"count":1,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000010000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000010000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[16,1]]},{"g_id":0,"p_id":7,"count":1,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000100000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000200000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[16,1]]},{"g_id":0,"p_id":8,"count":1,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000100000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000001000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[255,1]]},{"g_id":1,"p_id":9,"count":1,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000003000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000200000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[0,1]]},{"g_id":1,"p_id":10,"count":1,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000010000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000200000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[3,1]]},{"g_id":1,"p_id":12,"count":1,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000010000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000200000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[2,1]]},{"g_id":1,"p_id":13,"count":1,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000100000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000200000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[3,1]]},{"g_id":1,"p_id":14,"count":1,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000002000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000200000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[2,1]]},{"g_id":1,"p_id":15,"count":1,"values_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000020000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","subjects_hll":"00000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000200000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000","last_modified_t":1,"datatypes":[[2,1]]}]} \ No newline at end of file diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ns@v2/ann/main.commits.jsonl b/fluree-db-api/tests/fixtures/prelink-annotations/ns@v2/ann/main.commits.jsonl new file mode 100644 index 0000000000..0e04141333 --- /dev/null +++ b/fluree-db-api/tests/fixtures/prelink-annotations/ns@v2/ann/main.commits.jsonl @@ -0,0 +1 @@ +{"t":1,"cid":"bagaybqabciql6twr7f7i2ykiinxoi44njq7bav5h6gr2vry77gj76ircnkrzd3q"} diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ns@v2/ann/main.index.json b/fluree-db-api/tests/fixtures/prelink-annotations/ns@v2/ann/main.index.json new file mode 100644 index 0000000000..6916dd6fbd --- /dev/null +++ b/fluree-db-api/tests/fixtures/prelink-annotations/ns@v2/ann/main.index.json @@ -0,0 +1,9 @@ +{ + "@context": { + "f": "https://ns.flur.ee/db#" + }, + "f:ledgerIndex": { + "f:cid": "baghybqabciqpolv6tmtp2l3kz62zkaxuprobbngd7rw4l5zlz7f7we5i45gptwq", + "f:t": 1 + } +} \ No newline at end of file diff --git a/fluree-db-api/tests/fixtures/prelink-annotations/ns@v2/ann/main.json b/fluree-db-api/tests/fixtures/prelink-annotations/ns@v2/ann/main.json new file mode 100644 index 0000000000..90f352b7d8 --- /dev/null +++ b/fluree-db-api/tests/fixtures/prelink-annotations/ns@v2/ann/main.json @@ -0,0 +1,18 @@ +{ + "@context": { + "f": "https://ns.flur.ee/db#" + }, + "@id": "ann:main", + "@type": [ + "f:LedgerSource" + ], + "f:ledger": { + "@id": "ann" + }, + "f:branch": "main", + "f:commitCid": "bagaybqabciql6twr7f7i2ykiinxoi44njq7bav5h6gr2vry77gj76ircnkrzd3q", + "f:t": 1, + "f:status": "ready", + "f:statusV": 1, + "f:configV": 0 +} \ No newline at end of file diff --git a/fluree-db-api/tests/it_indexing_stats.rs b/fluree-db-api/tests/it_indexing_stats.rs index 16f027b78a..11aa21a2ff 100644 --- a/fluree-db-api/tests/it_indexing_stats.rs +++ b/fluree-db-api/tests/it_indexing_stats.rs @@ -73,6 +73,7 @@ async fn apply_index( string_watermark: root.string_watermark, graph_iris: root.graph_iris, has_annotations: root.has_annotations, + needs_link_reindex: root.has_annotations && root.term_dict.is_none(), has_list_meta: root.has_list_meta, }; let mut db = LedgerSnapshot::new_meta(meta).expect("seed graph registry from root"); diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 6ed652bac2..984c04d991 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2244,6 +2244,84 @@ async fn construct_writes_triple_terms_as_reifications() { ); } +/// A ledger written and indexed by 4.2.3, whose annotations are `f:reifies*` +/// bundles with no links (`tests/fixtures/prelink-annotations`). Every read of +/// its annotations refuses with the reindex error rather than answering as if +/// it had none; reads that never touch annotations answer. +#[tokio::test] +async fn a_release_built_pre_link_index_refuses_every_annotation_read() { + fn copy_dir(from: &std::path::Path, to: &std::path::Path) { + std::fs::create_dir_all(to).expect("mkdir"); + for entry in std::fs::read_dir(from).expect("read fixture") { + let entry = entry.expect("entry"); + let target = to.join(entry.file_name()); + if entry.file_type().expect("type").is_dir() { + copy_dir(&entry.path(), &target); + } else { + std::fs::copy(entry.path(), &target).expect("copy"); + } + } + } + let tmp = tempfile::TempDir::new().expect("tmp"); + copy_dir( + &std::path::Path::new(env!("CARGO_MANIFEST_DIR")) + .join("tests/fixtures/prelink-annotations"), + tmp.path(), + ); + let fluree = FlureeBuilder::file(tmp.path().to_string_lossy().to_string()) + .build() + .expect("build"); + let ledger = fluree.ledger("ann:main").await.expect("load"); + assert!(ledger.snapshot.needs_link_reindex); + let refused = |err: String| assert!(err.contains("fluree reindex"), "{err}"); + + for format in [ + fluree_db_api::export::ExportFormat::Turtle, + fluree_db_api::export::ExportFormat::JsonLd, + ] { + let mut out = Vec::new(); + let result = fluree + .export("ann:main") + .format(format) + .write_to(&mut out) + .await; + refused(match result { + Ok(_) => format!("{format:?} exported:\n{}", String::from_utf8_lossy(&out)), + Err(e) => e.to_string(), + }); + } + let crawl = json!({ + "@context": {"ex": "http://example.org/"}, + "select": {"ex:alice": ["*"]} + }); + refused( + match support::query_jsonld_formatted(&fluree, &ledger, &crawl).await { + Ok(rows) => format!("crawled: {rows}"), + Err(e) => e.to_string(), + }, + ); + refused( + support::query_sparql( + &fluree, + &ledger, + "PREFIX ex: \n\ + SELECT ?role WHERE { << ex:alice ex:worksFor ex:acme >> ex:role ?role }", + ) + .await + .expect_err("a reified pattern refuses") + .to_string(), + ); + + let plain = support::query_sparql_formatted( + &fluree, + &ledger, + "PREFIX ex: \nSELECT ?o WHERE { ex:bob ex:knows ?o }", + ) + .await + .expect("a read without annotations answers"); + assert_eq!(plain.as_array().map(Vec::len), Some(1), "{plain}"); +} + /// An annotated ledger whose index predates links (simulated by dropping the /// root's term dictionary) refuses link reads until a full rebuild links its /// annotations. A new link declines the incremental build, whose term @@ -2303,10 +2381,36 @@ async fn link_reads_refuse_an_index_built_before_links() { .expect("publish root"); let ledger = fluree.ledger(ledger_id).await.expect("load"); + assert!(ledger.snapshot.needs_link_reindex); let err = support::query_sparql(&fluree, &ledger, query) .await .expect_err("a pre-link index refuses link reads"); assert!(err.to_string().contains("fluree reindex"), "{err}"); + // Export and the crawl's `@annotation` read links too; without them each + // would drop every annotation's attachment and report success. + for format in [ + fluree_db_api::export::ExportFormat::Turtle, + fluree_db_api::export::ExportFormat::JsonLd, + ] { + let err = fluree + .export(ledger_id) + .format(format) + .write_to(&mut Vec::new()) + .await + .expect_err("a pre-link index refuses export"); + assert!( + err.to_string().contains("fluree reindex"), + "{format:?}: {err}" + ); + } + let crawl = json!({ + "@context": {"ex": "http://example.org/"}, + "select": {"ex:s1": ["*"]} + }); + let err = support::query_jsonld_formatted(&fluree, &ledger, &crawl) + .await + .expect_err("a pre-link index refuses the crawl's annotations"); + assert!(err.to_string().contains("fluree reindex"), "{err}"); fluree .upsert_turtle(ledger, &claim(2)) diff --git a/fluree-db-binary-index/src/format/index_root.rs b/fluree-db-binary-index/src/format/index_root.rs index a74c7d3dd4..79838089f7 100644 --- a/fluree-db-binary-index/src/format/index_root.rs +++ b/fluree-db-binary-index/src/format/index_root.rs @@ -824,6 +824,7 @@ impl IndexRoot { string_watermark: self.string_watermark, graph_iris: self.graph_iris.clone(), has_annotations: self.has_annotations, + needs_link_reindex: self.has_annotations && self.term_dict.is_none(), has_list_meta: self.has_list_meta, }) } diff --git a/fluree-db-core/src/db.rs b/fluree-db-core/src/db.rs index 708dafa9c7..07f373f82c 100644 --- a/fluree-db-core/src/db.rs +++ b/fluree-db-core/src/db.rs @@ -63,6 +63,10 @@ pub struct LedgerSnapshotMetadata { /// the per-retract POST scan entirely. pub has_annotations: bool, + /// The index holds annotations but predates RDF 1.2 triple-term links, + /// so it cannot answer a link read until it is rebuilt from commits. + pub needs_link_reindex: bool, + /// Whether any indexed row carries an RDF-list position. `Some(false)` /// lets the write path skip list-meta hydration; `None` means the root /// predates tracking and lists must be assumed possible. See @@ -159,6 +163,10 @@ pub struct LedgerSnapshot { /// cost. pub has_annotations: bool, + /// The index holds annotations but predates RDF 1.2 triple-term links, + /// so it cannot answer a link read until it is rebuilt from commits. + pub needs_link_reindex: bool, + /// Whether any indexed row carries an RDF-list position. `Some(false)` /// lets the write path skip list-meta hydration; `None` means the root /// predates tracking and lists must be assumed possible. See @@ -185,6 +193,7 @@ impl Clone for LedgerSnapshot { range_provider: self.range_provider.clone(), graph_registry: self.graph_registry.clone(), has_annotations: self.has_annotations, + needs_link_reindex: self.needs_link_reindex, has_list_meta: self.has_list_meta, } } @@ -240,6 +249,7 @@ impl LedgerSnapshot { string_watermark: 0, range_provider: None, has_annotations: false, + needs_link_reindex: false, // An empty snapshot has no indexed rows, so "no list rows" is // exact — everything lives in novelty, which tracks its own bit. has_list_meta: Some(false), @@ -279,6 +289,7 @@ impl LedgerSnapshot { range_provider: None, graph_registry, has_annotations: meta.has_annotations, + needs_link_reindex: meta.needs_link_reindex, has_list_meta: meta.has_list_meta, }) } @@ -982,6 +993,7 @@ fn decode_fir6_metadata(bytes: &[u8]) -> std::io::Result string_watermark, graph_iris, has_annotations, + needs_link_reindex: has_annotations && !has_term_dict_section, has_list_meta, }) } @@ -1043,6 +1055,7 @@ mod tests { string_watermark: 0, graph_iris: vec![], has_annotations: false, + needs_link_reindex: false, has_list_meta: None, }) .unwrap(); diff --git a/fluree-db-query/src/term_components.rs b/fluree-db-query/src/term_components.rs index f71d588f1a..115f5e23c7 100644 --- a/fluree-db-query/src/term_components.rs +++ b/fluree-db-query/src/term_components.rs @@ -397,27 +397,27 @@ impl TermComponentsOperator { /// object tree falls back to the predicate scan. const OBJECT_SITE: &str = "term-object"; -/// A component no dictionary can name is a miss, not an error. +pub(crate) fn require_indexed_links(ctx: &ExecutionContext<'_>) -> Result<()> { + require_link_index(ctx.active_snapshot) +} + /// Refuse to read links from an index built before them. Such an index holds /// an annotated ledger's annotations only as `f:reifies*` bundles, so a link -/// read would answer as if they did not exist. -pub(crate) fn require_indexed_links(ctx: &ExecutionContext<'_>) -> Result<()> { - let pre_link = ctx.active_snapshot.has_annotations - && ctx - .binary_store - .as_ref() - .is_some_and(|store| !store.has_term_dict()); - if pre_link { +/// read (a query, an export, a crawl's `@annotation`) would answer as if they +/// did not exist. +pub fn require_link_index(snapshot: &fluree_db_core::LedgerSnapshot) -> Result<()> { + if snapshot.needs_link_reindex { return Err(QueryError::UnsupportedFeature( - "this ledger's index predates RDF 1.2 triple-term links, so it cannot answer \ - reified-triple patterns over its annotations; rebuild the index \ - (`fluree reindex `)" + "this ledger's index predates RDF 1.2 triple-term links, so it cannot read \ + its annotations; rebuild the index in full (`fluree reindex ` from \ + the CLI, `Fluree::reindex` from the API)" .to_string(), )); } Ok(()) } +/// A component no dictionary can name is a miss, not an error. fn missing(r: std::io::Result) -> std::io::Result> { match r { Ok(v) => Ok(Some(v)), From 2b23f9e22f0f2b3584d8c99a0d43eb16c6388b84 Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 07:52:56 -0400 Subject: [PATCH 77/92] perf(query): seed the semijoin per chunk; build first for a large outer side Past 1,024 distinct outer keys the semijoin fell back to an unseeded build over the whole body: an annotated edge's EXISTS on a large predicate cost the predicate, not the query (IC7, and 1,100 hub likes over 1M ex:likes at 2-3x main's time and ~11x its memory). And because the outer side was pulled before the build, a child holding state (a hash table, OPTIONAL buckets) kept it through the inner build: NOT EXISTS after OPTIONAL rose from 106 to 139 MiB peak with no annotations involved. The planner now passes estimates of the outer block and the body. An outer side estimated small against a triples-only body is seeded a chunk of up to 1,024 keys at a time, so the cost follows the outer side; if the seeded keys outgrow the estimate, it switches to one unseeded build. Otherwise the unseeded build runs before the child is opened, as on main. Buffered outer rows and each chunk's keys are charged to the query budget and released as they drain. The outer estimate treats a term decomposition as the join it is rather than a scan of every term. --- fluree-db-api/tests/it_query_negation.rs | 68 ++++ fluree-db-query/src/execute/where_plan.rs | 77 ++++- fluree-db-query/src/semijoin.rs | 362 ++++++++++++++++------ 3 files changed, 406 insertions(+), 101 deletions(-) diff --git a/fluree-db-api/tests/it_query_negation.rs b/fluree-db-api/tests/it_query_negation.rs index 4023a7eb0e..45780fc1a9 100644 --- a/fluree-db-api/tests/it_query_negation.rs +++ b/fluree-db-api/tests/it_query_negation.rs @@ -1092,6 +1092,74 @@ SELECT ?p ?org WHERE { } } +/// An outer side past one seeded build's keys is seeded a chunk at a time, +/// so annotated edges on a much larger predicate are checked against their +/// base triples without building the whole predicate, and every chunk is +/// probed against its own keys. +#[tokio::test] +async fn semijoin_seeds_a_large_outer_side_chunk_by_chunk() { + use fluree_db_api::{QueryInput, ReindexOptions}; + + const HUB_LIKES: usize = 2_500; + const OTHER_LIKES: usize = 60_000; + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "negation:chunked-semijoin"; + let mut ttl = String::from("VERSION \"1.2\"\n@prefix ex: .\n"); + for i in 0..HUB_LIKES { + ttl.push_str(&format!("ex:post{i} ex:author ex:hub .\n")); + // Every tenth edge is reified without being asserted. + if i % 10 == 0 { + ttl.push_str(&format!( + "<< ex:liker{i} ex:likes ex:post{i} >> ex:at {i} .\n" + )); + } else { + ttl.push_str(&format!( + "ex:liker{i} ex:likes ex:post{i} {{| ex:at {i} |}} .\n" + )); + } + } + for i in 0..OTHER_LIKES { + ttl.push_str(&format!("ex:fan{i} ex:likes ex:other{} .\n", i % 97)); + } + fluree + .upsert_turtle(genesis_ledger(&fluree, ledger_id), &ttl) + .await + .expect("seed"); + fluree + .reindex(ledger_id, ReindexOptions::default()) + .await + .expect("reindex"); + + let asserted = HUB_LIKES - HUB_LIKES / 10; + let count = |body: &str| { + format!( + "PREFIX ex: \n\ + SELECT (COUNT(*) AS ?n) WHERE {{ ?post ex:author ex:hub . {body} }}" + ) + }; + let view = fluree.db(ledger_id).await.expect("view"); + for (query, expected) in [ + (count("?liker ex:likes ?post {| ex:at ?at |}"), asserted), + ( + count( + "<< ?liker ex:likes ?post >> ex:at ?at \ + FILTER NOT EXISTS { ?liker ex:likes ?post }", + ), + HUB_LIKES / 10, + ), + ] { + let result = fluree + .query(&view, QueryInput::Sparql(&query)) + .await + .expect("query"); + assert_eq!( + result.to_jsonld(&view.snapshot).unwrap(), + json!([[expected]]), + "{query}" + ); + } +} + /// A missing OPTIONAL binding must use a reusable existence lookup, while /// bound values still constrain the inner match and outer duplicates survive. #[tokio::test] diff --git a/fluree-db-query/src/execute/where_plan.rs b/fluree-db-query/src/execute/where_plan.rs index 2fde422c45..49a663a635 100644 --- a/fluree-db-query/src/execute/where_plan.rs +++ b/fluree-db-query/src/execute/where_plan.rs @@ -51,12 +51,10 @@ use super::pushdown::extract_bounds_from_filters; // ============================================================================ // // `Pattern::EdgeAnnotation { edge, annotation, body }` is flattened -// at planner time into the equivalent triple chain over the -// `f:reifies*` system predicates. The standard scan / join machinery -// handles the rest. This avoids a custom operator and exercises the -// existing visibility / policy / dedup paths automatically — the -// base edge triple's standard scan provides the visibility check -// "for free". +// at planner time into the body, the reifier's `rdf:reifies` link with +// its term components, and an existence check on the base edge. The +// standard scan / join machinery handles the rest. Policy on the link is +// checked on its triple term (`QueryPolicyEnforcer::term_visible`). /// Expand every `Pattern::EdgeAnnotation` in `patterns` into its /// triple-chain equivalent, recursing through every container pattern (`Optional`, `Union`, `Minus`, `Exists`, @@ -534,6 +532,53 @@ pub(crate) fn choose_exists_strategy( } } +/// Row estimate for the patterns ahead of an EXISTS. A term decomposition +/// whose term a triple of the block binds decodes one term per row, joining +/// its components to that triple; the branch estimate would price it as a +/// scan of every term. Estimate it as that join: each component variable +/// stands for the term variable. +fn estimate_outer_rows(patterns: &[Pattern], stats: &StatsView) -> f64 { + let triple_vars: HashSet = patterns + .iter() + .filter(|p| matches!(p, Pattern::Triple(_))) + .flat_map(Pattern::produced_vars) + .collect(); + let mut alias: HashMap = HashMap::new(); + for p in patterns { + if let Pattern::TermComponents(tc) = p { + if triple_vars.contains(&tc.term) { + for c in tc.components() { + if let crate::ir::Component::Var(v) = c { + alias.insert(*v, tc.term); + } + } + } + } + } + let aliased = |v: &VarId| *alias.get(v).unwrap_or(v); + let rows: Vec = patterns + .iter() + .filter(|p| !matches!(p, Pattern::TermComponents(tc) if triple_vars.contains(&tc.term))) + .map(|p| match p { + Pattern::Triple(tp) if !alias.is_empty() => { + let mut tp = tp.clone(); + if let Ref::Var(v) = &mut tp.s { + *v = aliased(v); + } + if let Ref::Var(v) = &mut tp.p { + *v = aliased(v); + } + if let Term::Var(v) = &mut tp.o { + *v = aliased(v); + } + Pattern::Triple(tp) + } + other => other.clone(), + }) + .collect(); + crate::planner::estimate_branch_cardinality(&rows, Some(stats)) +} + /// Build the operator for an EXISTS / NOT EXISTS using [`choose_exists_strategy`]. /// /// Picks `SemijoinOperator` (build-once + hash probe) when the inner pattern is @@ -542,8 +587,12 @@ pub(crate) fn choose_exists_strategy( /// / `Pattern::NotExists` dispatch and the `OPTIONAL { ... } FILTER(!bound(?v))` /// rewrite — the latter relied on `ExistsOperator` unconditionally before this /// helper existed, which timed out on large outer streams. +/// +/// `outer_patterns` are the patterns the child evaluates, whose estimate +/// against the inner body's lets the semijoin choose a seeded build. fn build_exists_strategy( child: BoxedOperator, + outer_patterns: &[Pattern], inner_patterns: &[Pattern], negated: bool, stats: Option>, @@ -559,14 +608,24 @@ fn build_exists_strategy( inner_pattern_count = inner_patterns.len(), "exists dispatch", ); - Box::new(SemijoinOperator::new( + let estimates = stats.as_deref().map(|s| { + ( + estimate_outer_rows(outer_patterns, s), + crate::planner::estimate_branch_cardinality(inner_patterns, Some(s)), + ) + }); + let op = SemijoinOperator::new( child, inner_patterns.to_vec(), key_vars, negated, stats, planning, - )) + ); + Box::new(match estimates { + Some((outer, inner)) => op.with_estimates(outer, inner), + None => op, + }) } ExistsStrategy::Exists { reason } => { tracing::debug!( @@ -3067,6 +3126,7 @@ pub fn build_where_operators_seeded_with_needed( if !v_bound_in_outer && !v_appears_later && v_bound_by_inner { operator = Some(build_exists_strategy( child, + &patterns[..i], inner_patterns, true, stats.clone(), @@ -3172,6 +3232,7 @@ pub fn build_where_operators_seeded_with_needed( let negated = matches!(&patterns[i], Pattern::NotExists(_)); operator = Some(build_exists_strategy( child, + &patterns[..i], inner_patterns, negated, stats.clone(), diff --git a/fluree-db-query/src/semijoin.rs b/fluree-db-query/src/semijoin.rs index a32ab7d8b7..1c13c3d2b7 100644 --- a/fluree-db-query/src/semijoin.rs +++ b/fluree-db-query/src/semijoin.rs @@ -3,9 +3,10 @@ //! Replaces per-row correlated subquery evaluation with a build-probe approach: //! //! 1. **Build phase** (`open`): Execute inner patterns once, collect distinct key -//! tuples (the correlation variables) into a `HashSet`. An outer side with -//! few distinct keys seeds the build with them, so a point query does not -//! pay for the whole inner relation. +//! tuples (the correlation variables) into a `HashSet`. When the outer side +//! is small against the inner relation, the build is instead seeded by the +//! outer keys, one bounded chunk of outer rows at a time, so the cost +//! follows the outer side rather than the whole inner relation. //! 2. **Probe phase** (`next_batch`): For each outer row, extract key var values and //! probe the set. EXISTS keeps matches; NOT EXISTS keeps non-matches. //! @@ -37,15 +38,18 @@ use std::sync::Arc; /// masks use the existing seeded evaluation; cached masks remain reusable. const MAX_PARTIAL_KEY_SETS: usize = 4; -/// Outer sides with at most this many distinct keys build the inner side -/// seeded by those keys. Each seeded key costs an index lookup where the -/// unseeded build costs a scan row, so the unseeded build wins once the -/// outer side is a sizeable fraction of the inner relation. +/// Distinct outer keys per seeded build; a larger outer side is seeded one +/// chunk of this many keys at a time. const SEEDED_BUILD_MAX_KEYS: usize = 1024; -/// Bound on the outer rows buffered while deciding, whatever their keys. +/// Bound on the outer rows buffered per chunk, whatever their keys. const SEEDED_BUILD_MAX_ROWS: usize = 64 * 1024; +/// A seeded key costs an index lookup where the unseeded build costs a scan +/// row: seeding wins while the outer side is under this fraction of the +/// inner relation. +const SEEDED_KEY_COST: f64 = 16.0; + /// Approximate retained key storage, shared by the base and projected sets. /// Counts the tuple and its cells; excludes table slack and shared payloads. fn key_entry_bytes(width: usize) -> usize { @@ -74,12 +78,21 @@ pub struct SemijoinOperator { partial_keys_safe: bool, /// Key positions (not batch columns) -> projected, normalized inner keys. partial_key_sets: FxHashMap, FxHashSet>, - /// Outer batches read during `open()` to size the build, replayed first. + /// The current chunk's outer rows, which `key_set` was seeded for. buffered: VecDeque, + /// Query-budget bytes charged for `buffered`, released as it drains. + buffered_bytes: usize, + /// Query-budget bytes charged for a seeded `key_set`, released per chunk. + key_set_bytes: usize, /// The child returned `None` while buffering. child_exhausted: bool, - /// The build was seeded by the outer keys. + /// Builds are seeded per chunk of outer rows; otherwise one unseeded build + /// precedes opening the child, so the two never hold memory together. seeded: bool, + /// Outer keys seeded so far, across chunks. + seeded_keys: usize, + /// Planner estimates of the outer and inner row counts, when stats exist. + estimates: Option<(f64, f64)>, /// Column indices of key_vars within child.schema(), computed in `open()`. key_col_indices: Vec, /// Stats for nested query building. @@ -117,8 +130,12 @@ impl SemijoinOperator { partial_keys_safe, partial_key_sets: FxHashMap::default(), buffered: VecDeque::new(), + buffered_bytes: 0, + key_set_bytes: 0, child_exhausted: false, seeded: false, + seeded_keys: 0, + estimates: None, norm: None, key_col_indices: Vec::new(), stats, @@ -178,6 +195,9 @@ impl SemijoinOperator { )); } ctx.record_alloc((projected.len() - charged_rows) * entry_bytes); + if self.seeded { + self.key_set_bytes += projected.len() * entry_bytes; + } ctx.checkpoint()?; tracing::debug!( bound_keys = positions.len(), @@ -240,68 +260,55 @@ impl SemijoinOperator { } impl SemijoinOperator { - /// Outer batches buffered by `open()` first, then the child's. + /// Planner estimates of the outer side's and the inner body's row counts, + /// which choose between seeded and unseeded builds. + pub fn with_estimates(mut self, outer_rows: f64, inner_rows: f64) -> Self { + self.estimates = Some((outer_rows, inner_rows)); + self + } + + /// Seed only a conjunction of triples (a seeded solution of it is a + /// solution of the unseeded body), and only an outer side estimated small + /// against the body. Without estimates, seed: memory stays per chunk. + fn seeds(&self) -> bool { + self.partial_keys_safe + && self + .estimates + .is_none_or(|(outer, inner)| outer * SEEDED_KEY_COST < inner) + } + + /// Seeded keys have passed the point where an unseeded build is cheaper: + /// the outer estimate was low. + fn seeding_outgrown(&self) -> bool { + self.estimates + .is_some_and(|(_, inner)| self.seeded_keys as f64 * SEEDED_KEY_COST >= inner) + } + + /// The current chunk's rows first; once it drains, the next chunk under a + /// seeded build, otherwise the child's. async fn next_child_batch(&mut self, ctx: &ExecutionContext<'_>) -> Result> { + if self.buffered.is_empty() && self.seeded && !self.child_exhausted { + self.load_chunk(ctx).await?; + } if let Some(batch) = self.buffered.pop_front() { + let bytes = batch_bytes(&batch).min(self.buffered_bytes); + ctx.release(bytes); + self.buffered_bytes -= bytes; return Ok(Some(batch)); } - if self.child_exhausted { + if self.child_exhausted || self.seeded { return Ok(None); } self.child.next_batch(ctx).await } -} - -/// Composite key over the columns `cols` of one row. -fn row_key( - batch: &Batch, - row_idx: usize, - cols: &[usize], - norm: &Option, -) -> CompositeGroupKey { - CompositeGroupKey::normalized(cols.iter().map(|&ci| batch.get_by_col(row_idx, ci)), norm) -} - -#[async_trait] -impl Operator for SemijoinOperator { - fn plan_children(&self) -> Vec> { - vec![crate::plan_node::PlanChild::child(self.child.as_ref())] - } - fn schema(&self) -> &[VarId] { - &self.schema - } - - async fn open(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { - if self.state != OperatorState::Created { - return Err(QueryError::Internal( - "SemijoinOperator::open() called in invalid state".into(), - )); - } - if self.norm.is_none() { - self.norm = equality_norm(ctx); - } - - // Compute key column indices for the child (outer) schema. - let child_schema = self.child.schema().to_vec(); - self.key_col_indices = self - .key_vars - .iter() - .map(|kv| { - child_schema.iter().position(|v| v == kv).ok_or_else(|| { - QueryError::Internal(format!("key var {kv:?} not found in child schema")) - }) - }) - .collect::>>()?; - self.child.open(ctx).await?; - // Read the outer side until it ends or shows more distinct keys than a - // seeded build should look up. Only a conjunction of triples seeds: a - // seeded solution of it is a solution of the unseeded body. + /// Buffer the next chunk of outer rows and seed `key_set` with its keys, + /// or, once seeding is outgrown, build unseeded and stop chunking. + async fn load_chunk(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { let mut seen: FxHashSet = FxHashSet::default(); let mut seed_rows: Vec> = Vec::new(); - let mut overflow = !self.partial_keys_safe; - let mut buffered_rows = 0usize; - while !overflow { + let mut rows = 0usize; + while seen.len() < SEEDED_BUILD_MAX_KEYS && rows < SEEDED_BUILD_MAX_ROWS { let Some(batch) = self.child.next_batch(ctx).await? else { self.child_exhausted = true; break; @@ -325,36 +332,52 @@ impl Operator for SemijoinOperator { .map(|&ci| batch.get_by_col(row_idx, ci).clone()) .collect(), ); - if seen.len() > SEEDED_BUILD_MAX_KEYS { - overflow = true; - break; - } } } - buffered_rows += batch.len(); - overflow |= buffered_rows > SEEDED_BUILD_MAX_ROWS; + rows += batch.len(); + let bytes = batch_bytes(&batch); + ctx.record_alloc(bytes); + self.buffered_bytes += bytes; + ctx.checkpoint()?; self.buffered.push_back(batch); } drop(seen); - self.seeded = !overflow; + if self.buffered.is_empty() { + return Ok(()); + } + self.key_set.clear(); + self.partial_key_sets.clear(); + ctx.release(self.key_set_bytes); + self.key_set_bytes = 0; + self.seeded_keys += seed_rows.len(); + if self.seeding_outgrown() { + self.seeded = false; + tracing::debug!(seeded_keys = self.seeded_keys, "semijoin build unseeded"); + return self.build(ctx, None).await; + } + let schema: Arc<[VarId]> = Arc::from(self.key_vars.clone().into_boxed_slice()); + let columns = (0..self.key_vars.len()) + .map(|col| seed_rows.iter().map(|r| r[col].clone()).collect()) + .collect(); + drop(seed_rows); + let before = ctx.mem_used(); + self.build(ctx, Some(Batch::new(schema, columns)?)).await?; + self.key_set_bytes = ctx.mem_used().saturating_sub(before); + Ok(()) + } - // Build phase: execute inner patterns once, collect distinct key tuples. - #[allow(clippy::box_default)] - let seed: BoxedOperator = if overflow { - Box::new(EmptyOperator::new()) - } else { - let schema: Arc<[VarId]> = Arc::from(self.key_vars.clone().into_boxed_slice()); - let columns = (0..self.key_vars.len()) - .map(|col| seed_rows.iter().map(|r| r[col].clone()).collect()) - .collect(); - Box::new(BatchSeedOperator::from_batch(Batch::new(schema, columns)?)) - }; + /// Execute the inner patterns, seeded by `seed` when given, into `key_set`. + async fn build(&mut self, ctx: &ExecutionContext<'_>, seed: Option) -> Result<()> { tracing::debug!( - seeded = self.seeded, - seed_keys = seed_rows.len(), + seeded = seed.is_some(), + seed_keys = seed.as_ref().map_or(0, Batch::len), "semijoin build" ); - drop(seed_rows); + #[allow(clippy::box_default)] + let seed: BoxedOperator = match seed { + Some(batch) => Box::new(BatchSeedOperator::from_batch(batch)), + None => Box::new(EmptyOperator::new()), + }; let mut inner_op = build_where_operators_seeded( Some(seed), &self.inner_patterns, @@ -396,7 +419,66 @@ impl Operator for SemijoinOperator { .await; // Also close the inner plan when its build exceeds the budget. inner_op.close(); - build_result?; + build_result + } +} + +/// Query-budget estimate for a buffered outer batch. +fn batch_bytes(batch: &Batch) -> usize { + batch.len() * batch.schema().len() * crate::context::BINDING_EST_BYTES +} + +/// Composite key over the columns `cols` of one row. +fn row_key( + batch: &Batch, + row_idx: usize, + cols: &[usize], + norm: &Option, +) -> CompositeGroupKey { + CompositeGroupKey::normalized(cols.iter().map(|&ci| batch.get_by_col(row_idx, ci)), norm) +} + +#[async_trait] +impl Operator for SemijoinOperator { + fn plan_children(&self) -> Vec> { + vec![crate::plan_node::PlanChild::child(self.child.as_ref())] + } + fn schema(&self) -> &[VarId] { + &self.schema + } + + async fn open(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { + if self.state != OperatorState::Created { + return Err(QueryError::Internal( + "SemijoinOperator::open() called in invalid state".into(), + )); + } + if self.norm.is_none() { + self.norm = equality_norm(ctx); + } + + // Compute key column indices for the child (outer) schema. + let child_schema = self.child.schema().to_vec(); + self.key_col_indices = self + .key_vars + .iter() + .map(|kv| { + child_schema.iter().position(|v| v == kv).ok_or_else(|| { + QueryError::Internal(format!("key var {kv:?} not found in child schema")) + }) + }) + .collect::>>()?; + + self.seeded = self.seeds(); + if self.seeded { + self.child.open(ctx).await?; + self.load_chunk(ctx).await?; + } else { + // Build before opening the child, so state the child holds once + // open (a hash table, OPTIONAL buckets) never coexists with it. + self.build(ctx, None).await?; + self.child.open(ctx).await?; + } self.state = OperatorState::Open; Ok(()) @@ -429,6 +511,8 @@ impl Operator for SemijoinOperator { self.key_set.clear(); self.partial_key_sets.clear(); self.buffered.clear(); + self.buffered_bytes = 0; + self.key_set_bytes = 0; self.state = OperatorState::Closed; } @@ -664,11 +748,49 @@ mod tests { assert!(op.partial_key_sets.is_empty()); } - /// A conjunction of triples seeds its build from an outer side with few - /// distinct keys (unbound keys included); more keys, or another body - /// shape, build the whole body. + /// An outer side delivered in batches of `size` rows, recording the query + /// memory in use when it is opened. + struct Batches { + batches: VecDeque, + schema: Arc<[VarId]>, + mem_at_open: Arc>>, + } + + impl Batches { + fn new(rows: Vec>, size: usize) -> Self { + let schema: Arc<[VarId]> = Arc::from(vec![VarId(0), VarId(1), VarId(2)]); + Self { + batches: rows + .chunks(size) + .map(|chunk| batch(chunk.to_vec())) + .collect(), + schema, + mem_at_open: Arc::default(), + } + } + } + + #[async_trait] + impl Operator for Batches { + fn schema(&self) -> &[VarId] { + &self.schema + } + async fn open(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { + *self.mem_at_open.lock().unwrap() = Some(ctx.mem_used()); + Ok(()) + } + async fn next_batch(&mut self, _: &ExecutionContext<'_>) -> Result> { + Ok(self.batches.pop_front()) + } + fn close(&mut self) {} + } + + /// A conjunction of triples seeds its build from the outer keys (unbound + /// keys included), a chunk at a time, unless the outer side is estimated + /// large against the body or seeded keys outgrow that estimate; another + /// body shape builds the whole body before the outer side is read. #[tokio::test] - async fn build_is_seeded_only_for_few_keys_over_triples() { + async fn build_is_seeded_per_chunk_for_a_small_outer_side_over_triples() { let snapshot = LedgerSnapshot::genesis("test:main"); let vars = VarRegistry::new(); let ctx = ExecutionContext::new(&snapshot, &vars); @@ -691,30 +813,84 @@ mod tests { vars: vec![VarId(0), VarId(1), VarId(2)], rows: vec![], }; - for (n, body, seeded) in [ - (SEEDED_BUILD_MAX_KEYS as u64, triple(), true), - (SEEDED_BUILD_MAX_KEYS as u64 + 1, triple(), false), - (2, values, false), + let keys = SEEDED_BUILD_MAX_KEYS as u64; + let few = Some((10.0, 1e6)); + // (outer rows, body, estimates, seeded once open, seeded keys at the end) + for (n, body, estimates, seeded, seeded_keys) in [ + (keys, triple(), None, true, keys), + (3 * keys, triple(), None, true, 3 * keys), + (3 * keys, triple(), few, true, 3 * keys), + (3 * keys, triple(), Some((1e6, 1e6)), false, 0), + // Seeding stops once its keys reach 1/16 of the body's estimate. + ( + 3 * keys, + triple(), + Some((10.0, 32.0 * keys as f64)), + true, + 2 * keys, + ), + (2, values, None, false, 0), ] { let mut op = SemijoinOperator::new( - Box::new(BatchSeedOperator::from_batch(batch(rows(n)))), + Box::new(Batches::new(rows(n), 256)), vec![body], vec![VarId(0), VarId(1), VarId(2)], false, None, PlanningContext::current(), ); + op.estimates = estimates; op.open(&ctx).await.unwrap(); - assert_eq!(op.seeded, seeded, "{n} keys"); + assert_eq!(op.seeded, seeded, "{n} keys, {estimates:?}"); let mut replayed = 0; while let Some(batch) = op.next_child_batch(&ctx).await.unwrap() { replayed += batch.len(); } assert_eq!(replayed as u64, n, "every outer row is replayed"); + assert_eq!( + op.seeded_keys as u64, seeded_keys, + "{n} keys, {estimates:?}" + ); + assert_eq!(op.buffered_bytes, 0); op.close(); } } + /// An unseeded build runs before the outer side is opened, so the two + /// never hold memory at once. + #[tokio::test] + async fn unseeded_build_precedes_opening_the_outer_side() { + let snapshot = LedgerSnapshot::genesis("test:main"); + let vars = VarRegistry::new(); + let cancellation = QueryCancellation::new(); + cancellation.set_memory_limit(usize::MAX); + let ctx = ExecutionContext::new(&snapshot, &vars).with_cancellation(cancellation); + let child = Batches::new( + vec![vec![ + Binding::encoded_sid(1), + Binding::encoded_sid(2), + Binding::Unbound, + ]], + 1, + ); + let mem_at_open = Arc::clone(&child.mem_at_open); + let mut op = SemijoinOperator::new( + Box::new(child), + vec![Pattern::Values { + vars: vec![VarId(0), VarId(1)], + rows: vec![vec![Binding::encoded_sid(1), Binding::encoded_sid(2)]], + }], + vec![VarId(0), VarId(1)], + false, + None, + PlanningContext::current(), + ); + op.open(&ctx).await.unwrap(); + assert!(!op.seeded); + assert_eq!(*mem_at_open.lock().unwrap(), Some(key_entry_bytes(2))); + assert_eq!(op.next_batch(&ctx).await.unwrap().unwrap().len(), 1); + } + #[tokio::test] async fn base_lookup_charges_only_distinct_keys_and_enforces_budget() { let snapshot = LedgerSnapshot::genesis("test:main"); From 441d047fc349e647751278ae3727cfa828c814ee Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 08:07:16 -0400 Subject: [PATCH 78/92] perf(export): read the link set without cloning it twice Export reads the ledger's live links up front. It cloned every link's term into the edge map, then cloned every edge key again to group the asserted check by (subject, predicate). Terms are now moved into the map and the grouping sorts borrowed keys; only edges found unasserted are cloned. 200k annotated edges, Turtle to a sink (debug build): peak RSS 395 to 287 MiB, against 56 MiB with --raw-reifies. The set is still held whole. --- fluree-db-api/src/export_annotations.rs | 50 +++++++++++++++---------- 1 file changed, 31 insertions(+), 19 deletions(-) diff --git a/fluree-db-api/src/export_annotations.rs b/fluree-db-api/src/export_annotations.rs index 28bc6bf960..b6cb2106de 100644 --- a/fluree-db-api/src/export_annotations.rs +++ b/fluree-db-api/src/export_annotations.rs @@ -9,7 +9,7 @@ use fluree_db_core::comparator::IndexType; use fluree_db_core::range::{range_with_overlay, RangeMatch, RangeOptions, RangeTest}; -use fluree_db_core::{EdgeKey, FlakeValue, GraphId, Sid}; +use fluree_db_core::{EdgeKey, FlakeValue, GraphId, Sid, TripleTermValue}; use std::collections::{HashMap, HashSet}; use std::io; use std::marker::PhantomData; @@ -127,33 +127,41 @@ impl<'a> AnnotationProbe<'a> { RangeOptions::new().with_to_t(as_of_t), ) .await?; + // Moved, not cloned: the map holds each term's parts. + let graph_links = links.entry(g_id).or_default(); for flake in flakes { - let FlakeValue::TripleTerm(term) = &flake.o else { + let FlakeValue::TripleTerm(term) = flake.o else { continue; }; - let edge = term_edge(term); - let reifiers = links.entry(g_id).or_default().entry(edge).or_default(); + let TripleTermValue { s, p, o, dt, lang } = *term; + let edge = EdgeKey { + g: None, + s, + p, + o, + dt, + lang, + list_i: None, + }; + let reifiers = graph_links.entry(edge).or_default(); if !reifiers.contains(&flake.s) { reifiers.push(flake.s); } } } for (&g_id, graph_links) in &mut links { - let mut by_subject_predicate: HashMap<(Sid, Sid), Vec> = HashMap::new(); - for edge in graph_links.keys() { - by_subject_predicate - .entry((edge.s.clone(), edge.p.clone())) - .or_default() - .push(edge.clone()); - } - for ((s, p), edges) in by_subject_predicate { + // One lookup per (subject, predicate), grouped over borrowed keys. + let mut edges: Vec<&EdgeKey> = graph_links.keys().collect(); + edges.sort_unstable_by(|a, b| (&a.s, &a.p).cmp(&(&b.s, &b.p))); + let mut missing: Vec = Vec::new(); + for group in edges.chunk_by(|a, b| a.s == b.s && a.p == b.p) { let asserted: HashSet = range_with_overlay( &ledger.snapshot, g_id, ledger.novelty.as_ref(), IndexType::Spot, RangeTest::Eq, - RangeMatch::subject_predicate(s, p), + RangeMatch::subject_predicate(group[0].s.clone(), group[0].p.clone()), RangeOptions::new().with_to_t(as_of_t), ) .await? @@ -164,12 +172,16 @@ impl<'a> AnnotationProbe<'a> { ..EdgeKey::from_flake(flake) }) .collect(); - for edge in edges { - if !asserted.contains(&edge) { - for reifier in graph_links.remove(&edge).unwrap_or_default() { - unasserted.insert((g_id, reifier, edge.clone())); - } - } + missing.extend( + group + .iter() + .filter(|edge| !asserted.contains(**edge)) + .map(|edge| (*edge).clone()), + ); + } + for edge in missing { + for reifier in graph_links.remove(&edge).unwrap_or_default() { + unasserted.insert((g_id, reifier, edge.clone())); } } } From 5cbbeeda9dd0843650fdfe23a5f929a4f9b306e8 Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 08:17:26 -0400 Subject: [PATCH 79/92] perf(query): OPTIONAL annotations keep the batched hash-join lane MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An annotated edge's expansion ends in an existence check on the base edge, which the OPTIONAL lane's safety check rejected, so OPTIONAL { s p o {| … |} } and a Cypher relationship binding went back to per-row rebuilds. An EXISTS of plain triples reads only the row's bindings, as a triple does, and is now admitted. It also counts as tolerating an unbound variable, so a row whose correlation key is unbound still takes the per-row path. --- fluree-db-api/tests/it_edge_annotations.rs | 51 ++++++++++++++++++++++ fluree-db-query/src/optional.rs | 24 +++++++++- 2 files changed, 73 insertions(+), 2 deletions(-) diff --git a/fluree-db-api/tests/it_edge_annotations.rs b/fluree-db-api/tests/it_edge_annotations.rs index 76e939b192..6e9fdb66c9 100644 --- a/fluree-db-api/tests/it_edge_annotations.rs +++ b/fluree-db-api/tests/it_edge_annotations.rs @@ -5097,6 +5097,57 @@ async fn policy_hiding_base_edge_hides_its_link() { check(ledger, "novelty over an index", 1, 2).await; } +/// An annotation inside OPTIONAL (a Cypher relationship binding's shape) takes +/// the batched hash-join lane: its expansion ends in an existence check on the +/// base edge, which reads only the row's bindings. That check still excludes a +/// reifier of an unasserted triple. +#[tokio::test] +async fn optional_annotation_takes_the_batched_lane() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/edge-annotations:optional-batched"; + fluree + .upsert_turtle( + genesis_ledger(&fluree, ledger_id), + "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:a a ex:P . ex:e a ex:P . ex:f a ex:P .\n\ + ex:a ex:knows ex:b {| ex:since 2020 |} .\n\ + ex:a ex:knows ex:c .\n\ + << ex:a ex:knows ex:d >> ex:since 1999 .\n\ + ex:e ex:knows ex:a {| ex:since 2021 |} .\n", + ) + .await + .expect("seed"); + let query = "PREFIX ex: \n\ + SELECT ?x ?y ?s WHERE { ?x a ex:P \ + OPTIONAL { ?x ex:knows ?y {| ex:since ?s |} } }"; + for phase in ["novelty", "indexed"] { + if phase == "indexed" { + support::rebuild_and_publish_index(&fluree, ledger_id).await; + } + let ledger = fluree.ledger(ledger_id).await.expect("load"); + let (spans, guard) = support::span_capture::init_test_tracing(); + let rows = support::query_sparql_formatted(&fluree, &ledger, query) + .await + .expect("query"); + drop(guard); + assert!( + !spans + .find_events("optional batched hash-join complete") + .is_empty(), + "[{phase}] the batched lane" + ); + assert_eq!( + support::normalize_rows(&rows), + support::normalize_rows(&json!([ + ["ex:a", "ex:b", 2020], + ["ex:e", "ex:a", 2021], + ["ex:f", null, null] + ])), + "[{phase}]" + ); + } +} + // ===================================================================== // #1467 — reification-aware COPY/MOVE/ADD re-homing (named-source cases) // ===================================================================== diff --git a/fluree-db-query/src/optional.rs b/fluree-db-query/src/optional.rs index d3c6f3898e..8bf41c962e 100644 --- a/fluree-db-query/src/optional.rs +++ b/fluree-db-query/src/optional.rs @@ -1728,6 +1728,8 @@ fn filter_tolerates_unbound(expr: &crate::ir::Expression) -> bool { fn pattern_filters_tolerate_unbound(p: &Pattern) -> bool { match p { Pattern::Filter(expr) => filter_tolerates_unbound(expr), + // An unbound variable is free inside EXISTS, not an error. + Pattern::Exists(_) => true, Pattern::DefaultGraphSource { patterns } => { patterns.iter().any(pattern_filters_tolerate_unbound) } @@ -1755,8 +1757,9 @@ fn pattern_filters_tolerate_unbound(p: &Pattern) -> bool { /// bound-subject shapes stay EXCLUDED pending their own differential evidence. /// /// The edge-annotation expansion (a Cypher relationship binding) yields the -/// annotation's `rdf:reifies` link and its term components: a pure relation -/// between the term and its components, so its per-seed evaluation is a pure +/// annotation's `rdf:reifies` link, its term components and an existence check +/// on the base edge: a pure relation between the term and its components, and +/// a check over the row's own bindings, so its per-seed evaluation is a pure /// restriction by the correlation tuple too. Under a default-graph union the /// chain is wrapped in `Pattern::DefaultGraphSource`, admitted recursively here /// and excluded at the `build_batch` dataset gate. @@ -1768,6 +1771,9 @@ fn inner_pattern_is_hash_join_safe(p: &Pattern) -> bool { // ~25ms/row of replanning that turned a 21k-row UNWIND reindex query // into minutes of CPU. Pattern::TermComponents(_) => true, + // The expansion's existence check on the base edge: like a triple, it + // reads only the row's own bindings. + Pattern::Exists(inner) => inner.iter().all(|p| matches!(p, Pattern::Triple(_))), Pattern::R2rml(rp) => { batched_optional_r2rml_enabled() && (r2rml_leaf_is_hash_join_safe(rp) @@ -2940,6 +2946,20 @@ mod tests { object: Component::Var(VarId(2)), } ))); + let edge = Pattern::Triple(TriplePattern::new( + Ref::Var(VarId(1)), + Ref::Var(VarId(3)), + Term::Var(VarId(2)), + )); + assert!(inner_pattern_is_hash_join_safe(&Pattern::Exists(vec![ + edge.clone() + ]))); + assert!(!inner_pattern_is_hash_join_safe(&Pattern::Exists(vec![ + Pattern::Optional(vec![edge.clone()]) + ]))); + assert!(pattern_filters_tolerate_unbound(&Pattern::Exists(vec![ + edge + ]))); } // PR-4b: the batched-OPTIONAL admission for R2RML inners is NARROW — only a From 44b3ac973cc1bb349600a6f035bc3384f66b51a3 Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 08:17:26 -0400 Subject: [PATCH 80/92] feat(query): triple terms in JSON-LD values cells `{"@id": {"@id": s, p: o}}` in a values cell is a triple term, the twin of SPARQL's `VALUES ?t { <<( s p o )>> }`. The object may be an IRI, a literal or a nested term, built as the SPARQL cell and `triple()` build it. --- docs/query/jsonld-query.md | 2 + fluree-db-api/tests/it_triple_term_links.rs | 36 ++++++++++++++++++ fluree-db-query/src/parse/ast.rs | 7 ++++ fluree-db-query/src/parse/lower.rs | 37 +++++++++++++++++++ fluree-db-query/src/parse/values.rs | 41 +++++++++++++++++++++ 5 files changed, 123 insertions(+) diff --git a/docs/query/jsonld-query.md b/docs/query/jsonld-query.md index 70be364526..7174651f53 100644 --- a/docs/query/jsonld-query.md +++ b/docs/query/jsonld-query.md @@ -1181,6 +1181,8 @@ As in RDF 1.2, `@reifies` matches the reifier's link only, so it also finds reif } ``` +The same shape, with constant components, is a triple term in a `values` cell: `"values": ["?t", [{ "@id": { "@id": "ex:alice", "ex:knows": { "@id": "ex:bob" } } }]]`. + See [Triple terms as values](../concepts/edge-annotations.md#triple-terms-as-values). **Subject expansion output:** diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 984c04d991..f1bfb3710e 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2776,6 +2776,42 @@ async fn assert_term_values( .await .expect("JSON-LD constant term"); assert_eq!(got, json!([]), "[{label}] another term does not match"); + + // A term as a JSON-LD `values` cell, twin of SPARQL's `VALUES ?t { <<( … )>> }`. + let values = |predicate: &str, term: JsonValue| { + json!({ + "@context": {"ex": "http://example.org/"}, + "select": ["?d"], + "values": ["?t", [{"@id": term}]], + "where": {"@id": "?d", predicate: "?t"} + }) + }; + for (predicate, term, expected) in [ + ( + "ex:mentions", + json!({"@id": "ex:s", "ex:p": {"@id": "ex:o"}}), + docs, + ), + ( + "ex:quotes", + json!({"@id": "ex:s", "ex:says": {"@value": "chat", "@language": "fr"}}), + 1, + ), + ( + "ex:mentions", + json!({"@id": "ex:s", "ex:p": {"@id": "ex:other"}}), + 0, + ), + ] { + let got = support::query_jsonld_formatted(fluree, ledger, &values(predicate, term.clone())) + .await + .unwrap_or_else(|e| panic!("[{label}] JSON-LD values term {term}: {e}")); + assert_eq!( + got.as_array().map(Vec::len), + Some(expected), + "[{label}] JSON-LD values term {term}: {got}" + ); + } } /// A triple term is a value under any predicate, not only as a link's diff --git a/fluree-db-query/src/parse/ast.rs b/fluree-db-query/src/parse/ast.rs index 46357df7d5..f4c6001885 100644 --- a/fluree-db-query/src/parse/ast.rs +++ b/fluree-db-query/src/parse/ast.rs @@ -61,6 +61,13 @@ pub enum UnresolvedValue { /// Datatype IRI or language tag constraint dtc: Option, }, + /// A triple term, `{"@id": {"@id": s, p: o}}`: subject and predicate + /// IRIs, and an object that is an IRI, a literal or a nested term. + TripleTerm { + subject: Arc, + predicate: Arc, + object: Box, + }, /// An already-resolved binding, passed through resolution verbatim. /// Never produced by any parser — the Cypher sequential write driver /// seeds row tables (bindings extracted from a prior query result) diff --git a/fluree-db-query/src/parse/lower.rs b/fluree-db-query/src/parse/lower.rs index 5e455b99c4..ae2e2aaf1a 100644 --- a/fluree-db-query/src/parse/lower.rs +++ b/fluree-db-query/src/parse/lower.rs @@ -515,6 +515,43 @@ fn lower_values_cell(cell: &UnresolvedValue, encoder: &E) -> Resu .ok_or_else(|| ParseError::UnknownNamespace(iri.to_string()))?; Ok(Binding::sid(sid)) } + UnresolvedValue::TripleTerm { + subject, + predicate, + object, + } => { + let encode = |iri: &str| { + encoder + .encode_iri(iri) + .ok_or_else(|| ParseError::UnknownNamespace(iri.to_string())) + }; + let (o, dt, lang) = match lower_values_cell(object, encoder)? { + Binding::Sid { sid, .. } => ( + FlakeValue::Ref(sid), + fluree_db_core::edge::id_datatype_sid(), + None, + ), + Binding::Lit { val, dtc, .. } => { + let lang = dtc.lang_tag().map(str::to_string); + (val, dtc.datatype().clone(), lang) + } + _ => { + return Err(ParseError::InvalidWhere( + "a triple term's object in values must be a constant".to_string(), + )) + } + }; + Ok(Binding::lit( + FlakeValue::TripleTerm(Box::new(fluree_db_core::TripleTermValue { + s: encode(subject)?, + p: encode(predicate)?, + o, + dt, + lang, + })), + fluree_db_core::triple_term_datatype_sid().clone(), + )) + } UnresolvedValue::Literal { value, dtc } => { // Build initial FlakeValue from the literal let initial_fv = match value { diff --git a/fluree-db-query/src/parse/values.rs b/fluree-db-query/src/parse/values.rs index a70c078d7d..e7b7c63268 100644 --- a/fluree-db-query/src/parse/values.rs +++ b/fluree-db-query/src/parse/values.rs @@ -178,11 +178,15 @@ fn parse_values_cell(cell: &JsonValue, ctx: &JsonLdParseCtx) -> Result, ctx: &JsonLdParseCtx, ) -> Result { + if let Some(JsonValue::Object(term)) = map.get("@id") { + return parse_triple_term(term, ctx); + } // Handle @id shorthand if let Some(id_val) = map.get("@id") { return parse_iri_binding(id_val, ctx); @@ -192,6 +196,43 @@ fn parse_jsonld_object( parse_typed_literal(map, ctx) } +/// Parse the `{"@id": s, p: o}` inside a triple-term cell: one predicate with +/// one object, which may itself be a triple term. +fn parse_triple_term( + term: &serde_json::Map, + ctx: &JsonLdParseCtx, +) -> Result { + let invalid = || { + ParseError::InvalidWhere( + "a triple term in values is {\"@id\": s, p: o}: an IRI subject and one \ + predicate with one object" + .to_string(), + ) + }; + let subject = term + .get("@id") + .and_then(JsonValue::as_str) + .ok_or_else(invalid)?; + let mut properties = term.iter().filter(|(k, _)| k.as_str() != "@id"); + let (Some((predicate, object)), None) = (properties.next(), properties.next()) else { + return Err(invalid()); + }; + let object = match object { + JsonValue::Array(items) if items.len() == 1 => &items[0], + JsonValue::Array(_) => return Err(invalid()), + other => other, + }; + let object = parse_values_cell(object, ctx)?; + if matches!(object, UnresolvedValue::Unbound) { + return Err(invalid()); + } + Ok(UnresolvedValue::TripleTerm { + subject: Arc::from(ctx.expand_vocab(subject)?.0), + predicate: Arc::from(ctx.expand_vocab(predicate)?.0), + object: Box::new(object), + }) +} + /// Parse IRI binding from `{"@id": "..."}` fn parse_iri_binding(id_val: &JsonValue, ctx: &JsonLdParseCtx) -> Result { let id_str = id_val From 18040ba0d2ab563fabc291bc20f6ac71345459f3 Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 08:37:49 -0400 Subject: [PATCH 81/92] bench: annotation link queries, with an outer side past one seeded chunk The arena's annotation_planner bench went with the arena, leaving no bench on the lanes this branch adds. query_hot_annotation_links covers the base-edge EXISTS below and past one seeded build's 1,024 keys over a much larger predicate (constant and variable predicate), a full reified-pattern count, the batched wildcard join and the COUNT product, and checks each scenario's count once before timing. annotation_varlength_probe's doc now describes the link expansion instead of the retired sidecar drain. --- fluree-db-api/Cargo.toml | 4 + .../benches/annotation_varlength_probe.rs | 39 ++-- .../benches/query_hot_annotation_links.rs | 216 ++++++++++++++++++ regression-budget.json | 5 + 4 files changed, 245 insertions(+), 19 deletions(-) create mode 100644 fluree-db-api/benches/query_hot_annotation_links.rs diff --git a/fluree-db-api/Cargo.toml b/fluree-db-api/Cargo.toml index a1bb97f36c..c9bce6204a 100644 --- a/fluree-db-api/Cargo.toml +++ b/fluree-db-api/Cargo.toml @@ -672,6 +672,10 @@ harness = false name = "query_hot_negation_count" harness = false +[[bench]] +name = "query_hot_annotation_links" +harness = false + [[bench]] name = "query_hot_limit_startup" harness = false diff --git a/fluree-db-api/benches/annotation_varlength_probe.rs b/fluree-db-api/benches/annotation_varlength_probe.rs index 99cfe9af1b..b96f0c7c40 100644 --- a/fluree-db-api/benches/annotation_varlength_probe.rs +++ b/fluree-db-api/benches/annotation_varlength_probe.rs @@ -9,13 +9,14 @@ //! `*1..1` is 1 hop, `*1..3` is 1+2+3 = 6, `*1..5` is 15 — so anything paid //! *per probe operator* is multiplied by that count. //! -//! Each probe plans to an `AnnotationValueOptionalBuilder` -//! (`fluree-db-query/src/optional.rs`), which answers its rows from sidecar -//! maps drained out of the three `f:reifies*` predicates. The drain is -//! O(#annotations in the ledger) and is **independent of the result size**, so -//! it is the term this bench is built to hold down: the scenarios below bind a -//! relationship variable and never read an edge property, which is the shape -//! that pays the drain for nothing. +//! Each probe expands to the reifier's `rdf:reifies` link, its term +//! components and an existence check on the base edge, which the OPTIONAL +//! batched hash-join lane (`PlanTreeOptionalBuilder::build_batch`, +//! `fluree-db-query/src/optional.rs`) evaluates once per batch of driving rows +//! rather than per row. Whatever a probe operator pays up front is +//! **independent of the result size**, so it is the term this bench is built +//! to hold down: the scenarios below bind a relationship variable and never +//! read an edge property. //! //! ## Scenarios //! @@ -24,28 +25,28 @@ //! //! 1. **`unbound_1_3`** — `-[:KNOWS*1..3]->` with no relationship variable. //! Zero probes: the floor, and the control for scenarios 2-4. -//! 2. **`bound_1_1`** — `-[rs:KNOWS*1..1]->`. One probe — one sidecar drain. +//! 2. **`bound_1_1`** — `-[rs:KNOWS*1..1]->`. One probe. //! 3. **`bound_1_3`** — `-[rs:KNOWS*1..3]->`. Six probes. //! 4. **`bound_1_5`** — `-[rs:KNOWS*1..5]->`. Fifteen probes. //! -//! With a per-query sidecar cache, 2-4 all cost one drain and the gap between -//! them is join work only. Without one, 3 and 4 are 6x and 15x scenario 2, and -//! every multiple grows with the ledger's annotation count rather than with -//! anything the query asked for. That divergence is what a regression here -//! means. +//! The gap between 2-4 should be join work only. If a probe starts paying in +//! proportion to the ledger's annotation count (a scan of every link, say), +//! 3 and 4 become 6x and 15x scenario 2 and every multiple grows with the +//! ledger rather than with anything the query asked for. That divergence is +//! what a regression here means. //! //! ## Setup discipline //! //! Mirrors `query_hot_optional.rs`: build once per scale, populate a //! file-backed ledger, full reindex behind the binary columnar index, then //! reuse one `GraphDb` for every `b.iter` call (warm-cache). Indexed matters — -//! the sidecar drain runs as ordinary planned scans, and an unindexed ledger -//! answers them from novelty instead. +//! the link reads take the index lanes, and an unindexed ledger answers them +//! from novelty instead. //! //! ## Matrix //! //! inputs: BenchScale -> n_claims, reified `KNOWS` edges NOT reachable -//! from the anchor, so they inflate the sidecar without +//! from the anchor, so they inflate the link set without //! inflating the result (Tiny=1_000, Small=5_000, Medium=20_000, //! Large=50_000), plus a 5-hop reified anchor chain //! metric: ns/query (criterion default) @@ -83,7 +84,7 @@ fn scale_n_claims(scale: BenchScale) -> usize { } /// Zero probes: the range binds no relationship variable, so the lowering -/// emits plain join chains and never touches the annotation sidecar. +/// emits plain join chains and never reads a link. const Q_UNBOUND_1_3: &str = r#"MATCH (a:Person {name: "Anchor"})-[:KNOWS*1..3]->(b) RETURN b"#; /// One probe. The pre-existing single-hop cost, and the unit scenarios 3 and 4 @@ -120,8 +121,8 @@ fn anchor_chain() -> Vec { } /// One batch of reified `KNOWS` edges over disjoint node pairs — unreachable -/// from the anchor, so they enlarge the `f:reifies*` sidecar (and therefore -/// every drain) without enlarging any scenario's result. +/// from the anchor, so they enlarge the ledger's link set without enlarging +/// any scenario's result. fn claim_batch(start: usize, end: usize) -> JsonValue { let graph: Vec = (start..end) .map(|i| { diff --git a/fluree-db-api/benches/query_hot_annotation_links.rs b/fluree-db-api/benches/query_hot_annotation_links.rs new file mode 100644 index 0000000000..c9ed3a3d85 --- /dev/null +++ b/fluree-db-api/benches/query_hot_annotation_links.rs @@ -0,0 +1,216 @@ +//! Hot-cache latency of annotation queries over RDF 1.2 links. +//! +//! An annotated edge is a reifier with an `rdf:reifies <<( s p o )>>` link. +//! Annotation syntax joins the link and checks the base edge exists; each +//! scenario pins a lane that keeps that cost proportional to the query rather +//! than to the predicate or the ledger: +//! +//! 1. **`annotated_edges_small_outer`** — 200 annotated edges into one hub's +//! posts. The base-edge EXISTS seeds its build from the outer keys +//! (`fluree-db-query/src/semijoin.rs`). +//! 2. **`annotated_edges_past_one_chunk`** — the same over 1,100 edges, past +//! one seeded build's 1,024 keys: seeded a chunk at a time, not built over +//! the whole `ex:likes` predicate. +//! 3. **`annotated_edges_var_predicate`** — scenario 2 with the predicate a +//! variable, where an unseeded build would cover the whole graph. +//! 4. **`reified_pattern_count`** — `COUNT(*)` over every link of one +//! predicate, `<< ?s ex:likes ?o >> ex:at ?at`. +//! 5. **`wildcard_into_posts`** — `?x ?p ?post` into one hub's posts: the +//! batched wildcard-predicate join (`fluree-db-query/src/join/wildcard.rs`). +//! 6. **`independent_count_product`** — `COUNT(*)` over two independent +//! sides, answered as a product (`fluree-db-query/src/join/replay.rs`). +//! +//! ## Matrix +//! +//! inputs: BenchScale → n_likes annotated `ex:likes` edges, 1,300 of +//! them into the two hubs' posts +//! (Tiny=20_000, Small=100_000, Medium=400_000, Large=1_000_000) +//! metric: ns/query (criterion default) +//! +//! ## Running +//! +//! cargo bench -p fluree-db-api --bench query_hot_annotation_links +//! cargo bench -p fluree-db-api --bench query_hot_annotation_links -- --test +//! FLUREE_BENCH_SCALE=medium cargo bench -p fluree-db-api --bench query_hot_annotation_links + +use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion}; +use fluree_bench_support::{ + bench_runtime, current_profile, current_scale, init_tracing_for_bench, next_ledger_alias, + BenchScale, +}; +use fluree_db_api::admin::ReindexOptions; +use fluree_db_api::{CommitOpts, Fluree, FlureeBuilder, IndexConfig, TxnOpts}; +use std::fmt::Write as _; + +/// Annotated edges into each hub's posts, one like per post. +const SMALL_HUB: usize = 200; +const LARGE_HUB: usize = 1_100; + +fn scale_n_likes(scale: BenchScale) -> usize { + match scale { + BenchScale::Tiny => 20_000, + BenchScale::Small => 100_000, + BenchScale::Medium => 400_000, + BenchScale::Large => 1_000_000, + } +} + +/// `(name, query, expected count)`. +const SCENARIOS: [(&str, &str, usize); 6] = [ + ( + "annotated_edges_small_outer", + r"PREFIX ex: +SELECT (COUNT(*) AS ?n) WHERE { ?post ex:author ex:hub0 . ?liker ex:likes ?post {| ex:at ?at |} }", + SMALL_HUB, + ), + ( + "annotated_edges_past_one_chunk", + r"PREFIX ex: +SELECT (COUNT(*) AS ?n) WHERE { ?post ex:author ex:hub1 . ?liker ex:likes ?post {| ex:at ?at |} }", + LARGE_HUB, + ), + ( + "annotated_edges_var_predicate", + r"PREFIX ex: +SELECT (COUNT(*) AS ?n) WHERE { ?post ex:author ex:hub1 . ?liker ?rel ?post {| ex:at ?at |} }", + LARGE_HUB, + ), + ( + "reified_pattern_count", + r"PREFIX ex: +SELECT (COUNT(*) AS ?n) WHERE { << ?s ex:likes ?o >> ex:at ?at }", + 0, // n_likes, checked separately + ), + ( + "wildcard_into_posts", + r"PREFIX ex: +SELECT (COUNT(*) AS ?n) WHERE { ?post ex:author ex:hub1 . ?x ?p ?post }", + LARGE_HUB, + ), + ( + "independent_count_product", + r"PREFIX ex: +SELECT (COUNT(*) AS ?n) WHERE { ?a ex:author ex:hub0 . ?b ex:author ex:hub1 }", + SMALL_HUB * LARGE_HUB, + ), +]; + +/// Hub posts each liked once; the remaining likes spread over a hundred +/// other posts. Every like is annotated. +fn likes_turtle(n_likes: usize) -> String { + let mut ttl = String::with_capacity(n_likes * 120); + ttl.push_str("VERSION \"1.2\"\n@prefix ex: .\n"); + let like = |ttl: &mut String, i: usize, post: &str| { + let _ = writeln!(ttl, "ex:fan{i} ex:likes {post} {{| ex:at {i} |}} ."); + }; + for i in 0..SMALL_HUB { + let _ = writeln!(ttl, "ex:hub0-post{i} ex:author ex:hub0 ."); + like(&mut ttl, i, &format!("ex:hub0-post{i}")); + } + for i in 0..LARGE_HUB { + let _ = writeln!(ttl, "ex:hub1-post{i} ex:author ex:hub1 ."); + like(&mut ttl, SMALL_HUB + i, &format!("ex:hub1-post{i}")); + } + for i in (SMALL_HUB + LARGE_HUB)..n_likes { + like(&mut ttl, i, &format!("ex:other{}", i % 100)); + } + ttl +} + +/// Populated, indexed file-backed Fluree (same discipline as +/// `query_hot_negation_count.rs`). +async fn setup_indexed(n_likes: usize) -> (tempfile::TempDir, Fluree, String) { + let db_dir = tempfile::tempdir().expect("db tmpdir"); + let fluree = FlureeBuilder::file(db_dir.path().to_string_lossy().to_string()) + .build() + .expect("build file-backed Fluree"); + + let alias = next_ledger_alias("query-hot-annotation-links"); + let ledger = fluree.create_ledger(&alias).await.expect("create_ledger"); + + // High thresholds so populate doesn't race background indexing; the + // explicit reindex below builds the index. + let index_config = IndexConfig { + reindex_min_bytes: 5_000_000_000, + reindex_max_bytes: 5_000_000_000, + }; + let _ = fluree + .insert_turtle_with_opts( + ledger, + &likes_turtle(n_likes), + TxnOpts::default(), + CommitOpts::default(), + &index_config, + None, + ) + .await + .expect("populate insert"); + + let _ = fluree + .reindex(&alias, ReindexOptions::default()) + .await + .expect("reindex"); + + (db_dir, fluree, alias) +} + +fn bench_query_hot_annotation_links(c: &mut Criterion) { + init_tracing_for_bench(); + let rt = bench_runtime(); + let scale = current_scale(); + let profile = current_profile(); + let n_likes = scale_n_likes(scale); + + eprintln!( + " [query_hot_annotation_links] scale={} n_likes={n_likes}", + scale.as_str() + ); + + let (_db_dir, fluree, alias) = rt.block_on(setup_indexed(n_likes)); + let snapshot = rt.block_on(async { fluree.graph(&alias).load().await.expect("graph load") }); + + // A lane that got faster by answering wrong is not a win. + for (name, query, expected) in SCENARIOS { + let expected = if expected == 0 { n_likes } else { expected }; + let count = rt.block_on(async { + let result = snapshot + .query() + .sparql(query) + .execute_formatted() + .await + .unwrap_or_else(|e| panic!("{name} execute: {e}")); + result["results"]["bindings"][0]["n"]["value"] + .as_str() + .and_then(|n| n.parse::().ok()) + .unwrap_or_else(|| panic!("{name}: {result}")) + }); + assert_eq!(count as usize, expected, "{name}"); + } + + let mut group = c.benchmark_group("query_hot_annotation_links"); + group.sample_size(profile.sample_size()); + group.sampling_mode(criterion::SamplingMode::Flat); + + for (name, query, _) in SCENARIOS { + group.bench_with_input(BenchmarkId::new(name, scale.as_str()), &n_likes, |b, _| { + b.iter(|| { + rt.block_on(async { + let result = snapshot + .query() + .sparql(query) + .execute() + .await + .unwrap_or_else(|e| panic!("{name} execute: {e}")); + black_box(result); + }); + }); + }); + } + + group.finish(); + drop(snapshot); + drop(fluree); +} + +criterion_group!(benches, bench_query_hot_annotation_links); +criterion_main!(benches); diff --git a/regression-budget.json b/regression-budget.json index e2610d8800..36a8157cd5 100644 --- a/regression-budget.json +++ b/regression-budget.json @@ -83,6 +83,11 @@ "small": 5.0, "medium": 3.0 }, + "query_hot_annotation_links": { + "tiny": 10.0, + "small": 5.0, + "medium": 3.0 + }, "query_hot_limit_startup": { "tiny": 10.0, "small": 5.0, From da463d7a53cfcc68f546ec8653e83b1772756f71 Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 08:37:49 -0400 Subject: [PATCH 82/92] docs: re-asserted edges, lane policy and pre-link refusal; drop EdgeKey bundle refs Re-asserting a deleted edge brings its earlier claims back, since the link names the triple: say so, and how to remove them. The design doc now states how the lanes keep term policy and which reads a pre-link index refuses. Comments citing the removed EdgeKey bundle decoder name fold_slots_into_links instead. --- docs/concepts/edge-annotations.md | 1 + docs/design/edge-annotations.md | 4 +-- docs/transactions/retractions.md | 2 ++ .../src/parse/edge_annotations.rs | 34 ++++++++----------- 4 files changed, 19 insertions(+), 22 deletions(-) diff --git a/docs/concepts/edge-annotations.md b/docs/concepts/edge-annotations.md index 4fd5904f65..8efadd424f 100644 --- a/docs/concepts/edge-annotations.md +++ b/docs/concepts/edge-annotations.md @@ -289,6 +289,7 @@ The two forms differ in visibility: an anonymous annotation is an edge property | Visible in `select: "*"` | No — hidden from wildcard subject expansion | Yes | | Visible in graph crawl | Only via `@annotation` projection | Yes, like any subject | | Retract base edge → link and body removed | Only in LPG mode | Only in LPG mode | +| Re-assert a retracted edge → earlier claims return | Yes, outside LPG mode; remove them through `@reifies` or `<< s p o ~ ?r >>` | Yes, outside LPG mode; delete the reifier's triples | The anonymous-hide rule means a user wildcard query against Alice doesn't suddenly start returning a sea of internal annotation SIDs once you adopt edge metadata. Annotations participate in queries that ask for them and stay out of the way otherwise. diff --git a/docs/design/edge-annotations.md b/docs/design/edge-annotations.md index 7064588050..5358374331 100644 --- a/docs/design/edge-annotations.md +++ b/docs/design/edge-annotations.md @@ -54,7 +54,7 @@ Every annotation pattern reads the link. Reified-triple patterns (`<< s p o ~ ?r Annotation syntax (`s p o ~ ?r {| ... |}`, JSON-LD `@annotation`, Cypher relationship properties) keeps its `Pattern::EdgeAnnotation` through lowering, and `expand_edge_annotation_patterns_for` (`where_plan.rs`) expands it at planning into the body, the link with its term components (`link_patterns`), and the base edge: the annotation syntax asserts its triple, so a reifier of an unasserted triple must not match it. The base edge comes last in the chain, so where estimates tie it is a bound existence probe. The chain is wrapped in `Pattern::DefaultGraphSource` only when the default graph is a union of two or more graphs (`PlanningContext::default_graph_union`). -A reified-triple pattern names its triple without joining it, so visibility is checked on the term: `QueryPolicyEnforcer` lets a flake whose object is a triple term through only when the triple that term names would be visible, recursively for nested terms. A policy hiding `ex:worksFor` therefore hides the links to `ex:worksFor` edges on every route. +A reified-triple pattern names its triple without joining it, so visibility is checked on the term: `QueryPolicyEnforcer` lets a flake whose object is a triple term through only when the triple that term names would be visible, recursively for nested terms. A policy hiding `ex:worksFor` therefore hides the links to `ex:worksFor` edges on every route. The scan and probe lanes do no per-flake filtering, so under a view policy that restricts anything they decline any predicate whose objects may be triple terms: `rdf:reifies`, and any predicate whose observed datatypes include the untagged kind triple terms carry, or are unknown (`cursor_fast_path_for_predicate`). A link is ordinary data: wildcard scans (`?s ?p ?o`) and wildcard hydration return it like any triple, and hydration renders its triple term as an embedded node. An `@annotation` body leaves its reifier's link out, since the body hangs from it, and so do Cypher property maps, where the link is the relationship itself. @@ -66,7 +66,7 @@ Earlier releases stored an annotation as an `f:reifies*` bundle (`f:reifiesSubje Index roots from those releases may also carry an annotation-arena section. Readers skip it, keeping only the arena's two branch CIDs, and the next index build releases the arena's blobs as garbage. -An annotated index built before links has no term dictionary. Link reads on it fail asking for a rebuild (`fluree reindex`) rather than answering without its annotations. An incremental build over it declines once its window holds a triple term — a term dictionary covering the window alone would lift that refusal — and the index build falls back to a full rebuild, which links every annotation in history. +An annotated index built before links has no term dictionary, and its snapshot says so (`LedgerSnapshot::needs_link_reindex`), so a host can schedule the rebuild. Every read of its annotations fails asking for a rebuild (`fluree reindex`, or `Fluree::reindex`) rather than answering without them: link queries, export, and hydration's `@annotation` (`require_link_index`). An incremental build over it declines once its window holds a triple term — a term dictionary covering the window alone would lift that refusal — and the index build falls back to a full rebuild, which links every annotation in history. ## See also diff --git a/docs/transactions/retractions.md b/docs/transactions/retractions.md index b374ecb79f..1d6be0a0d0 100644 --- a/docs/transactions/retractions.md +++ b/docs/transactions/retractions.md @@ -199,6 +199,8 @@ Deletes order and all its items. A transaction retracts the triples it names and nothing else (see [Insert: Edge Annotations](insert.md#edge-annotations)). As RDF 1.2 defines it, deleting a triple does not delete the statements of a reifier that reifies it: the reifier's `rdf:reifies` link and its body outlive the edge. The annotation syntax (`{| |}`, `~`, JSON-LD `@annotation`, Cypher relationships) joins the edge, so it stops matching those claims; the reified-triple form (`<< s p o ~ ?r >>`, `@reifies`) still finds them. +Because the link names the triple, not one assertion of it, re-asserting a deleted edge brings its earlier claims back: an `@annotation` deleted with its edge reappears when the bare edge is inserted again. To remove the claims for good, delete them through `@reifies` or `<< s p o ~ ?r >>` (an anonymous reifier is reachable only that way), or use LPG mode below. + **The annotation form of a *delete* is a base-edge retract.** `DELETE DATA { :alice :knows :bob ~ :claim1 {| … |} }` — and the bare `~ :claim1` tail with no block — expand to include the base triple, because RDF 1.2 annotation syntax both reifies *and asserts* the triple it annotates. The annotation syntax therefore stops matching **every** claim on that edge, not only the one named. An `upsert` that changes an annotated edge's object does the same thing, with no delete written at all. See [Which spelling does what](../concepts/edge-annotations.md#which-spelling-does-what) for the full table. **LPG mode** (`opts.lpgEdgeLifecycle: true`, which Cypher `DELETE` sets) adds the property-graph relationship lifecycle: retracting an edge retracts every link naming it, and a reifier left with no link loses its body. The cascade is graph-aware: named-graph links are retracted in the same named graph as the edge they reify. diff --git a/fluree-db-transact/src/parse/edge_annotations.rs b/fluree-db-transact/src/parse/edge_annotations.rs index 9501a52eb1..791d5c8706 100644 --- a/fluree-db-transact/src/parse/edge_annotations.rs +++ b/fluree-db-transact/src/parse/edge_annotations.rs @@ -90,8 +90,8 @@ pub(crate) enum ReifiedObjectShape { /// `value.get("@type")`. value: Value, /// `@language` payload if explicit. Drives `f:reifiesLang` - /// emission — required so `EdgeKey::from_reifies_facts` decodes - /// to the same `lang` the base flake carries via `flake.m.lang`. + /// emission — required so the link `fold_slots_into_links` builds + /// names the same `lang` the base flake carries via `flake.m.lang`. language: Option, }, /// Object is a triple term: the node naming its triple, @@ -345,8 +345,8 @@ pub(crate) fn classify_reified_object(map: &Map) -> Result Date: Tue, 6 Oct 2026 08:54:40 -0400 Subject: [PATCH 83/92] perf(export): bound annotation memory past a million links Export read every live link into an edge map before writing a line, which grows with the ledger. Up to 1M links (by the index's link counts) it still does, since a hash probe per edge is the fast path. Past that, or with no counts, each batch probes its own edges (a link lookup only for an edge whose term a dictionary holds) and checks each link row's triple, so memory follows the batch. The reifier bookkeeping keeps a hash per reifier instead of its IRI. 200k annotated edges (debug, Turtle to a sink): preloaded 287 MiB / 10.4 s, per-batch 133 MiB / 15.5 s, against 56 MiB / 2.0 s with --raw-reifies. Export also decoded a triple-term handle only through the persisted dictionary, so a term only novelty holds failed the export when written as a value, an unasserted link, or under --raw-reifies. It now falls back to dictionary novelty, as subjects and strings do. --- fluree-db-api/src/export.rs | 144 +++++---- fluree-db-api/src/export_annotations.rs | 396 ++++++++++++++++++------ fluree-db-api/src/export_builder.rs | 13 +- fluree-db-query/src/binary_scan.rs | 2 +- 4 files changed, 388 insertions(+), 167 deletions(-) diff --git a/fluree-db-api/src/export.rs b/fluree-db-api/src/export.rs index f957ca33c1..1ab4580c0b 100644 --- a/fluree-db-api/src/export.rs +++ b/fluree-db-api/src/export.rs @@ -340,6 +340,14 @@ impl<'a> ExportResolver<'a> { self.decode_string_novelty(ot.decode_kind(), o_key) } DecodeKind::IriRef => self.decode_iri_ref_novelty(o_key), + // A provisional handle names a term only novelty holds. + DecodeKind::TripleTermDict => { + fluree_db_core::triple_term::novelty_term_index(o_key) + .zip(self.dict_novelty.filter(|dn| dn.is_initialized())) + .and_then(|(index, dn)| dn.terms.resolve(index)) + .map(|term| FlakeValue::TripleTerm(Box::new(term.clone()))) + .ok_or(e) + } _ => Err(e), } } @@ -448,57 +456,65 @@ impl<'a> AnnotationContext<'a> { fn is_reifies_row(&self, p_id: u32) -> bool { self.reifies_p_ids.contains(&p_id) } +} - /// Whether a link row gives way to the `~ ` marker on its asserted - /// edge. A link whose triple is not asserted has no edge to carry it, - /// so it is written as a row. - fn link_becomes_marker( - &self, - resolver: &ExportResolver<'_>, - s_id: u64, - (o_type, o_key, p_id): (u16, u64, u32), - g_id: GraphId, - ) -> io::Result { - if !self.probe.has_unasserted() { - return Ok(true); - } - let FlakeValue::TripleTerm(term) = resolver.decode_value(o_type, o_key, p_id, g_id)? else { - return Ok(true); - }; - let reifier = resolver.resolve_subject_sid(s_id)?; - Ok(!self.probe.link_is_unasserted(g_id, &reifier, &term)) +/// A batch's annotation reads, row-aligned: each base edge's live reifiers, +/// and for each link row whether it gives way to the `~ ` marker on its +/// asserted edge. A link whose triple is not asserted has no edge to carry +/// it, so it is written as a row. +#[derive(Default)] +struct BatchAnnotations { + reifiers: Vec>, + markers: Vec, +} + +impl BatchAnnotations { + fn reifiers(&self, row: usize) -> &[Sid] { + self.reifiers.get(row).map_or(&[], Vec::as_slice) + } + + fn is_marker(&self, row: usize) -> bool { + self.markers.get(row).copied().unwrap_or(false) } } -/// Live reifiers for every row of `batch`, row-aligned. +/// Read `batch`'s annotations ([`BatchAnnotations`]). /// -/// Returns an empty vec when the export is not emitting annotation syntax; -/// callers treat a missing entry as "no reifiers", so no writer needs a -/// branch on the mode. +/// Returns empty annotations when the export is not emitting annotation +/// syntax; callers treat a missing entry as "no reifiers", so no writer needs +/// a branch on the mode. /// /// This re-decodes each row's subject, predicate and object to build its /// `EdgeKey` — work the row writer then does again. That duplication is /// deliberate path separation: it happens only for ledgers that carry /// annotations, and it keeps the row writers' existing loop untouched for /// every ledger that does not. -async fn batch_reifiers( +async fn batch_annotations( resolver: &ExportResolver<'_>, ann: Option<&AnnotationContext<'_>>, batch: &ColumnBatch, g_id: GraphId, -) -> io::Result>> { +) -> io::Result { let Some(ann) = ann else { - return Ok(Vec::new()); + return Ok(BatchAnnotations::default()); }; + let mut markers = vec![false; batch.row_count]; let mut edges: Vec = Vec::new(); let mut edge_row: Vec = Vec::new(); - for row in 0..batch.row_count { + for (row, marker) in markers.iter_mut().enumerate() { let p_id = batch.p_id.get_or(row, 0); let o_type = batch.o_type.get_or(row, 0); - if ann.is_reifies_row(p_id) || ann.is_link_row(p_id, o_type) { + let o_key = batch.o_key.get(row); + if ann.is_link_row(p_id, o_type) { + *marker = match resolver.decode_value(o_type, o_key, p_id, g_id)? { + FlakeValue::TripleTerm(term) => !ann.probe.link_is_unasserted(g_id, &term).await?, + _ => true, + }; + continue; + } + if ann.is_reifies_row(p_id) { continue; } - let o_key = batch.o_key.get(row); let Some(p) = resolver.resolve_predicate_sid(p_id) else { continue; }; @@ -536,18 +552,15 @@ async fn batch_reifiers( }); edge_row.push(row); } - let per_edge = ann.probe.live_reifiers(g_id, &edges); - let mut out = vec![Vec::new(); batch.row_count]; - for (i, row) in edge_row.into_iter().enumerate() { - out[row] = per_edge[i].clone(); + let per_edge = ann + .probe + .live_reifiers(g_id, &edges, resolver.store, resolver.dict_novelty) + .await?; + let mut reifiers = vec![Vec::new(); batch.row_count]; + for (row, edge_reifiers) in edge_row.into_iter().zip(per_edge) { + reifiers[row] = edge_reifiers; } - Ok(out) -} - -/// Reifiers for one row, or the empty slice. -#[inline] -fn row_reifiers(reifiers: &[Vec], row: usize) -> &[Sid] { - reifiers.get(row).map_or(&[], Vec::as_slice) + Ok(BatchAnnotations { reifiers, markers }) } /// IRI → prefixed name compression for Turtle/TriG (and JSON-LD compact IRIs). @@ -594,11 +607,11 @@ pub async fn export_graph_turtle( // to rather than appended after the stream, so a subject never opens twice // (see `UntranslatedBySubject`). let (untranslated, untranslated_reifiers) = - resolve_untranslated(ann.as_ref(), untranslated, config.g_id).await?; + resolve_untranslated(&resolver, ann.as_ref(), untranslated, config.g_id).await?; let mut untranslated = UntranslatedBySubject::new(store, untranslated); while let Some(batch) = cursor.next_batch()? { - let reifiers = batch_reifiers(&resolver, ann.as_ref(), &batch, config.g_id).await?; + let reifiers = batch_annotations(&resolver, ann.as_ref(), &batch, config.g_id).await?; write_turtle_batch( &resolver, ann.as_ref(), @@ -656,6 +669,7 @@ pub async fn export_graph_turtle( /// reifier in one probe call. Suppressed links are noted in scope, as the /// translated writers note theirs, so the unresolved count stays honest. async fn resolve_untranslated( + resolver: &ExportResolver<'_>, ann: Option<&AnnotationContext<'_>>, rows: Vec, g_id: GraphId, @@ -667,10 +681,12 @@ async fn resolve_untranslated( let mut base: Vec = Vec::with_capacity(rows.len()); for f in rows { if fluree_db_core::is_rdf_reifies(&f.p) { - let unasserted = matches!(&f.o, FlakeValue::TripleTerm(term) - if ann.probe.link_is_unasserted(g_id, &f.s, term)); + let unasserted = match &f.o { + FlakeValue::TripleTerm(term) => ann.probe.link_is_unasserted(g_id, term).await?, + _ => false, + }; if !unasserted { - ann.probe.note_link_sid(f.s.clone()); + ann.probe.note_link_sid(&f.s); continue; } } @@ -680,7 +696,10 @@ async fn resolve_untranslated( base.push(f); } let keys: Vec = base.iter().map(EdgeKey::from_flake).collect(); - let live = ann.probe.live_reifiers(g_id, &keys); + let live = ann + .probe + .live_reifiers(g_id, &keys, resolver.store, resolver.dict_novelty) + .await?; let mut map: HashMap> = HashMap::new(); for (key, reifiers) in keys.into_iter().zip(live) { if !reifiers.is_empty() { @@ -703,7 +722,7 @@ fn untranslated_reifiers_for<'m>(map: &'m HashMap>, flake: &Fl fn write_turtle_batch( resolver: &ExportResolver, ann: Option<&AnnotationContext<'_>>, - reifiers: &[Vec], + reifiers: &BatchAnnotations, untranslated_reifiers: &HashMap>, batch: &ColumnBatch, g_id: GraphId, @@ -721,9 +740,7 @@ fn write_turtle_batch( // Annotation syntax replaces each link with the `~ ` marker // emitted below; a legacy `f:reifies*` bundle is read as its link. if let Some(ann) = ann { - if ann.is_link_row(p_id, o_type) - && ann.link_becomes_marker(resolver, s_id, (o_type, o_key, p_id), g_id)? - { + if ann.is_link_row(p_id, o_type) && reifiers.is_marker(row) { ann.probe.note_link_in_scope(resolver, s_id); continue; } @@ -791,7 +808,7 @@ fn write_turtle_batch( // a random seek per reifier, out of scan order, at the exact moment // the base edge is written. if let Some(ann) = ann { - for reifier in row_reifiers(reifiers, row) { + for reifier in reifiers.reifiers(row) { let Some(iri) = resolver.store.sid_to_iri(reifier) else { continue; }; @@ -905,25 +922,18 @@ pub async fn export_graph_jsonld( let mut current_props: Vec<(String, Vec)> = Vec::new(); let mut first_node = true; let (untranslated, untranslated_reifiers) = - resolve_untranslated(ann.as_ref(), untranslated, config.g_id).await?; + resolve_untranslated(&resolver, ann.as_ref(), untranslated, config.g_id).await?; let mut untranslated = UntranslatedBySubject::new(store, untranslated); while let Some(batch) = cursor.next_batch()? { - let reifiers = batch_reifiers(&resolver, ann.as_ref(), &batch, config.g_id).await?; + let reifiers = batch_annotations(&resolver, ann.as_ref(), &batch, config.g_id).await?; for row in 0..batch.row_count { let s_id = batch.s_id.get(row); let p_id = batch.p_id.get_or(row, 0); let o_type = batch.o_type.get_or(row, 0); let o_key = batch.o_key.get(row); if let Some(ann) = ann.as_ref() { - if ann.is_link_row(p_id, o_type) - && ann.link_becomes_marker( - &resolver, - s_id, - (o_type, o_key, p_id), - config.g_id, - )? - { + if ann.is_link_row(p_id, o_type) && reifiers.is_marker(row) { ann.probe.note_link_in_scope(&resolver, s_id); continue; } @@ -961,7 +971,7 @@ pub async fn export_graph_jsonld( &resolver, ann, &jval, - row_reifiers(&reifiers, row), + reifiers.reifiers(row), prefixes, ), None => vec![jval], @@ -1613,7 +1623,7 @@ pub async fn export_graph_ntriples( let resolver = ExportResolver::new(store, config.dict_novelty, &ephemeral_preds); let ann = AnnotationContext::new(&resolver, config); let (untranslated, untranslated_reifiers) = - resolve_untranslated(ann.as_ref(), untranslated, config.g_id).await?; + resolve_untranslated(&resolver, ann.as_ref(), untranslated, config.g_id).await?; let mut stats = ExportStats::default(); let graph_term = config.graph_iri.as_deref().map(|iri| { @@ -1625,7 +1635,7 @@ pub async fn export_graph_ntriples( }); while let Some(batch) = cursor.next_batch()? { - let reifiers = batch_reifiers(&resolver, ann.as_ref(), &batch, config.g_id).await?; + let reifiers = batch_annotations(&resolver, ann.as_ref(), &batch, config.g_id).await?; write_batch( &resolver, ann.as_ref(), @@ -1664,7 +1674,7 @@ pub async fn export_graph_ntriples( fn write_batch( resolver: &ExportResolver, ann: Option<&AnnotationContext<'_>>, - reifiers: &[Vec], + reifiers: &BatchAnnotations, batch: &ColumnBatch, g_id: GraphId, graph_term: Option<&str>, @@ -1677,9 +1687,7 @@ fn write_batch( let o_type = batch.o_type.get_or(row, 0); let o_key = batch.o_key.get(row); if let Some(ann) = ann { - if ann.is_link_row(p_id, o_type) - && ann.link_becomes_marker(resolver, s_id, (o_type, o_key, p_id), g_id)? - { + if ann.is_link_row(p_id, o_type) && reifiers.is_marker(row) { ann.probe.note_link_in_scope(resolver, s_id); continue; } @@ -1732,7 +1740,7 @@ fn write_batch( // spelling is a triple term as the object of `rdf:reifies`, which // Fluree's Turtle and N-Quads readers both accept. if let Some(ann) = ann { - for reifier in row_reifiers(reifiers, row) { + for reifier in reifiers.reifiers(row) { let Some(r_iri) = resolver.store.sid_to_iri(reifier) else { continue; }; diff --git a/fluree-db-api/src/export_annotations.rs b/fluree-db-api/src/export_annotations.rs index b6cb2106de..de16089e00 100644 --- a/fluree-db-api/src/export_annotations.rs +++ b/fluree-db-api/src/export_annotations.rs @@ -4,40 +4,64 @@ //! (`s p o ~ `), so serializing it needs the edge → reifier direction of //! the `rdf:reifies` links. The scan is in SPOT order, and a reifier whose IRI //! sorts after its base edge's subject arrives too late to mark the line -//! already written, so the links are read once, up front: `O(annotations)`, -//! not `O(dataset)`. +//! already written, so each batch's edges look their links up as the batch +//! is written, and each link row checks whether its triple is asserted. Only +//! the reifier bookkeeping lasts the whole export, at a word per annotation. +use fluree_db_binary_index::BinaryIndexStore; use fluree_db_core::comparator::IndexType; +use fluree_db_core::dict_novelty::DictNovelty; use fluree_db_core::range::{range_with_overlay, RangeMatch, RangeOptions, RangeTest}; use fluree_db_core::{EdgeKey, FlakeValue, GraphId, Sid, TripleTermValue}; use std::collections::{HashMap, HashSet}; +use std::hash::{Hash, Hasher}; use std::io; -use std::marker::PhantomData; -use std::sync::Mutex; +use std::sync::{Arc, Mutex}; use crate::{LedgerState, Result}; -/// Edge → reifier lookup for the duration of one export. +/// Edge → reifier lookups for the duration of one export. /// /// Only `ExportBuilder` constructs one; the type is public because it appears /// on the public `ExportConfig`. An external caller building an `ExportConfig` /// by hand passes `annotations: None` and gets the links as ordinary triples. pub struct AnnotationProbe<'a> { - /// Live links at the export's `t` whose triple is asserted, per graph, - /// by the edge they name (its `g` cleared). Each becomes a marker on - /// its edge. - links: HashMap>>, - /// Live links whose triple is not asserted in their graph: no edge - /// carries their marker, so they are written as rows. - unasserted: HashSet<(GraphId, Sid, EdgeKey)>, - /// Reifiers named by a `~ ` marker somewhere in this export. - named: Mutex>, + ledger: &'a LedgerState, + as_of_t: i64, + /// Every live link, read up front when the index counts few enough; + /// otherwise each batch probes its own edges and links. + preloaded: Option, + /// Reifiers named by a `~ ` marker somewhere in this export, by + /// [`reifier_key`]. + named: Mutex>, /// Reifiers whose link the scan passed — i.e. whose own subject is inside /// the exported selection, so their properties are in the file. Populated /// from the rows the writers suppress, which costs nothing extra: they are /// already being visited and discarded. - in_scope: Mutex>, - _ledger: PhantomData<&'a LedgerState>, + in_scope: Mutex>, +} + +/// Links up to this many are read up front: lookups then cost a hash probe, +/// at a few hundred bytes per link. Past it, per-batch probes keep memory +/// to the batch. +pub(crate) const PRELOAD_MAX_LINKS: u64 = 1_000_000; + +/// The live links at the export's `t`, read once. +struct Preloaded { + /// Links whose triple is asserted, per graph, by the edge they name (its + /// `g` cleared). Each becomes a marker on its edge. + links: HashMap>>, + /// Edges, per graph, that live links name but the graph does not assert: + /// no edge carries their marker, so the links are written as rows. + unasserted: HashSet<(GraphId, EdgeKey)>, +} + +/// A reifier as the bookkeeping sets hold it: a hash, so they cost a word per +/// annotation rather than an IRI. +fn reifier_key(sid: &Sid) -> u64 { + let mut hasher = std::collections::hash_map::DefaultHasher::new(); + sid.hash(&mut hasher); + hasher.finish() } impl<'a> AnnotationProbe<'a> { @@ -45,9 +69,7 @@ impl<'a> AnnotationProbe<'a> { /// emitted for `reifier`. pub(crate) fn note_reifier_named(&self, reifier: &Sid) { if let Ok(mut named) = self.named.lock() { - if !named.contains(reifier) { - named.insert(reifier.clone()); - } + named.insert(reifier_key(reifier)); } } @@ -57,7 +79,7 @@ impl<'a> AnnotationProbe<'a> { let Ok(sid) = resolver.reifier_sid(s_id) else { return; }; - self.note_link_sid(sid); + self.note_link_sid(&sid); } /// As [`Self::note_link_in_scope`], for callers that already hold the @@ -67,9 +89,9 @@ impl<'a> AnnotationProbe<'a> { /// no resolver round-trip. They still have to be *counted*: suppressing a /// link without noting it turns a visible leak into an annotation that /// vanishes with no marker, no link and no number. - pub(crate) fn note_link_sid(&self, sid: Sid) { + pub(crate) fn note_link_sid(&self, sid: &Sid) { if let Ok(mut in_scope) = self.in_scope.lock() { - in_scope.insert(sid); + in_scope.insert(reifier_key(sid)); } } @@ -96,37 +118,66 @@ impl<'a> AnnotationProbe<'a> { in_scope.difference(&named).count() as u64 } - /// Read `ledger`'s live links as of `as_of_t`, or establish that it has - /// none. `Ok(None)` when neither the index nor novelty has ever held an - /// annotation, so a ledger without annotations pays two boolean reads. - /// Each link's triple is looked up once per `(graph, subject, - /// predicate)` to tell markers from rows. - pub(crate) async fn for_ledger(ledger: &'a LedgerState, as_of_t: i64) -> Result> { + /// A probe of `ledger`'s links as of `as_of_t`, or `Ok(None)` when + /// neither the index nor novelty has ever held an annotation, so a ledger + /// without annotations pays two boolean reads. Nothing is read up front: + /// the writers probe each batch's edges and links. + pub(crate) async fn for_ledger( + ledger: &'a LedgerState, + as_of_t: i64, + preload_max_links: u64, + ) -> Result> { if !ledger.snapshot.has_annotations && !ledger.novelty.has_annotations() { return Ok(None); } fluree_db_query::term_components::require_link_index(&ledger.snapshot)?; + let mut probe = Self { + ledger, + as_of_t, + preloaded: None, + named: Mutex::new(HashSet::new()), + in_scope: Mutex::new(HashSet::new()), + }; + // Unknown counts on an index mean an index of unknown size; without + // one, the links are novelty's. + let indexed_links = match ledger.snapshot.stats.as_ref() { + Some(stats) => stats + .links + .as_ref() + .map(|links| links.iter().map(|l| l.count).sum::()), + None => Some(0), + }; + if indexed_links.is_some_and(|n| n <= preload_max_links) { + probe.preloaded = Some( + probe + .preload() + .await + .map_err(|e| crate::ApiError::internal(e.to_string()))?, + ); + } + Ok(Some(probe)) + } + /// Read every live link, telling markers from rows by one lookup per + /// `(graph, subject, predicate)`. + async fn preload(&self) -> io::Result { let graphs = std::iter::once(0).chain( - ledger + self.ledger .snapshot .graph_registry .iter_entries() .map(|(g_id, _)| g_id), ); let mut links: HashMap>> = HashMap::new(); - let mut unasserted: HashSet<(GraphId, Sid, EdgeKey)> = HashSet::new(); + let mut unasserted: HashSet<(GraphId, EdgeKey)> = HashSet::new(); for g_id in graphs { - let flakes = range_with_overlay( - &ledger.snapshot, - g_id, - ledger.novelty.as_ref(), - IndexType::Psot, - RangeTest::Eq, - RangeMatch::predicate(fluree_db_core::rdf_reifies_sid().clone()), - RangeOptions::new().with_to_t(as_of_t), - ) - .await?; + let flakes = self + .range( + g_id, + IndexType::Psot, + RangeMatch::predicate(fluree_db_core::rdf_reifies_sid().clone()), + ) + .await?; // Moved, not cloned: the map holds each term's parts. let graph_links = links.entry(g_id).or_default(); for flake in flakes { @@ -155,23 +206,20 @@ impl<'a> AnnotationProbe<'a> { edges.sort_unstable_by(|a, b| (&a.s, &a.p).cmp(&(&b.s, &b.p))); let mut missing: Vec = Vec::new(); for group in edges.chunk_by(|a, b| a.s == b.s && a.p == b.p) { - let asserted: HashSet = range_with_overlay( - &ledger.snapshot, - g_id, - ledger.novelty.as_ref(), - IndexType::Spot, - RangeTest::Eq, - RangeMatch::subject_predicate(group[0].s.clone(), group[0].p.clone()), - RangeOptions::new().with_to_t(as_of_t), - ) - .await? - .iter() - .map(|flake| EdgeKey { - g: None, - list_i: None, - ..EdgeKey::from_flake(flake) - }) - .collect(); + let asserted: HashSet = self + .range( + g_id, + IndexType::Spot, + RangeMatch::subject_predicate(group[0].s.clone(), group[0].p.clone()), + ) + .await? + .iter() + .map(|flake| EdgeKey { + g: None, + list_i: None, + ..EdgeKey::from_flake(flake) + }) + .collect(); missing.extend( group .iter() @@ -180,59 +228,140 @@ impl<'a> AnnotationProbe<'a> { ); } for edge in missing { - for reifier in graph_links.remove(&edge).unwrap_or_default() { - unasserted.insert((g_id, reifier, edge.clone())); - } + graph_links.remove(&edge); + unasserted.insert((g_id, edge)); } } for reifiers in links.values_mut().flat_map(HashMap::values_mut) { reifiers.sort(); } - Ok(Some(Self { - links, - unasserted, - named: Mutex::new(HashSet::new()), - in_scope: Mutex::new(HashSet::new()), - _ledger: PhantomData, - })) + Ok(Preloaded { links, unasserted }) } - /// Whether any live link names a triple its graph does not assert. - pub(crate) fn has_unasserted(&self) -> bool { - !self.unasserted.is_empty() + async fn range( + &self, + g_id: GraphId, + index: IndexType, + rm: RangeMatch, + ) -> io::Result> { + range_with_overlay( + &self.ledger.snapshot, + g_id, + self.ledger.novelty.as_ref(), + index, + RangeTest::Eq, + rm, + RangeOptions::new().with_to_t(self.as_of_t), + ) + .await + .map_err(|e| io::Error::other(e.to_string())) } - /// Whether `reifier`'s link to `term` in graph `g_id` names a triple the - /// graph does not assert, so the link is written as a row rather than - /// replaced by a marker. - pub(crate) fn link_is_unasserted( + /// Whether graph `g_id` does not assert the triple `term` names, so a + /// link to it is written as a row rather than replaced by a marker. + pub(crate) async fn link_is_unasserted( &self, g_id: GraphId, - reifier: &Sid, - term: &fluree_db_core::TripleTermValue, - ) -> bool { - !self.unasserted.is_empty() - && self - .unasserted - .contains(&(g_id, reifier.clone(), term_edge(term))) + term: &TripleTermValue, + ) -> io::Result { + let edge = term_edge(term); + if let Some(preloaded) = &self.preloaded { + return Ok(preloaded.unasserted.contains(&(g_id, edge))); + } + let rows = self + .range( + g_id, + IndexType::Spot, + RangeMatch::subject_predicate(term.s.clone(), term.p.clone()), + ) + .await?; + Ok(!rows.iter().any(|flake| { + EdgeKey { + g: None, + list_i: None, + ..EdgeKey::from_flake(flake) + } == edge + })) } /// Live reifiers for each edge of graph `g_id`, index-aligned with - /// `edges`; entry `i` is empty when `edges[i]` carries no annotation. - pub(crate) fn live_reifiers(&self, g_id: GraphId, edges: &[EdgeKey]) -> Vec> { - let Some(links) = self.links.get(&g_id) else { - return vec![Vec::new(); edges.len()]; - }; - edges - .iter() - .map(|edge| { - let key = EdgeKey { - g: None, - ..edge.clone() - }; - links.get(&key).cloned().unwrap_or_default() - }) - .collect() + /// `edges` and sorted; entry `i` is empty when `edges[i]` carries no + /// annotation. An edge no term dictionary names has no link, so only an + /// annotated edge pays a link lookup. + pub(crate) async fn live_reifiers( + &self, + g_id: GraphId, + edges: &[EdgeKey], + store: &BinaryIndexStore, + dict_novelty: Option<&Arc>, + ) -> io::Result>> { + if let Some(preloaded) = &self.preloaded { + let links = preloaded.links.get(&g_id); + return Ok(edges + .iter() + .map(|edge| { + let key = EdgeKey { + g: None, + ..edge.clone() + }; + links + .and_then(|links| links.get(&key)) + .cloned() + .unwrap_or_default() + }) + .collect()); + } + let mut out = Vec::with_capacity(edges.len()); + for edge in edges { + let term = TripleTermValue { + s: edge.s.clone(), + p: edge.p.clone(), + o: edge.o.clone(), + dt: edge.dt.clone(), + lang: edge.lang.clone(), + }; + if !may_be_linked(&term, store, dict_novelty)? { + out.push(Vec::new()); + continue; + } + let links = self + .range( + g_id, + IndexType::Post, + RangeMatch::predicate_object( + fluree_db_core::rdf_reifies_sid().clone(), + FlakeValue::TripleTerm(Box::new(term)), + ), + ) + .await?; + let mut reifiers: Vec = links.into_iter().map(|flake| flake.s).collect(); + reifiers.sort(); + reifiers.dedup(); + out.push(reifiers); + } + Ok(out) + } +} + +/// Whether any link can name `term`: a dictionary holds it. A term neither +/// the index nor dictionary novelty names has never been written, so its edge +/// has no link; without initialized dictionary novelty, assume it may. +fn may_be_linked( + term: &TripleTermValue, + store: &BinaryIndexStore, + dict_novelty: Option<&Arc>, +) -> io::Result { + match fluree_db_query::binary_scan::compose_term_handle(term, store, dict_novelty) { + Ok(_) => Ok(true), + Err(e) + if matches!( + e.kind(), + io::ErrorKind::NotFound | io::ErrorKind::Unsupported + ) => + { + Ok(dict_novelty.is_none_or(|dn| !dn.is_initialized() || dn.terms.find(term).is_some())) + } + Err(e) => Err(e), } } @@ -256,3 +385,78 @@ fn term_edge(term: &fluree_db_core::TripleTermValue) -> EdgeKey { pub(crate) trait ReifierSubject { fn reifier_sid(&self, s_id: u64) -> io::Result; } + +#[cfg(test)] +mod tests { + use crate::export::ExportFormat; + use crate::{FlureeBuilder, ReindexOptions}; + + /// Per-batch probes write exactly what the up-front read writes, in every + /// format, over indexed links and novelty ones; a term only novelty holds + /// decodes through dictionary novelty. + #[tokio::test] + async fn per_batch_probes_match_the_preloaded_links() { + let fluree = FlureeBuilder::memory().build_memory(); + let id = "export/probe-modes:main"; + let ledger = fluree.create_ledger(id).await.unwrap(); + let ledger = fluree + .upsert_turtle( + ledger, + "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:alice ex:worksFor ex:acme {| ex:role \"Engineer\" |} .\n\ + ex:bob ex:knows ex:carol ~ ex:claim1 {| ex:confidence 0.9 |} .\n\ + ex:bob ex:knows ex:carol ~ ex:claim2 {| ex:confidence 0.5 |} .\n\ + ex:bob ex:says \"chat\"@fr ~ ex:claim3 {| ex:src ex:hr |} .\n\ + << ex:dave ex:knows ex:erin ~ ex:claim4 >> ex:confidence 0.1 .\n", + ) + .await + .unwrap() + .ledger; + drop(ledger); + fluree.reindex(id, ReindexOptions::default()).await.unwrap(); + let ledger = fluree.ledger(id).await.unwrap(); + fluree + .upsert_turtle( + ledger, + "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:frank ex:knows ex:alice ~ ex:claim5 {| ex:confidence 0.7 |} .\n\ + << ex:gina ex:knows ex:hal ~ ex:claim6 >> ex:confidence 0.2 .\n\ + ex:doc ex:mentions <<( ex:ivy ex:knows ex:jo )>> .\n", + ) + .await + .unwrap(); + + for format in [ + ExportFormat::Turtle, + ExportFormat::JsonLd, + ExportFormat::NTriples, + ] { + let mut outputs = Vec::new(); + for limit in [u64::MAX, 0] { + let mut out = Vec::new(); + fluree + .export(id) + .format(format) + .preload_max_links(limit) + .write_to(&mut out) + .await + .unwrap(); + outputs.push(String::from_utf8(out).unwrap()); + } + assert_eq!(outputs[0], outputs[1], "{format:?}"); + assert!( + outputs[0].contains("ivy"), + "{format:?} writes a novelty term value" + ); + if matches!(format, ExportFormat::Turtle) { + for marker in ["claim1", "claim2", "claim3", "claim5"] { + assert!( + outputs[0].contains(&format!("~ ")), + "{format:?} marks {marker}:\n{}", + outputs[0] + ); + } + } + } + } +} diff --git a/fluree-db-api/src/export_builder.rs b/fluree-db-api/src/export_builder.rs index 4886562706..b557de4648 100644 --- a/fluree-db-api/src/export_builder.rs +++ b/fluree-db-api/src/export_builder.rs @@ -31,6 +31,7 @@ pub struct ExportBuilder<'a> { graph_iri: Option, context_override: Option, time_spec: Option, + preload_max_links: u64, } impl<'a> ExportBuilder<'a> { @@ -45,9 +46,17 @@ impl<'a> ExportBuilder<'a> { graph_iri: None, context_override: None, time_spec: None, + preload_max_links: crate::export_annotations::PRELOAD_MAX_LINKS, } } + /// Read links up front only when the index counts at most `n`. + #[cfg(test)] + pub(crate) fn preload_max_links(mut self, n: u64) -> Self { + self.preload_max_links = n; + self + } + /// Set the output format (default: `Turtle`). pub fn format(mut self, format: ExportFormat) -> Self { self.format = format; @@ -260,13 +269,13 @@ impl<'a> ExportBuilder<'a> { let overlay: &dyn fluree_db_core::OverlayProvider = ledger.novelty.as_ref(); let dict_novelty = &ledger.dict_novelty; - // Edge → reifier lookup, read once for the whole export. `None` on a + // Edge → reifier lookup for the whole export. `None` on a // ledger that has never carried an annotation — and on // `raw_reifies()`, which writes the links as triples. let annotations = if self.raw_reifies && !matches!(self.format, ExportFormat::JsonLd) { None } else { - AnnotationProbe::for_ledger(&ledger, to_t).await? + AnnotationProbe::for_ledger(&ledger, to_t, self.preload_max_links).await? }; // `EdgeKey.g` for a graph being scanned. Computed per graph rather // than per row, and not at all when nothing will probe it. diff --git a/fluree-db-query/src/binary_scan.rs b/fluree-db-query/src/binary_scan.rs index ed43663521..9da06e1299 100644 --- a/fluree-db-query/src/binary_scan.rs +++ b/fluree-db-query/src/binary_scan.rs @@ -4616,7 +4616,7 @@ pub(crate) fn term_object_key( /// provisional handle dictionary novelty gives a term the index has not /// interned. A term neither holds is `NotFound`: no link names it, so the /// pattern cannot match. -pub(crate) fn compose_term_handle( +pub fn compose_term_handle( term: &fluree_db_core::TripleTermValue, store: &BinaryIndexStore, dict_novelty: Option<&Arc>, From 1b24482a2f0969b3864bdd2ae3dbdc4e6468388d Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 08:59:33 -0400 Subject: [PATCH 84/92] fix(query): without estimates, seed only an outer side one chunk holds With no planner estimates the semijoin seeded chunk after chunk however large the outer side, rebuilding its partial-key lookup per chunk where it used to build the body once. It now seeds only when the first chunk holds the whole outer side and otherwise builds unseeded once, as before; with estimates, chunking is unchanged. optional_exists_reuses_partial_keys pins the lookup to once per seeded chunk at most, never per row. --- fluree-db-api/tests/it_query_negation.rs | 10 ++++++---- fluree-db-query/src/semijoin.rs | 18 +++++++++++------- 2 files changed, 17 insertions(+), 11 deletions(-) diff --git a/fluree-db-api/tests/it_query_negation.rs b/fluree-db-api/tests/it_query_negation.rs index 45780fc1a9..0e56f05d1b 100644 --- a/fluree-db-api/tests/it_query_negation.rs +++ b/fluree-db-api/tests/it_query_negation.rs @@ -1233,11 +1233,13 @@ async fn optional_exists_reuses_partial_keys_across_batches() { normalize_rows(&result.to_jsonld(&view.snapshot).unwrap()), expected ); + // One lookup per seeded chunk of the 1,500 outer keys, or one for an + // unseeded build: reused across rows either way. let builds = spans.find_events("semijoin partial-key lookup built"); - assert_eq!( - builds.len(), - 1, - "indexed={indexed}: expected one reused lookup" + assert!( + (1..=2).contains(&builds.len()), + "indexed={indexed}: expected a reused lookup, got {}", + builds.len() ); let probes = spans.find_events("semijoin partial-key probes"); let projected: usize = probes diff --git a/fluree-db-query/src/semijoin.rs b/fluree-db-query/src/semijoin.rs index 1c13c3d2b7..984ac76fd9 100644 --- a/fluree-db-query/src/semijoin.rs +++ b/fluree-db-query/src/semijoin.rs @@ -269,7 +269,7 @@ impl SemijoinOperator { /// Seed only a conjunction of triples (a seeded solution of it is a /// solution of the unseeded body), and only an outer side estimated small - /// against the body. Without estimates, seed: memory stays per chunk. + /// against the body. Without estimates, try: the first chunk decides. fn seeds(&self) -> bool { self.partial_keys_safe && self @@ -278,10 +278,13 @@ impl SemijoinOperator { } /// Seeded keys have passed the point where an unseeded build is cheaper: - /// the outer estimate was low. + /// the outer estimate was low. Without estimates, seed only an outer side + /// one chunk holds. fn seeding_outgrown(&self) -> bool { - self.estimates - .is_some_and(|(_, inner)| self.seeded_keys as f64 * SEEDED_KEY_COST >= inner) + match self.estimates { + Some((_, inner)) => self.seeded_keys as f64 * SEEDED_KEY_COST >= inner, + None => !self.child_exhausted, + } } /// The current chunk's rows first; once it drains, the next chunk under a @@ -308,7 +311,7 @@ impl SemijoinOperator { let mut seen: FxHashSet = FxHashSet::default(); let mut seed_rows: Vec> = Vec::new(); let mut rows = 0usize; - while seen.len() < SEEDED_BUILD_MAX_KEYS && rows < SEEDED_BUILD_MAX_ROWS { + while seen.len() <= SEEDED_BUILD_MAX_KEYS && rows <= SEEDED_BUILD_MAX_ROWS { let Some(batch) = self.child.next_batch(ctx).await? else { self.child_exhausted = true; break; @@ -818,7 +821,8 @@ mod tests { // (outer rows, body, estimates, seeded once open, seeded keys at the end) for (n, body, estimates, seeded, seeded_keys) in [ (keys, triple(), None, true, keys), - (3 * keys, triple(), None, true, 3 * keys), + // Without estimates, an outer side past one chunk builds unseeded. + (3 * keys, triple(), None, false, keys + 256), (3 * keys, triple(), few, true, 3 * keys), (3 * keys, triple(), Some((1e6, 1e6)), false, 0), // Seeding stops once its keys reach 1/16 of the body's estimate. @@ -827,7 +831,7 @@ mod tests { triple(), Some((10.0, 32.0 * keys as f64)), true, - 2 * keys, + 2 * (keys + 256), ), (2, values, None, false, 0), ] { From 8fa018aa891025e6658698c5057d0180cdb2a27e Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 09:03:41 -0400 Subject: [PATCH 85/92] fix(policy): check a triple term before the schema exemption A flake under a schema predicate (rdfs:range, rdfs:domain, ...) skipped the policy check, the triple-term check with it, so `?d rdfs:range ?t` returned a term naming a hidden triple, and OBJECT(?t) its value. The term is now checked first in both the batch and single-flake filters. --- fluree-db-api/tests/it_edge_annotations.rs | 13 ++++++++++-- fluree-db-query/src/policy/enforcer.rs | 23 ++++++++++++---------- 2 files changed, 24 insertions(+), 12 deletions(-) diff --git a/fluree-db-api/tests/it_edge_annotations.rs b/fluree-db-api/tests/it_edge_annotations.rs index 6e9fdb66c9..3cc144594e 100644 --- a/fluree-db-api/tests/it_edge_annotations.rs +++ b/fluree-db-api/tests/it_edge_annotations.rs @@ -4959,7 +4959,7 @@ async fn policy_hiding_base_edge_blocks_annotation_rooted_query() { /// A link `r rdf:reifies <<( s p o )>>` names its triple, so a policy hiding /// the triple hides the link on every route to it, not only `@reifies`, and so -/// does a triple term held under any other predicate. Checked from novelty, from +/// does a triple term held under any other predicate, a schema one included. Checked from novelty, from /// an index (where the scan and probe lanes run), and from novelty over one. #[tokio::test] async fn policy_hiding_base_edge_hides_its_link() { @@ -4978,7 +4978,9 @@ async fn policy_hiding_base_edge_hides_its_link() { }, { "@id": "ex:doc", - "ex:mentions": {"@id": {"@id": "ex:bob", "ex:worksFor": {"@id": "ex:initech"}}} + "ex:mentions": {"@id": {"@id": "ex:bob", "ex:worksFor": {"@id": "ex:initech"}}}, + "http://www.w3.org/2000/01/rdf-schema#range": + {"@id": {"@id": "ex:dan", "ex:worksFor": {"@id": "ex:umbrella"}}} } ] }); @@ -5045,6 +5047,13 @@ async fn policy_hiding_base_edge_hides_its_link() { format!("SELECT ?s ?o WHERE {{ ?d {mentions_iri} <<( ?s {works_for} ?o )>> }}"), mentions, ), + // A schema predicate's exemption does not cover the term. + ( + "SELECT ?s WHERE { ?d ?t \ + BIND(SUBJECT(?t) AS ?s) }" + .to_string(), + 1, + ), ] { let count = |db: fluree_db_api::GraphDb| { let sparql = sparql.clone(); diff --git a/fluree-db-query/src/policy/enforcer.rs b/fluree-db-query/src/policy/enforcer.rs index 7b31871776..3c2d85294b 100644 --- a/fluree-db-query/src/policy/enforcer.rs +++ b/fluree-db-query/src/policy/enforcer.rs @@ -136,17 +136,19 @@ impl QueryPolicyEnforcer { let mut result = Vec::with_capacity(flakes.len()); for flake in flakes { - // Schema flakes always allowed - if is_schema_flake(&flake.p, &flake.o) { - result.push(flake); - continue; - } + // A term names its triple even under a schema predicate, so it is + // checked before the schema exemption. if !self .term_visible(g_id, to_t, &flake.o, &executor, tracker) .await? { continue; } + // Schema flakes always allowed + if is_schema_flake(&flake.p, &flake.o) { + result.push(flake); + continue; + } // Get subject classes from cache let subject_classes = self @@ -198,15 +200,11 @@ impl QueryPolicyEnforcer { return Ok(true); } - // Schema flakes always allowed - if is_schema_flake(&flake.p, &flake.o) { - return Ok(true); - } - // Create executor using the GRAPH's snapshot/overlay/to_t let executor = QueryPolicyExecutor::with_overlay(snapshot, overlay, to_t); self.cache_term_subject_classes(snapshot, g_id, overlay, to_t, std::slice::from_ref(flake)) .await?; + // Before the schema exemption: a term names its triple under any predicate. if !self .term_visible(g_id, to_t, &flake.o, &executor, tracker) .await? @@ -214,6 +212,11 @@ impl QueryPolicyEnforcer { return Ok(false); } + // Schema flakes always allowed + if is_schema_flake(&flake.p, &flake.o) { + return Ok(true); + } + // Get subject classes from cache let subject_classes = self .policy From 71a2266b9a488460490a60d8c3fbaf441ff07a2c Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 09:07:44 -0400 Subject: [PATCH 86/92] fix(query): a triple term is not a literal isLITERAL classed every non-resource as a literal, so it returned true for a triple term, and SPARQL DATATYPE returned f:tripleTerm from novelty and @id from an index. RDF 1.2 makes a triple term its own kind: isLITERAL is now false, SPARQL DATATYPE is a type error, and JSON-LD's datatype names f:tripleTerm on both paths, as it names @id for an IRI. The indexed literal count no longer counts TRIPLE_TERM rows. --- fluree-db-api/tests/it_triple_term_links.rs | 68 +++++++++++++++++++++ fluree-db-query/src/eval/rdf.rs | 18 ++++++ fluree-db-query/src/eval/types.rs | 15 ++++- fluree-db-query/src/fast_count.rs | 3 +- 4 files changed, 100 insertions(+), 4 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index f1bfb3710e..9b1a655097 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2748,6 +2748,23 @@ async fn assert_term_values( assert_eq!(got.len(), docs, "[{label}] a built term joins: {got:?}"); let got = run("SELECT ?r WHERE { ?r rdf:reifies ?t }").await; assert!(got.is_empty(), "[{label}] a value is not a link: {got:?}"); + // A triple term is its own kind of term: no literal, IRI or blank node, + // and DATATYPE is a type error. Stored and built alike. + for source in [ + "ex:doc ex:mentions ?t", + "BIND(TRIPLE(ex:s, ex:p, ex:o) AS ?t)", + ] { + let got = run(&format!( + "SELECT ?lit ?iri ?blank ?dt WHERE {{ {source} BIND(isLITERAL(?t) AS ?lit) \ + BIND(isIRI(?t) AS ?iri) BIND(isBLANK(?t) AS ?blank) BIND(DATATYPE(?t) AS ?dt) }}" + )) + .await; + assert_eq!( + got, + strings(&[&["false", "false", "false", "null"]]), + "[{label}] {source}" + ); + } let jsonld = |term: JsonValue| { json!({ @@ -2777,6 +2794,31 @@ async fn assert_term_values( .expect("JSON-LD constant term"); assert_eq!(got, json!([]), "[{label}] another term does not match"); + // JSON-LD's twin: not a literal, and its datatype names the term kind. + let got = support::query_jsonld_formatted( + fluree, + ledger, + &json!({ + "@context": {"ex": "http://example.org/"}, + "select": ["?lit", "?dt"], + "where": [ + {"@id": "ex:doc", "ex:mentions": "?t"}, + ["bind", "?lit", "(isLiteral ?t)", "?dt", "(datatype ?t)"] + ] + }), + ) + .await + .unwrap_or_else(|e| panic!("[{label}] JSON-LD term kind: {e}")); + assert_eq!( + got[0][0], + json!(false), + "[{label}] JSON-LD isLiteral: {got}" + ); + assert!( + got[0][1].to_string().contains("tripleTerm"), + "[{label}] JSON-LD datatype: {got}" + ); + // A term as a JSON-LD `values` cell, twin of SPARQL's `VALUES ?t { <<( … )>> }`. let values = |predicate: &str, term: JsonValue| { json!({ @@ -2814,6 +2856,32 @@ async fn assert_term_values( } } +/// A triple term is not a literal to the indexed literal count either. +#[tokio::test] +async fn literal_count_leaves_triple_terms_out() { + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/triple-term-links:literal-count"; + fluree + .upsert_turtle( + support::genesis_ledger(&fluree, ledger_id), + "@prefix ex: .\n\ + ex:a ex:name \"x\" .\n\ + ex:a ex:knows ex:b .\n\ + ex:doc ex:mentions <<( ex:s ex:p ex:o )>> .\n", + ) + .await + .expect("seed"); + support::rebuild_and_publish_index(&fluree, ledger_id).await; + let ledger = fluree.ledger(ledger_id).await.expect("load"); + let got = run_link_query( + &fluree, + &ledger, + "SELECT (COUNT(?o) AS ?n) WHERE { ?s ?p ?o FILTER(isLITERAL(?o)) }".to_string(), + ) + .await; + assert_eq!(got, strings(&[&["1"]])); +} + /// A triple term is a value under any predicate, not only as a link's /// object: it reads back, decomposes and matches as a constant from novelty, /// a full rebuild and an incremental build, and is never read as a link. diff --git a/fluree-db-query/src/eval/rdf.rs b/fluree-db-query/src/eval/rdf.rs index 0eb285daa5..0238557265 100644 --- a/fluree-db-query/src/eval/rdf.rs +++ b/fluree-db-query/src/eval/rdf.rs @@ -101,6 +101,13 @@ fn datatype_of_binding( strict: bool, ) -> Result> { match binding { + Binding::Lit { + val: fluree_db_core::FlakeValue::TripleTerm(_), + .. + } => Ok(triple_term_datatype(strict)), + Binding::EncodedLit { o_kind, .. } if *o_kind == ObjKind::TRIPLE_TERM.as_u8() => { + Ok(triple_term_datatype(strict)) + } Binding::Lit { dtc, .. } => Ok(Some(ComparableValue::Sid(dtc.datatype().clone()))), // A NUM_BIG `dt_id` reads DECIMAL for overflow integers too; only the // decoded value names the datatype (issue #1329). @@ -182,6 +189,13 @@ fn datatype_of_binding( } } +/// A triple term is neither literal nor IRI: SPARQL's DATATYPE is a type +/// error; the JSON-LD surface names the term's kind, as it names `@id` for +/// an IRI. +fn triple_term_datatype(strict: bool) -> Option { + (!strict).then(|| ComparableValue::Sid(fluree_db_core::triple_term_datatype_sid().clone())) +} + /// Datatype of a computed value (the `DATATYPE()` path): the datatype /// the value would carry if bound — matching storage/arithmetic tagging /// (a plain integer result is `xsd:integer`, RDF 1.1). @@ -205,6 +219,10 @@ fn datatype_of_comparable( ComparableValue::Time(_) => dts.xsd_time.clone(), ComparableValue::Vector(_) => dts.fluree_vector.clone(), ComparableValue::GeoPoint(_) => dts.geo_wkt_literal.clone(), + ComparableValue::TypedLiteral { + val: fluree_db_core::FlakeValue::TripleTerm(_), + .. + } => return Ok(triple_term_datatype(strict)), ComparableValue::TypedLiteral { dtc, .. } => match dtc { Some(UnresolvedDatatypeConstraint::LangTag(_)) => dts.rdf_lang_string.clone(), Some(UnresolvedDatatypeConstraint::Explicit(iri)) => { diff --git a/fluree-db-query/src/eval/types.rs b/fluree-db-query/src/eval/types.rs index 602fa77026..959c27a01c 100644 --- a/fluree-db-query/src/eval/types.rs +++ b/fluree-db-query/src/eval/types.rs @@ -52,9 +52,18 @@ pub fn eval_is_literal( check_arity(args, 1, "isLiteral")?; let val = args[0].eval_to_comparable(row, ctx)?; Ok(Some(ComparableValue::Bool(val.is_some_and(|v| { - // In SPARQL, a term is a literal iff it is not an IRI and not a blank node. - // At this layer, non-literals are represented as `Sid` (node ref) or `Iri`. - !matches!(v, ComparableValue::Sid(_) | ComparableValue::Iri(_)) + // In SPARQL, a term is a literal iff it is not an IRI, a blank node or + // a triple term. At this layer, IRIs and blank nodes are `Sid` (node + // ref) or `Iri`; a triple term is a `TypedLiteral` holding one. + !matches!( + v, + ComparableValue::Sid(_) + | ComparableValue::Iri(_) + | ComparableValue::TypedLiteral { + val: fluree_db_core::FlakeValue::TripleTerm(_), + .. + } + ) })))) } diff --git a/fluree-db-query/src/fast_count.rs b/fluree-db-query/src/fast_count.rs index c865620f5a..2d31cea2bb 100644 --- a/fluree-db-query/src/fast_count.rs +++ b/fluree-db-query/src/fast_count.rs @@ -2024,9 +2024,10 @@ fn count_literal_rows_from_stats(stats: &fluree_db_core::IndexStats, g_id: Graph (literals > 0).then_some(literals) } +/// A triple term is neither a node reference nor a literal. fn is_literal_otype(ot_u16: u16) -> bool { let ot = OType::from_u16(ot_u16); - !ot.is_node_ref() + !ot.is_node_ref() && ot != OType::TRIPLE_TERM } fn count_literal_rows_psot(store: &BinaryIndexStore, g_id: GraphId) -> Result { From aaf49037cd42ad42076b4f7f5fe18d6382e3c586 Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 09:13:05 -0400 Subject: [PATCH 87/92] fix(query): TermComponents streams within batch size and the budget The operator gathered every candidate of an input batch into one output batch, so decomposing a predicate's 2,500 terms with a batch size of 4 came back as one batch of 2,500, and none of the dictionary reads were charged. It now resumes mid-row across calls and emits at most a batch at a time, charges the dictionary reads it caches per input batch (released when the next batch starts) and the novelty term set, and checkpoints while it iterates. Novelty's terms are collected when a row first needs them, not on every open. --- fluree-db-api/tests/it_triple_term_links.rs | 93 ++++ fluree-db-query/src/term_components.rs | 517 ++++++++++++-------- 2 files changed, 416 insertions(+), 194 deletions(-) diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 9b1a655097..7bbda3a3ef 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -12,6 +12,7 @@ use crate::support; use fluree_db_api::{FlureeBuilder, LedgerState}; use serde_json::{json, Value as JsonValue}; +use std::sync::Arc; const CLAIMS: &str = r#"VERSION "1.2" @prefix ex: . @@ -2856,6 +2857,98 @@ async fn assert_term_values( } } +/// Decomposing many terms streams them a batch at a time and charges the +/// dictionary reads it caches to the query budget. +#[tokio::test] +async fn term_components_stream_in_batches_within_the_budget() { + use fluree_db_core::QueryCancellation; + use fluree_db_query::{ + binding::Batch, + context::ExecutionContext, + error::QueryError, + ir::{Component, TermComponentsPattern}, + operator::Operator, + seed::BatchSeedOperator, + term_components::TermComponentsOperator, + VarRegistry, + }; + + const TERMS: usize = 2_500; + let fluree = FlureeBuilder::memory().build_memory(); + let ledger_id = "it/triple-term-links:stream"; + let mut ttl = String::from("@prefix ex: .\n"); + for i in 0..TERMS { + ttl.push_str(&format!( + "ex:doc{i} ex:mentions <<( ex:s{i} ex:p ex:o )>> .\n" + )); + } + fluree + .upsert_turtle(support::genesis_ledger(&fluree, ledger_id), &ttl) + .await + .expect("seed"); + support::rebuild_and_publish_index(&fluree, ledger_id).await; + let ledger = fluree.ledger(ledger_id).await.expect("load"); + let store = ledger + .binary_store + .as_ref() + .unwrap() + .0 + .clone() + .downcast::() + .unwrap(); + + let mut vars = VarRegistry::new(); + let term = vars.get_or_insert("?t"); + let subject = vars.get_or_insert("?s"); + let pattern = TermComponentsPattern { + term, + subject: Component::Var(subject), + predicate: Component::Node(ledger.snapshot.encode_iri("http://example.org/p").unwrap()), + object: Component::Any, + }; + let operator = || { + TermComponentsOperator::new( + Box::new(BatchSeedOperator::from_batch(Batch::empty_schema_with_len( + 1, + ))), + pattern.clone(), + ) + }; + + let cancellation = QueryCancellation::new(); + cancellation.set_memory_limit(usize::MAX); + let ctx = ExecutionContext::new(&ledger.snapshot, &vars) + .with_binary_store(Arc::clone(&store), 0) + .with_batch_size(4) + .with_cancellation(cancellation); + let mut op = operator(); + op.open(&ctx).await.unwrap(); + let mut rows = 0; + while let Some(batch) = op.next_batch(&ctx).await.unwrap() { + assert!(batch.len() <= 4, "a batch holds {} rows", batch.len()); + if rows == 0 { + assert!(ctx.mem_used() > 0, "the cached dictionary read is charged"); + } + rows += batch.len(); + } + assert_eq!(rows, TERMS); + op.close(); + + let cancellation = QueryCancellation::new(); + cancellation.set_memory_limit(4 * 1024); + let ctx = ExecutionContext::new(&ledger.snapshot, &vars) + .with_binary_store(store, 0) + .with_batch_size(4) + .with_cancellation(cancellation); + let mut op = operator(); + op.open(&ctx).await.unwrap(); + let err = op.next_batch(&ctx).await.expect_err("past the budget"); + assert!( + matches!(err, QueryError::MemoryBudgetExceeded { .. }), + "{err:?}" + ); +} + /// A triple term is not a literal to the indexed literal count either. #[tokio::test] async fn literal_count_leaves_triple_terms_out() { diff --git a/fluree-db-query/src/term_components.rs b/fluree-db-query/src/term_components.rs index 115f5e23c7..50b0b6a4bf 100644 --- a/fluree-db-query/src/term_components.rs +++ b/fluree-db-query/src/term_components.rs @@ -16,7 +16,9 @@ //! the graph, time, policy and novelty checks. //! //! Terms the index has not interned yet appear only in novelty's links; they -//! are collected once per open and offered alongside the dictionary's. +//! are collected once, when a row first needs them, and offered alongside the +//! dictionary's. Output is emitted a batch at a time, resuming mid-row, and +//! the dictionary reads a batch caches are charged to the query budget. use crate::binding::{Batch, Binding}; use crate::context::ExecutionContext; @@ -58,6 +60,27 @@ struct NoveltyTerm { encoded: Option<(u64, TermKey)>, } +/// Dictionary terms read for one lookup, shared by the rows that ask for it. +type Found = Arc>; + +/// One input row's candidates, consumed across `next_batch` calls. +struct RowCursor { + row: Vec, + t: i64, + /// The bound term's own candidate. + single: Option, + found: Option, + found_pos: usize, + /// Novelty's terms follow the dictionary's for an unbound term. + with_novelty: bool, + novelty_pos: usize, +} + +/// Query-budget bytes for `n` cached dictionary terms. +fn found_bytes(n: usize) -> usize { + n * std::mem::size_of::<(TermKey, u64)>() +} + pub struct TermComponentsOperator { child: BoxedOperator, pattern: TermComponentsPattern, @@ -67,7 +90,17 @@ pub struct TermComponentsOperator { store: Option>, constants: Option, reifies_p_id: u32, - novelty_terms: Vec, + /// Collected when a row first needs them. + novelty_terms: Option>, + novelty_bytes: usize, + input: Option, + input_row: usize, + cursor: Option, + /// Distinct prefixes, objects and predicates are read once per input batch. + by_prefix: HashMap<(u64, Option), Found>, + by_predicate: HashMap, Found>, + by_object: HashMap<(u16, u64, Option), Found>, + cache_bytes: usize, } impl TermComponentsOperator { @@ -88,10 +121,262 @@ impl TermComponentsOperator { store: None, constants: None, reifies_p_id: 0, - novelty_terms: Vec::new(), + novelty_terms: None, + novelty_bytes: 0, + input: None, + input_row: 0, + cursor: None, + by_prefix: HashMap::new(), + by_predicate: HashMap::new(), + by_object: HashMap::new(), + cache_bytes: 0, } } + fn clear_caches(&mut self, ctx: &ExecutionContext<'_>) { + self.by_prefix.clear(); + self.by_predicate.clear(); + self.by_object.clear(); + ctx.release(self.cache_bytes); + self.cache_bytes = 0; + } + + /// Charge a dictionary read the batch caches. + fn cache(&mut self, found: Vec<(TermKey, u64)>, ctx: &ExecutionContext<'_>) -> Result { + let bytes = found_bytes(found.len()); + ctx.record_alloc(bytes); + self.cache_bytes += bytes; + ctx.checkpoint()?; + Ok(Arc::new(found)) + } + + fn ensure_novelty_terms(&mut self, ctx: &ExecutionContext<'_>) -> Result<()> { + if self.novelty_terms.is_none() { + let terms = self.collect_novelty_terms(ctx)?; + self.novelty_bytes = terms.len() + * (std::mem::size_of::() + std::mem::size_of::()); + ctx.record_alloc(self.novelty_bytes); + ctx.checkpoint()?; + self.novelty_terms = Some(terms); + } + Ok(()) + } + + /// The next input row with candidates, as a cursor; `None` once the + /// child is exhausted. A new input batch drops the previous one's caches. + async fn next_row(&mut self, ctx: &ExecutionContext<'_>) -> Result> { + loop { + if self + .input + .as_ref() + .is_none_or(|input| self.input_row >= input.len()) + { + self.clear_caches(ctx); + self.input = None; + match self.child.next_batch(ctx).await? { + Some(batch) if !batch.is_empty() => { + self.input = Some(batch); + self.input_row = 0; + } + Some(_) => continue, + None => return Ok(None), + } + } + let row_idx = self.input_row; + self.input_row += 1; + let input = self.input.as_ref().expect("input batch"); + let row: Vec = (0..self.schema.len()) + .map(|col| { + if col < self.child_width { + input.get_by_col(row_idx, col).clone() + } else { + Binding::Unbound + } + }) + .collect(); + if let Some(cursor) = self.row_cursor(row, ctx)? { + return Ok(Some(cursor)); + } + } + } + + /// Where `row`'s candidates come from; `None` when it has none. + fn row_cursor( + &mut self, + row: Vec, + ctx: &ExecutionContext<'_>, + ) -> Result> { + let cursor = |row, t, single, found| RowCursor { + row, + t, + single, + found, + found_pos: 0, + with_novelty: false, + novelty_pos: 0, + }; + match self.value(&row, self.pattern.term).cloned() { + Some(Binding::EncodedLit { + o_kind, o_key, t, .. + }) if o_kind == ObjKind::TRIPLE_TERM.as_u8() => { + let Some(store) = &self.store else { + return Ok(None); + }; + let key = crate::binary_scan::term_key_for_handle( + o_key, + store, + ctx.dict_novelty.as_ref(), + ) + .map_err(|e| QueryError::from_io("resolve_term_key", e))?; + // A provisional handle whose components no dictionary + // encodes is offered as its novelty term. + let candidate = match key { + Some(key) => Candidate::Encoded { handle: o_key, key }, + None => match novelty_term_index(o_key) + .zip(ctx.dict_novelty.as_ref()) + .and_then(|(index, dn)| dn.terms.resolve(index)) + { + Some(term) => Candidate::Materialized(Box::new(term.clone())), + None => { + return Err(QueryError::Internal(format!( + "triple-term handle {o_key:#x} has no dictionary entry" + ))) + } + }, + }; + Ok(Some(cursor(row, t, Some(candidate), None))) + } + Some(Binding::Lit { + val: FlakeValue::TripleTerm(term), + .. + }) => Ok(Some(cursor( + row, + 0, + Some(Candidate::Materialized(term)), + None, + ))), + // Bound to something that is not a term. + Some(_) => Ok(None), + None => { + let found = match self.store.clone() { + Some(store) => self.dictionary_candidates(&row, &store, ctx)?, + None => None, + }; + self.ensure_novelty_terms(ctx)?; + let mut cursor = cursor(row, 0, None, found); + cursor.with_novelty = true; + Ok(Some(cursor)) + } + } + } + + /// The dictionary terms an unbound term can be: by the subject anchor, + /// else the object anchor, else every term of the predicate (or of all). + fn dictionary_candidates( + &mut self, + row: &[Binding], + store: &BinaryIndexStore, + ctx: &ExecutionContext<'_>, + ) -> Result> { + let Some(terms) = store.term_dict() else { + return Ok(None); + }; + let p_id = self.fixed_predicate(row, store); + match self.anchor_subject(row, store, ctx)? { + Some(Ok(s_id)) => { + if let Some(found) = self.by_prefix.get(&(s_id, p_id)) { + return Ok(Some(Arc::clone(found))); + } + let found = terms + .terms_with_subject(s_id, p_id) + .map_err(|e| QueryError::from_io("term components: subject", e))?; + let found = self.cache(found, ctx)?; + self.by_prefix.insert((s_id, p_id), Arc::clone(&found)); + return Ok(Some(found)); + } + // No interned subject: only novelty can match. + Some(Err(())) => return Ok(None), + None => {} + } + match self.anchor_object(row, store, ctx)? { + Some(Err(())) => return Ok(None), + Some(Ok((o_type, o_key))) => { + if let Some(found) = self.by_object.get(&(o_type, o_key, p_id)) { + return Ok(Some(Arc::clone(found))); + } + let found = terms + .terms_with_object(o_type, o_key, p_id) + .map_err(|e| QueryError::from_io("term components: object", e))?; + stamp_fast_path( + OBJECT_SITE, + match found { + Some(_) => FastPathOutcome::Proceed, + None => FastPathOutcome::Fallback(FastPathFallback::GateDeclined), + }, + ); + // A dictionary without the object tree falls to the scan. + if let Some(found) = found { + let found = self.cache(found, ctx)?; + self.by_object + .insert((o_type, o_key, p_id), Arc::clone(&found)); + return Ok(Some(found)); + } + } + None => {} + } + if let Some(found) = self.by_predicate.get(&p_id) { + return Ok(Some(Arc::clone(found))); + } + let predicates: Vec = match p_id { + Some(p) => vec![p], + None => terms.predicates().collect(), + }; + let mut all = Vec::new(); + for p in predicates { + all.extend( + terms + .terms_of_predicate(p) + .map_err(|e| QueryError::from_io("term components: scan", e))?, + ); + ctx.checkpoint()?; + } + let found = self.cache(all, ctx)?; + self.by_predicate.insert(p_id, Arc::clone(&found)); + Ok(Some(found)) + } + + /// The cursor's next candidate: the bound term, then the dictionary's, + /// then novelty's that can match the row's subject anchor. + fn next_candidate(&self, cursor: &mut RowCursor) -> Option { + if let Some(candidate) = cursor.single.take() { + return Some(candidate); + } + if let Some((key, handle)) = cursor + .found + .as_ref() + .and_then(|found| found.get(cursor.found_pos)) + { + cursor.found_pos += 1; + return Some(Candidate::Encoded { + handle: *handle, + key: *key, + }); + } + if cursor.with_novelty { + let terms = self.novelty_terms.as_deref().unwrap_or(&[]); + while let Some(nt) = terms.get(cursor.novelty_pos) { + cursor.novelty_pos += 1; + if self.novelty_anchor_matches(&cursor.row, nt) { + return Some(match nt.encoded { + Some((handle, key)) => Candidate::Encoded { handle, key }, + None => Candidate::Materialized(nt.term.clone()), + }); + } + } + } + None + } + /// The terms novelty holds that the dictionary does not: those of any /// predicate, and those nested in them. The relation may offer a term no /// asserted flake still holds; the pattern binding the term joins it out. @@ -556,7 +841,6 @@ impl Operator for TermComponentsOperator { )) .unwrap_or(0); } - self.novelty_terms = self.collect_novelty_terms(ctx)?; self.state = OperatorState::Open; Ok(()) } @@ -565,207 +849,52 @@ impl Operator for TermComponentsOperator { if self.state != OperatorState::Open { return Ok(None); } + let limit = ctx.batch_size.max(1); + let mut columns: Vec> = vec![Vec::new(); self.schema.len()]; + let mut produced = 0usize; + let mut visited = 0usize; loop { - let input = match self.child.next_batch(ctx).await? { - Some(b) if !b.is_empty() => b, - Some(_) => continue, - None => { - self.state = OperatorState::Exhausted; - return Ok(None); - } - }; - let mut columns: Vec> = vec![Vec::new(); self.schema.len()]; - // Distinct prefixes and predicates are read once per batch. - let mut by_prefix: HashMap<(u64, Option), Arc>> = - HashMap::new(); - let mut by_predicate: HashMap, Arc>> = HashMap::new(); - let mut by_object: HashMap<(u16, u64, Option), Arc>> = - HashMap::new(); - for row_idx in 0..input.len() { - let mut row: Vec = (0..self.schema.len()) - .map(|col| { - if col < self.child_width { - input.get_by_col(row_idx, col).clone() - } else { - Binding::Unbound - } - }) - .collect(); - let (candidates, t): (Vec, i64) = match self - .value(&row, self.pattern.term) - .cloned() - { - Some(Binding::EncodedLit { - o_kind, o_key, t, .. - }) if o_kind == ObjKind::TRIPLE_TERM.as_u8() => { - let Some(store) = &self.store else { - continue; - }; - let key = crate::binary_scan::term_key_for_handle( - o_key, - store, - ctx.dict_novelty.as_ref(), - ) - .map_err(|e| QueryError::from_io("resolve_term_key", e))?; - // A provisional handle whose components no dictionary - // encodes is offered as its novelty term. - let candidate = match key { - Some(key) => Candidate::Encoded { handle: o_key, key }, - None => match novelty_term_index(o_key) - .zip(ctx.dict_novelty.as_ref()) - .and_then(|(index, dn)| dn.terms.resolve(index)) - { - Some(term) => Candidate::Materialized(Box::new(term.clone())), - None => { - return Err(QueryError::Internal(format!( - "triple-term handle {o_key:#x} has no dictionary entry" - ))) - } - }, - }; - (vec![candidate], t) - } - Some(Binding::Lit { - val: FlakeValue::TripleTerm(term), - .. - }) => (vec![Candidate::Materialized(term)], 0), - // Bound to something that is not a term. - Some(_) => continue, + let mut cursor = match self.cursor.take() { + Some(cursor) => cursor, + None => match self.next_row(ctx).await? { + Some(cursor) => cursor, None => { - let mut candidates = Vec::new(); - if let Some(store) = self.store.clone() { - if let Some(terms) = store.term_dict() { - let p_id = self.fixed_predicate(&row, &store); - let found = match self.anchor_subject(&row, &store, ctx)? { - Some(Ok(s_id)) => match by_prefix.get(&(s_id, p_id)) { - Some(found) => Some(Arc::clone(found)), - None => { - let found = Arc::new( - terms.terms_with_subject(s_id, p_id).map_err( - |e| { - QueryError::from_io( - "term components: subject", - e, - ) - }, - )?, - ); - by_prefix.insert((s_id, p_id), Arc::clone(&found)); - Some(found) - } - }, - // No interned subject: only novelty can match. - Some(Err(())) => None, - None => { - let by_obj = match self.anchor_object(&row, &store, ctx)? { - Some(Err(())) => Some(Arc::new(Vec::new())), - Some(Ok((o_type, o_key))) => { - match by_object.get(&(o_type, o_key, p_id)) { - Some(found) => Some(Arc::clone(found)), - None => { - let found = terms - .terms_with_object(o_type, o_key, p_id) - .map_err(|e| { - QueryError::from_io( - "term components: object", - e, - ) - })?; - stamp_fast_path( - OBJECT_SITE, - match found { - Some(_) => FastPathOutcome::Proceed, - None => FastPathOutcome::Fallback( - FastPathFallback::GateDeclined, - ), - }, - ); - found.map(|found| { - let found = Arc::new(found); - by_object.insert( - (o_type, o_key, p_id), - Arc::clone(&found), - ); - found - }) - } - } - } - None => None, - }; - match by_obj { - Some(found) => Some(found), - // No object anchor, or a dictionary - // without the object tree: scan. - None => match by_predicate.get(&p_id) { - Some(found) => Some(Arc::clone(found)), - None => { - let predicates: Vec = match p_id { - Some(p) => vec![p], - None => terms.predicates().collect(), - }; - let mut all = Vec::new(); - for p in predicates { - all.extend( - terms.terms_of_predicate(p).map_err( - |e| { - QueryError::from_io( - "term components: scan", - e, - ) - }, - )?, - ); - } - let found = Arc::new(all); - by_predicate.insert(p_id, Arc::clone(&found)); - Some(found) - } - }, - } - } - }; - if let Some(found) = found { - candidates.extend(found.iter().map(|(key, handle)| { - Candidate::Encoded { - handle: *handle, - key: *key, - } - })); - } - } - } - candidates.extend( - self.novelty_terms - .iter() - .filter(|nt| self.novelty_anchor_matches(&row, nt)) - .map(|nt| match nt.encoded { - Some((handle, key)) => Candidate::Encoded { handle, key }, - None => Candidate::Materialized(nt.term.clone()), - }), - ); - (candidates, 0) + self.state = OperatorState::Exhausted; + self.clear_caches(ctx); + return Ok((produced > 0) + .then(|| Batch::new(Arc::clone(&self.schema), columns)) + .transpose()?); } - }; - for candidate in &candidates { - let mut out = row.clone(); - if self.apply(candidate, t, &mut out, ctx)? { - for (col, value) in out.into_iter().enumerate() { - columns[col].push(value); - } + }, + }; + while let Some(candidate) = self.next_candidate(&mut cursor) { + visited += 1; + if visited.is_multiple_of(1024) { + ctx.checkpoint()?; + } + let mut out = cursor.row.clone(); + if self.apply(&candidate, cursor.t, &mut out, ctx)? { + for (col, value) in out.into_iter().enumerate() { + columns[col].push(value); + } + produced += 1; + if produced >= limit { + self.cursor = Some(cursor); + return Ok(Some(Batch::new(Arc::clone(&self.schema), columns)?)); } } - row.clear(); - } - if columns.first().is_some_and(Vec::is_empty) { - continue; } - return Ok(Some(Batch::new(Arc::clone(&self.schema), columns)?)); } } fn close(&mut self) { self.child.close(); + self.input = None; + self.cursor = None; + self.by_prefix.clear(); + self.by_predicate.clear(); + self.by_object.clear(); + self.novelty_terms = None; self.state = OperatorState::Closed; } From cb788608cc1b54c418d04915e58aed38909007c3 Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 09:15:38 -0400 Subject: [PATCH 88/92] fix(export): count novelty's links against the preload bound The bound read only the index's link counts, so an unindexed ledger counted as zero and preloaded however many links novelty held. Preloading now stops and falls back to per-batch probes as soon as the links it has read pass the bound, novelty's included. --- fluree-db-api/src/export_annotations.rs | 48 +++++++++++++++++++------ 1 file changed, 38 insertions(+), 10 deletions(-) diff --git a/fluree-db-api/src/export_annotations.rs b/fluree-db-api/src/export_annotations.rs index de16089e00..e8e46b29e2 100644 --- a/fluree-db-api/src/export_annotations.rs +++ b/fluree-db-api/src/export_annotations.rs @@ -139,7 +139,7 @@ impl<'a> AnnotationProbe<'a> { in_scope: Mutex::new(HashSet::new()), }; // Unknown counts on an index mean an index of unknown size; without - // one, the links are novelty's. + // one, the links are novelty's, which `preload` counts as it reads. let indexed_links = match ledger.snapshot.stats.as_ref() { Some(stats) => stats .links @@ -148,19 +148,19 @@ impl<'a> AnnotationProbe<'a> { None => Some(0), }; if indexed_links.is_some_and(|n| n <= preload_max_links) { - probe.preloaded = Some( - probe - .preload() - .await - .map_err(|e| crate::ApiError::internal(e.to_string()))?, - ); + probe.preloaded = probe + .preload(preload_max_links) + .await + .map_err(|e| crate::ApiError::internal(e.to_string()))?; } Ok(Some(probe)) } /// Read every live link, telling markers from rows by one lookup per - /// `(graph, subject, predicate)`. - async fn preload(&self) -> io::Result { + /// `(graph, subject, predicate)`; `None` once more than `max_links` are + /// read, novelty's included. + async fn preload(&self, max_links: u64) -> io::Result> { + let mut read: u64 = 0; let graphs = std::iter::once(0).chain( self.ledger .snapshot @@ -178,6 +178,10 @@ impl<'a> AnnotationProbe<'a> { RangeMatch::predicate(fluree_db_core::rdf_reifies_sid().clone()), ) .await?; + read += flakes.len() as u64; + if read > max_links { + return Ok(None); + } // Moved, not cloned: the map holds each term's parts. let graph_links = links.entry(g_id).or_default(); for flake in flakes { @@ -235,7 +239,7 @@ impl<'a> AnnotationProbe<'a> { for reifiers in links.values_mut().flat_map(HashMap::values_mut) { reifiers.sort(); } - Ok(Preloaded { links, unasserted }) + Ok(Some(Preloaded { links, unasserted })) } async fn range( @@ -391,6 +395,30 @@ mod tests { use crate::export::ExportFormat; use crate::{FlureeBuilder, ReindexOptions}; + /// The preload bound counts novelty's links, which index stats do not. + #[tokio::test] + async fn preload_bound_counts_novelty_links() { + let fluree = FlureeBuilder::memory().build_memory(); + let id = "export/preload-bound:main"; + let ledger = fluree.create_ledger(id).await.unwrap(); + let ledger = fluree + .upsert_turtle( + ledger, + "VERSION \"1.2\"\n@prefix ex: .\n\ + ex:alice ex:worksFor ex:acme ~ ex:claim1 {| ex:role \"Engineer\" |} .\n", + ) + .await + .unwrap() + .ledger; + for (max, preloaded) in [(0, false), (1, true)] { + let probe = super::AnnotationProbe::for_ledger(&ledger, ledger.t(), max) + .await + .unwrap() + .expect("an annotated ledger"); + assert_eq!(probe.preloaded.is_some(), preloaded, "max_links={max}"); + } + } + /// Per-batch probes write exactly what the up-front read writes, in every /// format, over indexed links and novelty ones; a term only novelty holds /// decodes through dictionary novelty. From efce5dbc61438013618bdb02da6da5f92450cc05 Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 09:19:31 -0400 Subject: [PATCH 89/92] refactor(query): one triple-term object conversion, Binding::term_object The formatters and OBJECT()/TermComponents each turned a term's object into a binding, keeping its datatype or language tag, in their own copy. --- fluree-db-api/src/format/mod.rs | 16 +--------------- fluree-db-query/src/binding.rs | 19 +++++++++++++++++++ fluree-db-query/src/eval/rdf.rs | 21 +-------------------- fluree-db-query/src/term_components.rs | 2 +- 4 files changed, 22 insertions(+), 36 deletions(-) diff --git a/fluree-db-api/src/format/mod.rs b/fluree-db-api/src/format/mod.rs index fc82164f98..d73a093ef9 100644 --- a/fluree-db-api/src/format/mod.rs +++ b/fluree-db-api/src/format/mod.rs @@ -65,25 +65,11 @@ mod typed; pub(crate) fn triple_term_components( term: &fluree_db_core::TripleTermValue, ) -> [fluree_db_query::binding::Binding; 3] { - use fluree_db_core::{DatatypeConstraint, FlakeValue}; use fluree_db_query::binding::Binding; - let object = match &term.o { - FlakeValue::Ref(sid) => Binding::sid(sid.clone()), - other => Binding::Lit { - val: other.clone(), - dtc: match &term.lang { - Some(lang) => DatatypeConstraint::LangTag(std::sync::Arc::from(lang.as_str())), - None => DatatypeConstraint::Explicit(term.dt.clone()), - }, - t: None, - op: None, - p_id: None, - }, - }; [ Binding::sid(term.s.clone()), Binding::sid(term.p.clone()), - object, + Binding::term_object(term), ] } diff --git a/fluree-db-query/src/binding.rs b/fluree-db-query/src/binding.rs index 2800285b49..41ad137e6f 100644 --- a/fluree-db-query/src/binding.rs +++ b/fluree-db-query/src/binding.rs @@ -327,6 +327,25 @@ impl UnmatchedOptional { } impl Binding { + /// A triple term's object as a binding: a node, or a literal with the + /// term's datatype or language tag, which `"chat"@fr` and `"5"^^xsd:int` + /// must keep (a nested term is such a literal). + pub fn term_object(term: &fluree_db_core::TripleTermValue) -> Binding { + match &term.o { + FlakeValue::Ref(sid) => Binding::sid(sid.clone()), + other => Binding::Lit { + val: other.clone(), + dtc: match &term.lang { + Some(lang) => DatatypeConstraint::LangTag(Arc::from(lang.as_str())), + None => DatatypeConstraint::Explicit(term.dt.clone()), + }, + t: None, + op: None, + p_id: None, + }, + } + } + /// Create a new literal binding /// /// # Panics diff --git a/fluree-db-query/src/eval/rdf.rs b/fluree-db-query/src/eval/rdf.rs index 0238557265..f17d81a98f 100644 --- a/fluree-db-query/src/eval/rdf.rs +++ b/fluree-db-query/src/eval/rdf.rs @@ -671,29 +671,10 @@ pub(crate) fn term_component_binding( Ok(Some(match func { Function::TripleSubject => Binding::sid(term.s), Function::TriplePredicate => Binding::sid(term.p), - _ => materialized_term_object(&term), + _ => Binding::term_object(&term), })) } -/// A materialized term's object with its datatype or language tag, which -/// `OBJECT()` must keep: `"chat"@fr` is not `"chat"`, `"5"^^xsd:int` is not -/// `5`. -pub(crate) fn materialized_term_object(term: &fluree_db_core::TripleTermValue) -> Binding { - match &term.o { - fluree_db_core::FlakeValue::Ref(sid) => Binding::sid(sid.clone()), - other => Binding::Lit { - val: other.clone(), - dtc: match &term.lang { - Some(lang) => fluree_db_core::DatatypeConstraint::LangTag(Arc::from(lang.as_str())), - None => fluree_db_core::DatatypeConstraint::Explicit(term.dt.clone()), - }, - t: None, - op: None, - p_id: None, - }, - } -} - /// One accessor: the component's binding converted exactly as a bound /// variable is. fn eval_term_accessor( diff --git a/fluree-db-query/src/term_components.rs b/fluree-db-query/src/term_components.rs index 50b0b6a4bf..f2e043f903 100644 --- a/fluree-db-query/src/term_components.rs +++ b/fluree-db-query/src/term_components.rs @@ -626,7 +626,7 @@ impl TermComponentsOperator { (Component::Var(_), Candidate::Materialized(term)) => match position { 0 => Binding::sid(term.s.clone()), 1 => Binding::sid(term.p.clone()), - _ => crate::eval::rdf::materialized_term_object(term), + _ => Binding::term_object(term), }, }; let Component::Var(v) = component else { From ba480bc87167f4d83587b4eb33a673efc0361bfa Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 09:19:31 -0400 Subject: [PATCH 90/92] refactor(json-ld): share the triple-term shape check across surfaces Transactions, import and query values each split `{"@id": {"@id": s, p: o}}` on their own and had drifted: values took any key but `@id` as the predicate, and only import rejected an object node with properties. `triple_term_parts` now checks the shape once (a subject, one predicate, one object that is a reference, value or nested term); each surface keeps its own lowering. --- fluree-db-query/src/parse/values.rs | 31 ++++------- fluree-db-transact/src/parse/jsonld.rs | 12 ++--- fluree-graph-json-ld/src/adapter.rs | 25 ++------- fluree-graph-json-ld/src/lib.rs | 1 + fluree-graph-json-ld/src/triple_term.rs | 72 +++++++++++++++++++++++++ 5 files changed, 90 insertions(+), 51 deletions(-) create mode 100644 fluree-graph-json-ld/src/triple_term.rs diff --git a/fluree-db-query/src/parse/values.rs b/fluree-db-query/src/parse/values.rs index e7b7c63268..b867b944b3 100644 --- a/fluree-db-query/src/parse/values.rs +++ b/fluree-db-query/src/parse/values.rs @@ -202,29 +202,16 @@ fn parse_triple_term( term: &serde_json::Map, ctx: &JsonLdParseCtx, ) -> Result { - let invalid = || { - ParseError::InvalidWhere( - "a triple term in values is {\"@id\": s, p: o}: an IRI subject and one \ - predicate with one object" - .to_string(), - ) - }; - let subject = term - .get("@id") - .and_then(JsonValue::as_str) - .ok_or_else(invalid)?; - let mut properties = term.iter().filter(|(k, _)| k.as_str() != "@id"); - let (Some((predicate, object)), None) = (properties.next(), properties.next()) else { - return Err(invalid()); - }; - let object = match object { - JsonValue::Array(items) if items.len() == 1 => &items[0], - JsonValue::Array(_) => return Err(invalid()), - other => other, - }; - let object = parse_values_cell(object, ctx)?; + let invalid = |msg: &str| ParseError::InvalidWhere(format!("triple term in values: {msg}")); + let parts = fluree_graph_json_ld::triple_term::triple_term_parts(term).map_err(invalid)?; + let subject = parts + .subject + .as_str() + .ok_or_else(|| invalid("its subject must be an IRI"))?; + let predicate = parts.predicate; + let object = parse_values_cell(parts.object, ctx)?; if matches!(object, UnresolvedValue::Unbound) { - return Err(invalid()); + return Err(invalid("its object must be a constant")); } Ok(UnresolvedValue::TripleTerm { subject: Arc::from(ctx.expand_vocab(subject)?.0), diff --git a/fluree-db-transact/src/parse/jsonld.rs b/fluree-db-transact/src/parse/jsonld.rs index 6ad7788ced..bd09b442e3 100644 --- a/fluree-db-transact/src/parse/jsonld.rs +++ b/fluree-db-transact/src/parse/jsonld.rs @@ -1734,14 +1734,10 @@ fn parse_expanded_triple_term_with_ctx( ctx: &mut TemplateParseCtx<'_>, ) -> Result { let invalid = |msg: &str| TransactError::Parse(format!("triple term: {msg}")); - let s = match term.get("@id") { - Some(id) => parse_expanded_id_with_ctx(id, ctx)?, - None => return Err(invalid("@id must name the subject")), - }; - let mut pairs = term.iter().filter(|(k, _)| !k.starts_with('@')); - let (Some((key, values)), None) = (pairs.next(), pairs.next()) else { - return Err(invalid("it must describe exactly one triple")); - }; + let parts = fluree_graph_json_ld::triple_term::triple_term_parts(term).map_err(invalid)?; + let s = parse_expanded_id_with_ctx(parts.subject, ctx)?; + let key = parts.predicate; + let values = parts.object; let p = if key.starts_with('?') { TemplateTerm::Var(ctx.vars.get_or_insert(key)) } else { diff --git a/fluree-graph-json-ld/src/adapter.rs b/fluree-graph-json-ld/src/adapter.rs index 8acc945c17..a093486907 100644 --- a/fluree-graph-json-ld/src/adapter.rs +++ b/fluree-graph-json-ld/src/adapter.rs @@ -302,31 +302,14 @@ fn process_triple_term( if !sink.supports_triple_terms() { return Err(invalid("this destination does not hold triple-term values")); } - let subject = match term.get("@id").and_then(Value::as_str) { + let parts = crate::triple_term::triple_term_parts(term).map_err(invalid)?; + let subject = match parts.subject.as_str() { Some(id) if id.starts_with("_:") => sink.term_blank(Some(strip_blank_prefix(id))), Some(id) => sink.term_iri(id), None => return Err(invalid("@id must name the subject")), }; - let mut pairs = term.iter().filter(|(k, _)| !k.starts_with('@')); - let (Some((predicate, values)), None) = (pairs.next(), pairs.next()) else { - return Err(invalid("it must describe exactly one triple")); - }; - let value = match values { - Value::Array(items) if items.len() == 1 => &items[0], - Value::Array(_) => return Err(invalid("it must describe exactly one triple")), - value => value, - }; - // A node with properties would assert them. - let reference_or_value = match value { - Value::Object(o) => o.contains_key("@value") || (o.len() == 1 && o.contains_key("@id")), - _ => true, - }; - if !reference_or_value { - return Err(invalid( - "its object must be a reference, a value or a triple term", - )); - } - let ProcessedValue::Single(object) = process_value(value, sink)? else { + let predicate = parts.predicate; + let ProcessedValue::Single(object) = process_value(parts.object, sink)? else { return Err(invalid("its object must be a reference or a value")); }; let predicate = sink.term_iri(predicate); diff --git a/fluree-graph-json-ld/src/lib.rs b/fluree-graph-json-ld/src/lib.rs index f928183432..f75f7fc489 100644 --- a/fluree-graph-json-ld/src/lib.rs +++ b/fluree-graph-json-ld/src/lib.rs @@ -37,6 +37,7 @@ pub mod error; pub mod expand; pub mod iri; pub mod normalize; +pub mod triple_term; // GraphSink adapter for emitting triples to fluree-graph-ir pub mod adapter; diff --git a/fluree-graph-json-ld/src/triple_term.rs b/fluree-graph-json-ld/src/triple_term.rs new file mode 100644 index 0000000000..6552df74d5 --- /dev/null +++ b/fluree-graph-json-ld/src/triple_term.rs @@ -0,0 +1,72 @@ +//! The shape of a JSON-LD-star triple term, `{"@id": {"@id": s, p: o}}`, +//! shared by every surface that reads one; each lowers the parts its own way. + +use serde_json::{Map, Value}; + +/// A triple term's parts: the subject's `@id` value, its one predicate, and +/// that predicate's one object. +pub struct TripleTermParts<'a> { + pub subject: &'a Value, + pub predicate: &'a str, + pub object: &'a Value, +} + +/// Split the node inside a triple term's `@id` into its parts. The node +/// describes exactly one triple: keywords aside, one predicate with one +/// object, which is a reference, a value or a nested term, never a node with +/// properties of its own (those would be asserted). The error is the reason. +pub fn triple_term_parts(term: &Map) -> Result, &'static str> { + let subject = term.get("@id").ok_or("@id must name the subject")?; + let mut pairs = term.iter().filter(|(k, _)| !k.starts_with('@')); + let (Some((predicate, object)), None) = (pairs.next(), pairs.next()) else { + return Err("it must describe exactly one triple"); + }; + let object = match object { + Value::Array(items) if items.len() == 1 => &items[0], + Value::Array(_) => return Err("it must describe exactly one triple"), + object => object, + }; + if let Value::Object(o) = object { + if !(o.contains_key("@value") || (o.len() == 1 && o.contains_key("@id"))) { + return Err("its object must be a reference, a value or a triple term"); + } + } + Ok(TripleTermParts { + subject, + predicate, + object, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use serde_json::json; + + fn parts(v: Value) -> Result<(String, String), &'static str> { + let Value::Object(map) = v else { + unreachable!() + }; + triple_term_parts(&map).map(|p| (p.predicate.to_string(), p.object.to_string())) + } + + #[test] + fn one_predicate_with_one_object() { + assert_eq!( + parts(json!({"@id": "ex:s", "ex:p": [{"@id": "ex:o"}]})).unwrap(), + ("ex:p".to_string(), r#"{"@id":"ex:o"}"#.to_string()) + ); + assert!(parts(json!({"@id": "ex:s", "ex:p": "v", "@type": "ex:T"})).is_ok()); + assert!(parts(json!({"@id": "ex:s", "ex:p": {"@id": {"@id": "ex:a", "ex:q": 1}}})).is_ok()); + for bad in [ + json!({"ex:p": "v"}), + json!({"@id": "ex:s"}), + json!({"@id": "ex:s", "ex:p": "v", "ex:q": "w"}), + json!({"@id": "ex:s", "ex:p": ["v", "w"]}), + json!({"@id": "ex:s", "ex:p": {"@id": "ex:o", "ex:q": "w"}}), + json!({"@id": "ex:s", "ex:p": {"@list": ["v"]}}), + ] { + assert!(parts(bad.clone()).is_err(), "{bad}"); + } + } +} From eeaa3c264062b1e2deb5524a26899fd6c7cf2dc4 Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 09:21:06 -0400 Subject: [PATCH 91/92] fix(policy): no store, no triple-term decline in the lane gate Without a binary store no lane reads raw rows, so the triple-term check declined fast paths for nothing and broke the gate's coverage test. It now answers no there; rdf:reifies still declines under a restricting policy, which the test now pins. --- fluree-db-query/src/fast_path_common.rs | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/fluree-db-query/src/fast_path_common.rs b/fluree-db-query/src/fast_path_common.rs index ecfaf62204..b29f915911 100644 --- a/fluree-db-query/src/fast_path_common.rs +++ b/fluree-db-query/src/fast_path_common.rs @@ -3936,13 +3936,14 @@ pub fn cursor_fast_path_for_predicate( /// Whether `pred_sid`'s objects in the active graph may include triple terms. /// Triple terms carry no datatype tag of their own, so an `UNKNOWN` tag, an -/// unknown set, or novelty the stats do not see answers yes. +/// unknown set, or novelty the stats do not see answers yes. Without a store +/// no lane reads raw rows, so there is nothing to decline. fn predicate_may_hold_triple_terms(ctx: &ExecutionContext<'_>, pred_sid: &Sid) -> bool { if fluree_db_core::is_rdf_reifies(pred_sid) { return true; } let Some(store) = ctx.binary_store.as_ref() else { - return true; + return false; }; let Some(p_id) = store.sid_to_p_id(pred_sid) else { return true; @@ -4406,6 +4407,12 @@ mod tests { cursor_fast_path_for_predicate(&ctx_allow, &name), PredicateFastPath::Allow ); + // Links' triple terms are checked as the triples they name, which + // the view may cover: never a raw read under a restricting policy. + assert_eq!( + cursor_fast_path_for_predicate(&ctx_allow, fluree_db_core::rdf_reifies_sid()), + PredicateFastPath::Decline + ); // Uncovered predicate + default-deny => short-circuit to Empty. let ctx_deny = make_ctx(false); assert_eq!( From 583c4dbfeeaddd2952048e01b8e2fb82e01b5579 Mon Sep 17 00:00:00 2001 From: bplatz Date: Tue, 6 Oct 2026 09:42:07 -0400 Subject: [PATCH 92/92] feat: say when an annotated ledger's index predates links A ledger whose index was built before triple-term links surfaced only when a read of its annotations failed. Loading one now logs a warning, `fluree info` prints one (local, and tracked from the server's response), and `ledger-info` carries `"needs-link-reindex": true` in its `ledger` block. Pinned against the 4.2.3-written fixture. --- docs/cli/info.md | 1 + docs/design/edge-annotations.md | 2 +- fluree-db-api/src/ledger_info.rs | 13 ++++- fluree-db-api/tests/it_triple_term_links.rs | 17 +++++++ fluree-db-cli/src/commands/info.rs | 25 +++++++++- fluree-db-cli/tests/integration.rs | 53 +++++++++++++++++++++ fluree-db-ledger/src/lib.rs | 7 +++ 7 files changed, 115 insertions(+), 3 deletions(-) diff --git a/docs/cli/info.md b/docs/cli/info.md index 3337ca6ba8..29f8141c97 100644 --- a/docs/cli/info.md +++ b/docs/cli/info.md @@ -29,6 +29,7 @@ For ledgers, displays: - Ledger ID, branch, and type - Current transaction number (t) - Commit and index details +- A warning when the ledger holds edge annotations and its index was built by a release before RDF 1.2 triple-term links: those annotations cannot be read until [`fluree reindex`](reindex.md). The `ledger-info` JSON carries `"needs-link-reindex": true` in its `ledger` block for the same case. For graph sources (Iceberg, R2RML, BM25, etc.), displays: - Name, branch, and type diff --git a/docs/design/edge-annotations.md b/docs/design/edge-annotations.md index 5358374331..706873f72c 100644 --- a/docs/design/edge-annotations.md +++ b/docs/design/edge-annotations.md @@ -66,7 +66,7 @@ Earlier releases stored an annotation as an `f:reifies*` bundle (`f:reifiesSubje Index roots from those releases may also carry an annotation-arena section. Readers skip it, keeping only the arena's two branch CIDs, and the next index build releases the arena's blobs as garbage. -An annotated index built before links has no term dictionary, and its snapshot says so (`LedgerSnapshot::needs_link_reindex`), so a host can schedule the rebuild. Every read of its annotations fails asking for a rebuild (`fluree reindex`, or `Fluree::reindex`) rather than answering without them: link queries, export, and hydration's `@annotation` (`require_link_index`). An incremental build over it declines once its window holds a triple term — a term dictionary covering the window alone would lift that refusal — and the index build falls back to a full rebuild, which links every annotation in history. +An annotated index built before links has no term dictionary, and its snapshot says so (`LedgerSnapshot::needs_link_reindex`), so a host can schedule the rebuild. Loading such a ledger logs a warning, `fluree info` prints one, and `ledger-info` reports `needs-link-reindex`. Every read of its annotations fails asking for a rebuild (`fluree reindex`, or `Fluree::reindex`) rather than answering without them: link queries, export, and hydration's `@annotation` (`require_link_index`). An incremental build over it declines once its window holds a triple term — a term dictionary covering the window alone would lift that refusal — and the index build falls back to a full rebuild, which links every annotation in history. ## See also diff --git a/fluree-db-api/src/ledger_info.rs b/fluree-db-api/src/ledger_info.rs index 9e272275df..8e7e787300 100644 --- a/fluree-db-api/src/ledger_info.rs +++ b/fluree-db-api/src/ledger_info.rs @@ -6,7 +6,8 @@ //! //! ```json //! { -//! "ledger": { "alias", "t", "commit-t", "index-t", "flakes", "size", "named-graphs" }, +//! "ledger": { "alias", "t", "commit-t", "index-t", "flakes", "size", "named-graphs", +//! "needs-link-reindex" (only when true) }, //! "graph": "urn:default", //! "stats": { "flakes", "size", "properties": { ... }, "classes": { ... } }, //! "commit": { ... }, @@ -151,6 +152,14 @@ pub struct Ledger { /// Registered named graphs (always includes `urn:default`). #[serde(rename = "named-graphs")] pub named_graphs: Vec, + /// The index predates RDF 1.2 triple-term links: the ledger's + /// annotations read only after a full reindex. Omitted when false. + #[serde( + rename = "needs-link-reindex", + default, + skip_serializing_if = "std::ops::Not::not" + )] + pub needs_link_reindex: bool, } /// One entry in `ledger.named-graphs`. @@ -818,6 +827,7 @@ fn build_ledger_block(ledger: &LedgerState, stats: &IndexStats) -> Ledger { flakes: Some(stats.flakes as i64), size: stats.size, named_graphs, + needs_link_reindex: ledger.snapshot.needs_link_reindex, } } @@ -1533,6 +1543,7 @@ pub fn build_virtual_ledger_info( flakes: total_rows, size: 0, }], + needs_link_reindex: false, }, graph: DEFAULT_GRAPH_IRI.to_string(), stats: Stats { diff --git a/fluree-db-api/tests/it_triple_term_links.rs b/fluree-db-api/tests/it_triple_term_links.rs index 7bbda3a3ef..3c855dfbe7 100644 --- a/fluree-db-api/tests/it_triple_term_links.rs +++ b/fluree-db-api/tests/it_triple_term_links.rs @@ -2272,8 +2272,25 @@ async fn a_release_built_pre_link_index_refuses_every_annotation_read() { let fluree = FlureeBuilder::file(tmp.path().to_string_lossy().to_string()) .build() .expect("build"); + let (spans, guard) = support::span_capture::init_test_tracing(); let ledger = fluree.ledger("ann:main").await.expect("load"); + drop(guard); assert!(ledger.snapshot.needs_link_reindex); + assert!( + !spans + .find_events( + "index predates RDF 1.2 triple-term links; annotation reads fail until a full \ + reindex (`fluree reindex `)" + ) + .is_empty(), + "loading warns" + ); + let info = fluree + .ledger_info("ann:main") + .execute() + .await + .expect("ledger info"); + assert_eq!(info["ledger"]["needs-link-reindex"], json!(true), "{info}"); let refused = |err: String| assert!(err.contains("fluree reindex"), "{err}"); for format in [ diff --git a/fluree-db-cli/src/commands/info.rs b/fluree-db-cli/src/commands/info.rs index 3de9dd6a55..c7173f398f 100644 --- a/fluree-db-cli/src/commands/info.rs +++ b/fluree-db-cli/src/commands/info.rs @@ -99,6 +99,13 @@ pub async fn run( { println!("Index ID: {index}"); } + if info + .pointer("/ledger/needs-link-reindex") + .and_then(serde_json::Value::as_bool) + == Some(true) + { + print_link_reindex_warning(&remote_alias); + } // Print full JSON if there are stats if info.get("stats").is_some() { @@ -138,7 +145,14 @@ pub async fn run( } println!("Index t: {}", record.index_t); match &record.index_head_id { - Some(id) => println!("Index ID: {id}"), + Some(id) => { + println!("Index ID: {id}"); + let root = fluree.content_store(&ledger_id).get(id).await?; + let snapshot = fluree_db_core::LedgerSnapshot::from_root_bytes(&root)?; + if snapshot.needs_link_reindex { + print_link_reindex_warning(&record.ledger_id); + } + } None => println!("Index ID: (none)"), } } else if let Some(gs) = fluree.nameservice().lookup_graph_source(&ledger_id).await? { @@ -158,3 +172,12 @@ pub async fn run( Ok(()) } + +/// The ledger's index predates triple-term links, so its annotations cannot be +/// read until a full reindex. +fn print_link_reindex_warning(ledger: &str) { + println!( + "Warning: index predates RDF 1.2 triple-term links; annotations cannot be \ + read until `fluree reindex {ledger}`" + ); +} diff --git a/fluree-db-cli/tests/integration.rs b/fluree-db-cli/tests/integration.rs index f1466a72c2..b39843e035 100644 --- a/fluree-db-cli/tests/integration.rs +++ b/fluree-db-cli/tests/integration.rs @@ -2574,6 +2574,59 @@ fn export_annotations_round_trip_in_every_format() { /// A reifier of a triple the ledger does not assert has no edge to carry a /// `~` marker: it is written as its `rdf:reifies` link (`@reifies` in /// JSON-LD), and re-importing it reifies the triple without asserting it. +/// `fluree info` says when an annotated ledger's index predates triple-term +/// links (a ledger written and indexed by 4.2.3), so annotation reads will +/// refuse until a reindex; a current ledger says nothing. +#[test] +fn info_flags_an_index_that_predates_links() { + fn copy_dir(from: &std::path::Path, to: &std::path::Path) { + std::fs::create_dir_all(to).unwrap(); + for entry in std::fs::read_dir(from).unwrap() { + let entry = entry.unwrap(); + let target = to.join(entry.file_name()); + if entry.file_type().unwrap().is_dir() { + copy_dir(&entry.path(), &target); + } else { + std::fs::copy(entry.path(), &target).unwrap(); + } + } + } + let dir = TempDir::new().unwrap(); + fluree_cmd(&dir).arg("init").assert().success(); + copy_dir( + &std::path::Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../fluree-db-api/tests/fixtures/prelink-annotations"), + &dir.path().join(".fluree/storage"), + ); + fluree_cmd(&dir) + .args(["info", "ann"]) + .assert() + .success() + .stdout(predicate::str::contains( + "predates RDF 1.2 triple-term links", + )); + + fluree_cmd(&dir) + .args(["create", "plain"]) + .assert() + .success(); + fluree_cmd(&dir) + .args([ + "insert", + "plain", + "-e", + "@prefix ex: . ex:a ex:b ex:c .", + ]) + .assert() + .success(); + fluree_cmd(&dir).args(["index", "plain"]).assert().success(); + fluree_cmd(&dir) + .args(["info", "plain"]) + .assert() + .success() + .stdout(predicate::str::contains("predates").not()); +} + #[test] fn export_round_trips_reifications_of_unasserted_triples() { let src = TempDir::new().unwrap(); diff --git a/fluree-db-ledger/src/lib.rs b/fluree-db-ledger/src/lib.rs index 2a4b97e2d4..27972b03b2 100644 --- a/fluree-db-ledger/src/lib.rs +++ b/fluree-db-ledger/src/lib.rs @@ -344,6 +344,13 @@ impl LedgerState { if snapshot.ledger_id != record.ledger_id { snapshot.ledger_id = record.ledger_id.clone(); } + if snapshot.needs_link_reindex { + tracing::warn!( + ledger_id = %record.ledger_id, + "index predates RDF 1.2 triple-term links; annotation reads fail until a full \ + reindex (`fluree reindex `)" + ); + } // Load novelty from commits since index_t let head_commit_id = match &record.commit_head_id {