From f464dfc7b34e5293cda3850d8a1a9ab8fc709f17 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 01:11:52 +0000 Subject: [PATCH 1/2] ogar-dir-core/ogar-ad/ogar-az: directory observation PoC Read-only encoding of observed Active Directory and Entra ID objects into a fixed 512-byte, NodeRow-shaped record: 128-bit source GUID (textual byte order, MS mixed-endian converted), scope GUID, an 8x16-bit OU HHTL with an explicit per-parent dictionary (no hashing, collision-free, reversible), schema family/version separate from the ABI version, out-of-line value pool, and a 64-byte GUID-pair edge record (SynchronizesTo, evidenced by onPremisesImmutableId). AD ingest via LDIF; Graph ingest via a page body with a schema-derived $select. No classid mint, no IAM semantics, no writes. Design and conflicts with the canonical NodeGuid/HHTL: docs/DIRECTORY-ADAPTERS-POC.md. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G22yT6htkcdyXsihxxXdrg --- Cargo.toml | 3 + crates/ogar-ad/Cargo.toml | 12 + crates/ogar-ad/src/ldif.rs | 87 ++++ crates/ogar-ad/src/lib.rs | 333 ++++++++++++++++ crates/ogar-ad/tests/fixtures/lab.ldif | 29 ++ crates/ogar-ad/tests/main.rs | 198 +++++++++ crates/ogar-az/Cargo.toml | 13 + crates/ogar-az/examples/az_ingest.rs | 52 +++ crates/ogar-az/src/lib.rs | 377 ++++++++++++++++++ crates/ogar-az/tests/fixtures/users_page.json | 39 ++ crates/ogar-az/tests/main.rs | 122 ++++++ crates/ogar-dir-core/Cargo.toml | 11 + crates/ogar-dir-core/src/base64.rs | 51 +++ crates/ogar-dir-core/src/dn.rs | 231 +++++++++++ crates/ogar-dir-core/src/edge.rs | 99 +++++ crates/ogar-dir-core/src/guid.rs | 170 ++++++++ crates/ogar-dir-core/src/hhtl.rs | 239 +++++++++++ crates/ogar-dir-core/src/lib.rs | 44 ++ crates/ogar-dir-core/src/pool.rs | 126 ++++++ crates/ogar-dir-core/src/record.rs | 320 +++++++++++++++ crates/ogar-dir-core/src/schema.rs | 110 +++++ crates/ogar-dir-core/tests/main.rs | 224 +++++++++++ docs/DIRECTORY-ADAPTERS-POC.md | 177 ++++++++ 23 files changed, 3067 insertions(+) create mode 100644 crates/ogar-ad/Cargo.toml create mode 100644 crates/ogar-ad/src/ldif.rs create mode 100644 crates/ogar-ad/src/lib.rs create mode 100644 crates/ogar-ad/tests/fixtures/lab.ldif create mode 100644 crates/ogar-ad/tests/main.rs create mode 100644 crates/ogar-az/Cargo.toml create mode 100644 crates/ogar-az/examples/az_ingest.rs create mode 100644 crates/ogar-az/src/lib.rs create mode 100644 crates/ogar-az/tests/fixtures/users_page.json create mode 100644 crates/ogar-az/tests/main.rs create mode 100644 crates/ogar-dir-core/Cargo.toml create mode 100644 crates/ogar-dir-core/src/base64.rs create mode 100644 crates/ogar-dir-core/src/dn.rs create mode 100644 crates/ogar-dir-core/src/edge.rs create mode 100644 crates/ogar-dir-core/src/guid.rs create mode 100644 crates/ogar-dir-core/src/hhtl.rs create mode 100644 crates/ogar-dir-core/src/lib.rs create mode 100644 crates/ogar-dir-core/src/pool.rs create mode 100644 crates/ogar-dir-core/src/record.rs create mode 100644 crates/ogar-dir-core/src/schema.rs create mode 100644 crates/ogar-dir-core/tests/main.rs create mode 100644 docs/DIRECTORY-ADAPTERS-POC.md diff --git a/Cargo.toml b/Cargo.toml index 579ca51..b66fbf7 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -37,6 +37,9 @@ members = [ "crates/ogar-ro", "crates/ogar-elk", "crates/ogar-osm", + "crates/ogar-dir-core", + "crates/ogar-ad", + "crates/ogar-az", ] [workspace.package] diff --git a/crates/ogar-ad/Cargo.toml b/crates/ogar-ad/Cargo.toml new file mode 100644 index 0000000..3fea74a --- /dev/null +++ b/crates/ogar-ad/Cargo.toml @@ -0,0 +1,12 @@ +[package] +name = "ogar-ad" +version.workspace = true +edition.workspace = true +license.workspace = true +repository.workspace = true +authors.workspace = true +rust-version.workspace = true +description = "Active Directory (AD DS / LDAP) source-native schema for OGAR directory observations: objectGUID identity, DN -> OU-HHTL location, an embedded versioned attribute table, and a read-only LDIF ingest path. No IAM, no writes." + +[dependencies] +ogar-dir-core = { path = "../ogar-dir-core" } diff --git a/crates/ogar-ad/src/ldif.rs b/crates/ogar-ad/src/ldif.rs new file mode 100644 index 0000000..575526d --- /dev/null +++ b/crates/ogar-ad/src/ldif.rs @@ -0,0 +1,87 @@ +//! Minimal read-only LDIF (RFC 2849) reader for `ldapsearch -LLL` / +//! `ldifde -f` output. +//! +//! Supported: `#` comments, a leading `version:` line, folded lines (a line +//! starting with one space continues the previous one), `attr: value`, +//! `attr:: base64` (binary or non-ASCII values — this is how `objectGUID` +//! arrives), blank-line entry separation. Rejected: `attr:< URL` and +//! change records (`changetype:`) — this is an observation reader. + +use crate::AdEntry; +use ogar_dir_core::base64; + +/// LDIF failure with 1-based line number. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum LdifError { + /// A line has no `:`. + NoColon(usize), + /// Invalid base64 payload. + BadBase64(usize), + /// `:<` URL value or `changetype` record. + Unsupported(usize), + /// An entry without a `dn:` first line. + MissingDn(usize), + /// A base64-encoded DN was not UTF-8. + DnNotUtf8(usize), +} + +/// Parse LDIF text into entries, in file order. +pub fn parse(text: &str) -> Result, LdifError> { + // Unfold first, remembering the starting line number of each logical line. + let mut logical: Vec<(usize, String)> = Vec::new(); + for (i, raw) in text.lines().enumerate() { + let line = raw.strip_suffix('\r').unwrap_or(raw); + if let Some(cont) = line.strip_prefix(' ') + && let Some((_, last)) = logical.last_mut() + { + last.push_str(cont); + continue; + } + logical.push((i + 1, line.to_string())); + } + + let mut out = Vec::new(); + let mut cur: Option = None; + for (ln, line) in logical { + if line.starts_with('#') { + continue; + } + if line.is_empty() { + if let Some(e) = cur.take() { + out.push(e); + } + continue; + } + let (name, rest) = line.split_once(':').ok_or(LdifError::NoColon(ln))?; + let value: Vec = if let Some(b) = rest.strip_prefix(':') { + base64::decode(b.trim()).ok_or(LdifError::BadBase64(ln))? + } else if rest.starts_with('<') { + return Err(LdifError::Unsupported(ln)); + } else { + rest.strip_prefix(' ').unwrap_or(rest).as_bytes().to_vec() + }; + if name.eq_ignore_ascii_case("version") && cur.is_none() && out.is_empty() { + continue; + } + if name.eq_ignore_ascii_case("changetype") { + return Err(LdifError::Unsupported(ln)); + } + match cur.as_mut() { + None => { + if !name.eq_ignore_ascii_case("dn") { + return Err(LdifError::MissingDn(ln)); + } + let dn = String::from_utf8(value).map_err(|_| LdifError::DnNotUtf8(ln))?; + cur = Some(AdEntry { + dn, + attrs: Vec::new(), + }); + } + Some(e) => e.attrs.push((name.to_string(), value)), + } + } + if let Some(e) = cur { + out.push(e); + } + Ok(out) +} diff --git a/crates/ogar-ad/src/lib.rs b/crates/ogar-ad/src/lib.rs new file mode 100644 index 0000000..0322b78 --- /dev/null +++ b/crates/ogar-ad/src/lib.rs @@ -0,0 +1,333 @@ +//! # ogar-ad — Active Directory as observed +//! +//! Encodes one LDAP entry into one [`DirRecord`]: +//! +//! * WHO = `objectGUID` (raw LDAP bytes, Microsoft mixed-endian → textual order) +//! * WHERE = the domain GUID (caller-supplied) + the OU chain of the DN as an +//! [`OuHhtl`](ogar_dir_core::OuHhtl), interned in a per-domain +//! [`OuDictionary`] +//! * WHAT = the attributes in [`SCHEMA_V1`], stored raw in a [`ValuePool`] +//! +//! It does not interpret `userAccountControl`, does not compute "enabled", +//! does not normalise addresses. Derived views are functions over the raw +//! record, never stored in it. +//! +//! Ingest is read-only: [`ldif::parse`] reads `ldapsearch`/`ldifde` output. + +pub mod ldif; + +use ogar_dir_core::dn::Dn; +use ogar_dir_core::record::{FLAG_DN_UNENCODED, FLAG_NON_OU_CONTAINER}; +use ogar_dir_core::{ + AttrDef, AttrKind, DirRecord, Guid128, OuDictionary, SchemaFamily, SchemaId, ValuePool, +}; + +/// Encoder schema version understood by this crate. +pub const SCHEMA_VERSION: u16 = 1; +/// This crate's schema id. +pub const SCHEMA: SchemaId = SchemaId { + family: SchemaFamily::AdDs, + version: SCHEMA_VERSION, +}; + +/// The embedded AD attribute table, v1. +/// +/// Selection: the attributes needed to (a) name an account in every form AD +/// and Exchange use (`sAMAccountName`, `userPrincipalName`, `mail`, +/// `mailNickname`, `proxyAddresses`, `targetAddress`), (b) say what it is +/// (`objectClass`, `objectSid`, `userAccountControl`), (c) say where it is +/// (`distinguishedName`) and (d) date it (`whenCreated`, `whenChanged`), plus +/// the display triplet. Group membership (`memberOf`/`member`) is a relation +/// and becomes edges later, never an inline list. Further Exchange attributes +/// (`msExch*`) are deferred until a consumer needs them. +pub const SCHEMA_V1: &[AttrDef] = &[ + AttrDef { + name: "distinguishedName", + slot: 0, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "objectSid", + slot: 1, + kind: AttrKind::Bytes, + since: 1, + }, + AttrDef { + name: "sAMAccountName", + slot: 2, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "userPrincipalName", + slot: 3, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "objectClass", + slot: 4, + kind: AttrKind::MultiStr, + since: 1, + }, + AttrDef { + name: "displayName", + slot: 5, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "givenName", + slot: 6, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "sn", + slot: 7, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "mail", + slot: 8, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "mailNickname", + slot: 9, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "proxyAddresses", + slot: 10, + kind: AttrKind::MultiStr, + since: 1, + }, + AttrDef { + name: "targetAddress", + slot: 11, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "whenCreated", + slot: 12, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "whenChanged", + slot: 13, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "userAccountControl", + slot: 0, + kind: AttrKind::U32, + since: 1, + }, +]; + +/// AD object kinds (family-scoped codes in `object_kind`). +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u16)] +pub enum AdKind { + /// None of the below. + Other = 0, + /// `user` (and not `computer`). + User = 1, + /// `group`. + Group = 2, + /// `computer` (which is also a `user` subclass in AD). + Computer = 3, + /// `contact`. + Contact = 4, + /// `organizationalUnit`. + OrganizationalUnit = 5, +} + +impl AdKind { + /// From `objectClass` values; most specific class wins. + pub fn from_object_class>(classes: &[S]) -> Self { + let has = |c: &str| classes.iter().any(|x| x.as_ref().eq_ignore_ascii_case(c)); + if has("computer") { + Self::Computer + } else if has("user") { + Self::User + } else if has("group") { + Self::Group + } else if has("contact") { + Self::Contact + } else if has("organizationalUnit") { + Self::OrganizationalUnit + } else { + Self::Other + } + } +} + +/// One LDAP entry as read: DN plus attribute values in source order. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct AdEntry { + /// The entry DN. + pub dn: String, + /// `(attribute name as written, raw value bytes)`, repeated for multi-values. + pub attrs: Vec<(String, Vec)>, +} + +impl AdEntry { + /// All values of an attribute (case-insensitive name; `;options` ignored). + pub fn values(&self, name: &str) -> Vec<&[u8]> { + self.attrs + .iter() + .filter(|(n, _)| n.split(';').next().unwrap_or(n).eq_ignore_ascii_case(name)) + .map(|(_, v)| v.as_slice()) + .collect() + } +} + +/// Encoding failure (the entry is skipped, never half-written). +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum AdError { + /// No `objectGUID`, or not 16 bytes. + MissingObjectGuid, + /// A string attribute was not UTF-8. + NotUtf8(&'static str), + /// A single-valued attribute had several values. + MultipleValues(&'static str), + /// A numeric attribute did not parse. + BadNumber(&'static str), + /// Pool or slot failure. + Storage(String), +} + +impl std::fmt::Display for AdError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "ad encode: {self:?}") + } +} +impl std::error::Error for AdError {} + +/// Result of encoding one entry. +#[derive(Debug, Clone)] +pub struct Encoded { + /// The fixed record. + pub record: DirRecord, + /// Attribute names present in the entry but not in the schema. They are + /// not stored and cannot affect the fixed ABI. + pub ignored: Vec, +} + +fn identity_or_dn(name: &str) -> bool { + name.eq_ignore_ascii_case("objectGUID") || name.eq_ignore_ascii_case("dn") +} + +/// Encode one entry. `domain` is the AD domain's GUID (the naming-context +/// head's `objectGUID`); `dict` must be that domain's OU dictionary. +pub fn encode( + entry: &AdEntry, + domain: Guid128, + dict: &mut OuDictionary, + pool: &mut ValuePool, + observed_at_ms: i64, +) -> Result { + let guid_vals = entry.values("objectGUID"); + let node = match guid_vals.as_slice() { + [g] => Guid128::from_ms_bytes(g).map_err(|_| AdError::MissingObjectGuid)?, + _ => return Err(AdError::MissingObjectGuid), + }; + + let classes: Vec = entry + .values("objectClass") + .iter() + .map(|v| String::from_utf8_lossy(v).into_owned()) + .collect(); + let kind = AdKind::from_object_class(&classes) as u16; + let mut rec = DirRecord::new(SCHEMA, kind, node, domain, observed_at_ms); + let st = |e: &dyn std::fmt::Debug| AdError::Storage(format!("{e:?}")); + + for def in SCHEMA_V1 { + let mut vals = entry.values(def.name); + // The DN line is authoritative for distinguishedName when the + // attribute itself was not requested. + if def.name == "distinguishedName" && vals.is_empty() && !entry.dn.is_empty() { + vals = vec![entry.dn.as_bytes()]; + } + if vals.is_empty() { + continue; + } + match def.kind { + AttrKind::Str | AttrKind::Bytes => { + let [v] = vals.as_slice() else { + return Err(AdError::MultipleValues(def.name)); + }; + if def.kind == AttrKind::Str && std::str::from_utf8(v).is_err() { + return Err(AdError::NotUtf8(def.name)); + } + let r = pool.push(v).map_err(|e| st(&e))?; + rec.set_str(def.slot as usize, r).map_err(|e| st(&e))?; + } + AttrKind::MultiStr => { + if vals.iter().any(|v| std::str::from_utf8(v).is_err()) { + return Err(AdError::NotUtf8(def.name)); + } + let r = pool.push_multi(&vals).map_err(|e| st(&e))?; + rec.set_str(def.slot as usize, r).map_err(|e| st(&e))?; + } + AttrKind::U32 | AttrKind::Bool => { + let [v] = vals.as_slice() else { + return Err(AdError::MultipleValues(def.name)); + }; + let s = std::str::from_utf8(v).map_err(|_| AdError::BadNumber(def.name))?; + // LDAP integers are signed decimal; UAC fits in i32/u32. + let n = s + .trim() + .parse::() + .ok() + .and_then(|n| { + u32::try_from(n) + .ok() + .or_else(|| i32::try_from(n).ok().map(|i| i as u32)) + }) + .ok_or(AdError::BadNumber(def.name))?; + rec.set_num(def.slot as usize, n).map_err(|e| st(&e))?; + } + } + } + + // WHERE: the OU chain of the DN. Never the leaf, never DC=. + if !entry.dn.is_empty() { + match Dn::parse(&entry.dn) { + Ok(dn) => { + if dn.has_non_ou_container() { + rec.add_flags(FLAG_NON_OU_CONTAINER); + } + match dict.intern(&dn.ou_path_root_first()) { + Ok(h) => rec.set_ou_hhtl(h), + Err(_) => rec.add_flags(FLAG_DN_UNENCODED), + } + } + Err(_) => rec.add_flags(FLAG_DN_UNENCODED), + } + } + + let mut ignored: Vec = entry + .attrs + .iter() + .map(|(n, _)| n.clone()) + .filter(|n| { + let base = n.split(';').next().unwrap_or(n); + !identity_or_dn(base) && !SCHEMA_V1.iter().any(|d| d.name.eq_ignore_ascii_case(base)) + }) + .collect(); + ignored.dedup(); + Ok(Encoded { + record: rec, + ignored, + }) +} diff --git a/crates/ogar-ad/tests/fixtures/lab.ldif b/crates/ogar-ad/tests/fixtures/lab.ldif new file mode 100644 index 0000000..317a59d --- /dev/null +++ b/crates/ogar-ad/tests/fixtures/lab.ldif @@ -0,0 +1,29 @@ +# Synthetic lab export (ldapsearch -LLL style). No real persons. +version: 1 + +dn:: Q049RXJpa2EgTcO8bGxlcixPVT1FeGNoYW5nZSxPVT1JbmZyYXN0cnVjdHVyZSxPVT1TdHV0dGdhcnQsREM9ZXhhbXBsZSxEQz1kZQ== +objectClass: top +objectClass: person +objectClass: organizationalPerson +objectClass: user +objectGUID:: 4AQlP4lP0xGaDAMF6CwzAQ== +sAMAccountName: emueller +userPrincipalName: erika.mueller@example.de +displayName:: RXJpa2EgTcO8bGxlcg== +mail: erika.mueller@example.de +proxyAddresses: SMTP:erika.mueller@example.de +proxyAddresses: smtp:emueller@example.mail.onmicrosoft.com +userAccountControl: 512 +whenChanged: 20261001120000.0Z +msDS-SomeFutureAttribute: unknown to schema v1 +thumbnailPhoto:: AAEC + +dn: CN=Svc Backup,CN=Users,DC=example,DC=de +objectClass: top +objectClass: user +objectClass: computer +objectGUID:: 1MOyoQAAAECAAAAAAAC+7w== +sAMAccountName: SVC-BACKUP$ +description: this attribute is not in v1 and the + value is folded over two lines +userAccountControl: 4096 diff --git a/crates/ogar-ad/tests/main.rs b/crates/ogar-ad/tests/main.rs new file mode 100644 index 0000000..582c521 --- /dev/null +++ b/crates/ogar-ad/tests/main.rs @@ -0,0 +1,198 @@ +use ogar_ad::{AdEntry, AdKind, SCHEMA, SCHEMA_V1, encode, ldif}; +use ogar_dir_core::record::{FLAG_NON_OU_CONTAINER, RECORD_BYTES}; +use ogar_dir_core::{DirRecord, Guid128, OuDictionary, ValuePool, schema}; + +const DOMAIN: &str = "d0d0d0d0-0000-4000-8000-000000000001"; + +fn domain() -> Guid128 { + Guid128::parse(DOMAIN).unwrap() +} + +fn slot(name: &str) -> usize { + SCHEMA_V1.iter().find(|d| d.name == name).unwrap().slot as usize +} + +#[test] +fn schema_table_is_well_formed() { + schema::validate(SCHEMA_V1).unwrap(); +} + +#[test] +fn ldif_lab_export_encodes() { + let text = include_str!("fixtures/lab.ldif"); + let entries = ldif::parse(text).unwrap(); + assert_eq!(entries.len(), 2); + let (mut dict, mut pool) = (OuDictionary::new(), ValuePool::new()); + + let user = encode(&entries[0], domain(), &mut dict, &mut pool, 1).unwrap(); + let r = user.record; + assert_eq!(r.schema(), SCHEMA); + assert_eq!(r.object_kind(), AdKind::User as u16); + // objectGUID arrived as raw MS bytes and lands in textual order. + assert_eq!( + r.node_guid().to_string(), + "3f2504e0-4f89-11d3-9a0c-0305e82c3301" + ); + assert_eq!(r.scope_guid(), domain()); + let ou = r.ou_hhtl().unwrap(); + assert_eq!( + dict.explain(&ou).unwrap(), + ["Stuttgart", "Infrastructure", "Exchange"] + ); + assert_eq!( + pool.get(r.str_ref(slot("displayName")).unwrap()).unwrap(), + "Erika Müller".as_bytes() + ); + assert_eq!( + pool.get_multi(r.str_ref(slot("proxyAddresses")).unwrap()) + .unwrap(), + vec![ + &b"SMTP:erika.mueller@example.de"[..], + b"smtp:emueller@example.mail.onmicrosoft.com" + ] + ); + assert_eq!(r.num(0), Some(512)); + // the DN line lands raw in the distinguishedName slot + assert!( + std::str::from_utf8( + pool.get(r.str_ref(slot("distinguishedName")).unwrap()) + .unwrap() + ) + .unwrap() + .starts_with("CN=Erika Müller,OU=Exchange") + ); + + let comp = encode(&entries[1], domain(), &mut dict, &mut pool, 1).unwrap(); + assert_eq!(comp.record.object_kind(), AdKind::Computer as u16); + // CN=Users is a container, not an OU: depth 0, but flagged as such + assert_eq!(comp.record.ou_hhtl().unwrap().depth(), 0); + assert_ne!(comp.record.flags() & FLAG_NON_OU_CONTAINER, 0); + assert_eq!(comp.ignored, ["description"]); +} + +fn entry(guid_ms_b64: &str, dn: &str) -> AdEntry { + AdEntry { + dn: dn.into(), + attrs: vec![ + ( + "objectGUID".into(), + ogar_dir_core::base64::decode(guid_ms_b64).unwrap(), + ), + ("objectClass".into(), b"user".to_vec()), + ("sAMAccountName".into(), b"emueller".to_vec()), + ], + } +} + +// 6. Moving the object between OUs changes HHTL, preserves NodeGuid. +#[test] +fn inv06_ou_move_changes_hhtl_not_identity() { + let (mut dict, mut pool) = (OuDictionary::new(), ValuePool::new()); + let g = "4AQlP4lP0xGaDAMF6CwzAQ=="; + let before = encode( + &entry(g, "CN=E M,OU=Exchange,OU=Stuttgart,DC=example,DC=de"), + domain(), + &mut dict, + &mut pool, + 1, + ) + .unwrap() + .record; + let after = encode( + &entry(g, "CN=E M,OU=Finance,OU=Berlin,DC=example,DC=de"), + domain(), + &mut dict, + &mut pool, + 2, + ) + .unwrap() + .record; + assert_eq!(before.node_guid(), after.node_guid()); + assert_ne!(before.ou_hhtl(), after.ou_hhtl()); +} + +// 7. Renaming the leaf CN preserves HHTL and NodeGuid. +#[test] +fn inv07_leaf_rename_preserves_hhtl_and_identity() { + let (mut dict, mut pool) = (OuDictionary::new(), ValuePool::new()); + let g = "4AQlP4lP0xGaDAMF6CwzAQ=="; + let a = encode( + &entry(g, "CN=Erika Mustermann,OU=Exchange,DC=example,DC=de"), + domain(), + &mut dict, + &mut pool, + 1, + ) + .unwrap() + .record; + let b = encode( + &entry(g, "CN=Erika Musterfrau,OU=Exchange,DC=example,DC=de"), + domain(), + &mut dict, + &mut pool, + 2, + ) + .unwrap() + .record; + assert_eq!(a.node_guid(), b.node_guid()); + assert_eq!(a.ou_hhtl(), b.ou_hhtl()); + assert_eq!(dict.len(), 1, "no new segment for a leaf rename"); +} + +// 12. Unknown/new source attributes do not corrupt the fixed ABI. +#[test] +fn inv12_unknown_attributes_do_not_touch_the_record() { + let (mut d1, mut p1) = (OuDictionary::new(), ValuePool::new()); + let (mut d2, mut p2) = (OuDictionary::new(), ValuePool::new()); + let g = "4AQlP4lP0xGaDAMF6CwzAQ=="; + let plain = entry(g, "CN=E,OU=Exchange,DC=x"); + let mut noisy = plain.clone(); + for i in 0..200 { + noisy + .attrs + .push((format!("msDS-Future{i}"), vec![0xFF; 300])); + } + noisy + .attrs + .push(("member;range=0-1499".into(), b"CN=a,DC=x".to_vec())); + let a = encode(&plain, domain(), &mut d1, &mut p1, 7).unwrap(); + let b = encode(&noisy, domain(), &mut d2, &mut p2, 7).unwrap(); + assert_eq!(a.record.as_bytes(), b.record.as_bytes()); + assert_eq!( + p1.as_bytes(), + p2.as_bytes(), + "unknown values never reach the pool" + ); + assert_eq!(b.ignored.len(), 201); + assert!(b.record.reserved_is_zero()); + assert_eq!(b.record.as_bytes().len(), RECORD_BYTES); + assert!(DirRecord::from_bytes(b.record.as_bytes()).is_ok()); +} + +#[test] +fn missing_or_malformed_object_guid_is_refused() { + let (mut d, mut p) = (OuDictionary::new(), ValuePool::new()); + let mut e = entry("4AQlP4lP0xGaDAMF6CwzAQ==", "CN=E,DC=x"); + e.attrs.retain(|(n, _)| n != "objectGUID"); + assert!(encode(&e, domain(), &mut d, &mut p, 0).is_err()); + e.attrs.push(("objectGUID".into(), vec![1, 2, 3])); + assert!(encode(&e, domain(), &mut d, &mut p, 0).is_err()); +} + +#[test] +fn too_deep_dn_is_flagged_not_truncated() { + let (mut d, mut p) = (OuDictionary::new(), ValuePool::new()); + let dn = "CN=E,OU=1,OU=2,OU=3,OU=4,OU=5,OU=6,OU=7,OU=8,OU=9,DC=x"; + let r = encode( + &entry("4AQlP4lP0xGaDAMF6CwzAQ==", dn), + domain(), + &mut d, + &mut p, + 0, + ) + .unwrap() + .record; + assert_eq!(r.ou_hhtl(), None); + assert_ne!(r.flags() & ogar_dir_core::record::FLAG_DN_UNENCODED, 0); + assert!(d.is_empty(), "nothing interned for a refused path"); +} diff --git a/crates/ogar-az/Cargo.toml b/crates/ogar-az/Cargo.toml new file mode 100644 index 0000000..c4f7cde --- /dev/null +++ b/crates/ogar-az/Cargo.toml @@ -0,0 +1,13 @@ +[package] +name = "ogar-az" +version.workspace = true +edition.workspace = true +license.workspace = true +repository.workspace = true +authors.workspace = true +rust-version.workspace = true +description = "Microsoft Entra ID / Microsoft Graph source-native schema for OGAR directory observations: Graph object id identity, tenant scope, onPremisesDistinguishedName -> OU-HHTL evidence, a schema-derived $select, read-only page ingest, and SynchronizesTo edges from onPremisesImmutableId. No IAM, no writes." + +[dependencies] +ogar-dir-core = { path = "../ogar-dir-core" } +serde_json = "1.0" diff --git a/crates/ogar-az/examples/az_ingest.rs b/crates/ogar-az/examples/az_ingest.rs new file mode 100644 index 0000000..cc9f635 --- /dev/null +++ b/crates/ogar-az/examples/az_ingest.rs @@ -0,0 +1,52 @@ +//! Read-only lab ingest of a Microsoft Graph `users` page. +//! +//! This example performs no HTTP. Fetch with any client holding a token that +//! has `User.Read.All` (application or delegated), e.g.: +//! +//! ```sh +//! URL=$(cargo run -q -p ogar-az --example az_ingest -- --url) +//! curl -s -H "Authorization: Bearer $TOKEN" "$URL" > page1.json +//! cargo run -p ogar-az --example az_ingest -- page1.json +//! ``` +//! +//! Follow `next_link` for further pages with the same dictionary and pool. + +use ogar_dir_core::{OuDictionary, ValuePool}; + +fn main() { + let args: Vec = std::env::args().skip(1).collect(); + if args.first().map(String::as_str) == Some("--url") { + println!("{}", ogar_az::users_url(ogar_az::SCHEMA_VERSION, 100)); + return; + } + let [path, tenant] = args.as_slice() else { + eprintln!("usage: az_ingest | --url"); + std::process::exit(2); + }; + let tenant = ogar_dir_core::Guid128::parse(tenant).expect("tenant must be a GUID"); + let body = std::fs::read_to_string(path).expect("read page"); + let (mut dict, mut pool) = (OuDictionary::new(), ValuePool::new()); + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_millis() as i64) + .unwrap_or(0); + let page = ogar_az::ingest_page(&body, tenant, &mut dict, &mut pool, now).expect("ingest"); + for r in &page.records { + let ou = r.ou_hhtl().and_then(|h| dict.explain(&h).map(|n| (h, n))); + println!( + "{} upn={:?} immutableId={:?} syncEnabled={:?} ou={:?}", + r.node_guid(), + ogar_az::attr_str(r, &pool, "userPrincipalName"), + ogar_az::attr_str(r, &pool, "onPremisesImmutableId"), + r.num(1), + ou, + ); + } + println!( + "records={} pool_bytes={} ignored={:?} next_link={:?}", + page.records.len(), + pool.as_bytes().len(), + page.ignored, + page.next_link + ); +} diff --git a/crates/ogar-az/src/lib.rs b/crates/ogar-az/src/lib.rs new file mode 100644 index 0000000..fd7a43d --- /dev/null +++ b/crates/ogar-az/src/lib.rs @@ -0,0 +1,377 @@ +//! # ogar-az — Entra ID as observed through Microsoft Graph +//! +//! Encodes one Graph `user` object into one [`DirRecord`]: +//! +//! * WHO = Graph `id` (the Entra object id). **Never** the AD `objectGUID`: +//! a synchronized identity is two nodes, related by a +//! [`DirEdge`] ([`sync_edges`]), not one. +//! * WHERE = tenant id + (when present) the OU chain of +//! `onPremisesDistinguishedName` as an [`OuHhtl`](ogar_dir_core::OuHhtl). +//! That HHTL is *evidence about the on-premises location*, not a Graph +//! location. Pass the on-premises domain's [`OuDictionary`] (the same one +//! `ogar-ad` uses) and the AD and AZ HHTLs of one OU coincide. +//! * WHAT = the attributes in [`SCHEMA_V1`], raw. +//! +//! The `$select` projection is derived from the schema ([`select_query`]), +//! never maintained as a separate list. Read-only: nothing here writes to +//! Graph; this crate does not even perform HTTP (see `examples/az_ingest.rs`). + +use ogar_dir_core::dn::Dn; +use ogar_dir_core::edge::{DirEdge, EdgeEvidence, EdgeKind}; +use ogar_dir_core::record::{FLAG_DN_UNENCODED, FLAG_NON_OU_CONTAINER}; +use ogar_dir_core::{ + AttrDef, AttrKind, DirRecord, Guid128, OuDictionary, SchemaFamily, SchemaId, ValuePool, base64, +}; +use serde_json::Value; +use std::collections::HashSet; + +/// Encoder schema version understood by this crate. +pub const SCHEMA_VERSION: u16 = 1; +/// This crate's schema id. +pub const SCHEMA: SchemaId = SchemaId { + family: SchemaFamily::MsGraph, + version: SCHEMA_VERSION, +}; + +/// The embedded Graph `user` attribute table, v1. +/// +/// Selection: naming (`userPrincipalName`, `mail`, `mailNickname`, +/// `proxyAddresses`, display triplet), state (`accountEnabled`, `userType`), +/// HR-ish context (`employeeId`, `department`, `companyName`, +/// `officeLocation`), and every `onPremises*` property that is *evidence about +/// the AD counterpart* (DN, sAMAccountName, domain, immutable id, SID, sync +/// flag, last sync). Licences, sign-in activity and group membership are out +/// of scope (the last becomes edges). +pub const SCHEMA_V1: &[AttrDef] = &[ + AttrDef { + name: "userPrincipalName", + slot: 0, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "displayName", + slot: 1, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "givenName", + slot: 2, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "surname", + slot: 3, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "mail", + slot: 4, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "mailNickname", + slot: 5, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "proxyAddresses", + slot: 6, + kind: AttrKind::MultiStr, + since: 1, + }, + AttrDef { + name: "employeeId", + slot: 7, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "department", + slot: 8, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "companyName", + slot: 9, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "officeLocation", + slot: 10, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "userType", + slot: 11, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "onPremisesDistinguishedName", + slot: 12, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "onPremisesSamAccountName", + slot: 13, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "onPremisesDomainName", + slot: 14, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "onPremisesImmutableId", + slot: 15, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "onPremisesSecurityIdentifier", + slot: 16, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "onPremisesLastSyncDateTime", + slot: 17, + kind: AttrKind::Str, + since: 1, + }, + AttrDef { + name: "accountEnabled", + slot: 0, + kind: AttrKind::Bool, + since: 1, + }, + AttrDef { + name: "onPremisesSyncEnabled", + slot: 1, + kind: AttrKind::Bool, + since: 1, + }, +]; + +/// AZ object kinds (family-scoped codes). Only users are ingested in v1. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u16)] +pub enum AzKind { + /// A Graph `user`. + User = 1, +} + +/// `$select` value for schema `version`: `id` plus every attribute defined at +/// or before that version, in table order. +pub fn select_query(version: u16) -> String { + std::iter::once("id") + .chain( + SCHEMA_V1 + .iter() + .filter(|a| a.since <= version) + .map(|a| a.name), + ) + .collect::>() + .join(",") +} + +/// The read-only list URL for users (v1.0 endpoint). +pub fn users_url(version: u16, page_size: u16) -> String { + format!( + "https://graph.microsoft.com/v1.0/users?$select={}&$top={}", + select_query(version), + page_size.clamp(1, 999) + ) +} + +/// Encoding failure. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum AzError { + /// The page is not `{ "value": [...] }`. + NotAPage, + /// Object `index` has no parseable `id`. + BadId(usize), + /// Object `index`, attribute `name` has the wrong JSON type. + BadType(usize, &'static str), + /// Pool or slot failure. + Storage(String), +} + +impl std::fmt::Display for AzError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "az encode: {self:?}") + } +} +impl std::error::Error for AzError {} + +/// One ingested page. +#[derive(Debug, Clone)] +pub struct Page { + /// Records, in page order. + pub records: Vec, + /// `@odata.nextLink`, if Graph returned one (follow it to continue). + pub next_link: Option, + /// Distinct property names seen but not in the schema (not stored). + pub ignored: Vec, +} + +/// Encode one Graph user object. +pub fn encode_user( + obj: &serde_json::Map, + index: usize, + tenant: Guid128, + onprem_dict: &mut OuDictionary, + pool: &mut ValuePool, + observed_at_ms: i64, +) -> Result { + let id = obj + .get("id") + .and_then(Value::as_str) + .and_then(|s| Guid128::parse(s).ok()) + .ok_or(AzError::BadId(index))?; + let mut rec = DirRecord::new(SCHEMA, AzKind::User as u16, id, tenant, observed_at_ms); + let st = |e: &dyn std::fmt::Debug| AzError::Storage(format!("{e:?}")); + + for def in SCHEMA_V1 { + // Graph returns `null` for unset properties: that is absence. + let Some(v) = obj.get(def.name).filter(|v| !v.is_null()) else { + continue; + }; + let bad = || AzError::BadType(index, def.name); + match def.kind { + AttrKind::Str | AttrKind::Bytes => { + let s = v.as_str().ok_or_else(bad)?; + let r = pool.push(s.as_bytes()).map_err(|e| st(&e))?; + rec.set_str(def.slot as usize, r).map_err(|e| st(&e))?; + } + AttrKind::MultiStr => { + let arr = v.as_array().ok_or_else(bad)?; + let vals: Vec<&str> = arr + .iter() + .map(|x| x.as_str().ok_or_else(bad)) + .collect::>()?; + let r = pool.push_multi(&vals).map_err(|e| st(&e))?; + rec.set_str(def.slot as usize, r).map_err(|e| st(&e))?; + } + AttrKind::Bool => { + let b = v.as_bool().ok_or_else(bad)?; + rec.set_num(def.slot as usize, b as u32) + .map_err(|e| st(&e))?; + } + AttrKind::U32 => { + let n = v + .as_u64() + .and_then(|n| u32::try_from(n).ok()) + .ok_or_else(bad)?; + rec.set_num(def.slot as usize, n).map_err(|e| st(&e))?; + } + } + } + + if let Some(dn) = obj + .get("onPremisesDistinguishedName") + .and_then(Value::as_str) + { + match Dn::parse(dn) { + Ok(dn) => { + if dn.has_non_ou_container() { + rec.add_flags(FLAG_NON_OU_CONTAINER); + } + match onprem_dict.intern(&dn.ou_path_root_first()) { + Ok(h) => rec.set_ou_hhtl(h), + Err(_) => rec.add_flags(FLAG_DN_UNENCODED), + } + } + Err(_) => rec.add_flags(FLAG_DN_UNENCODED), + } + } + Ok(rec) +} + +/// Ingest one Graph list-page body (`GET /users?$select=…`). +pub fn ingest_page( + body: &str, + tenant: Guid128, + onprem_dict: &mut OuDictionary, + pool: &mut ValuePool, + observed_at_ms: i64, +) -> Result { + let v: Value = serde_json::from_str(body).map_err(|_| AzError::NotAPage)?; + let items = v + .get("value") + .and_then(Value::as_array) + .ok_or(AzError::NotAPage)?; + let mut records = Vec::with_capacity(items.len()); + let mut ignored: Vec = Vec::new(); + for (i, item) in items.iter().enumerate() { + let obj = item.as_object().ok_or(AzError::BadId(i))?; + for k in obj.keys() { + if k != "id" && !SCHEMA_V1.iter().any(|d| d.name == k) && !ignored.contains(k) { + ignored.push(k.clone()); + } + } + records.push(encode_user( + obj, + i, + tenant, + onprem_dict, + pool, + observed_at_ms, + )?); + } + ignored.sort(); + let next_link = v + .get("@odata.nextLink") + .and_then(Value::as_str) + .map(str::to_string); + Ok(Page { + records, + next_link, + ignored, + }) +} + +/// Read a string slot by attribute name. +pub fn attr_str<'p>(rec: &DirRecord, pool: &'p ValuePool, name: &str) -> Option<&'p str> { + let def = SCHEMA_V1 + .iter() + .find(|d| d.name == name && d.kind.is_pooled())?; + std::str::from_utf8(pool.get(rec.str_ref(def.slot as usize)?)?).ok() +} + +/// `SynchronizesTo` edges for AZ records whose `onPremisesImmutableId` +/// base64-decodes to the `objectGUID` of an **observed** AD object. +/// +/// This is evidence, not a decision: anchors that are not a 16-byte GUID +/// (e.g. custom source anchors) yield no edge, and an immutable id matching no +/// observed AD object yields no dangling edge. The two nodes stay distinct. +pub fn sync_edges( + az: &[DirRecord], + pool: &ValuePool, + observed_ad: &HashSet, +) -> Vec { + az.iter() + .filter_map(|r| { + let raw = base64::decode(attr_str(r, pool, "onPremisesImmutableId")?)?; + let ad = Guid128::from_ms_bytes(&raw).ok()?; + observed_ad.contains(&ad).then_some(DirEdge { + kind: EdgeKind::SynchronizesTo, + evidence: EdgeEvidence::ImmutableIdIsObjectGuid, + src: (SchemaFamily::AdDs, ad), + dst: (SchemaFamily::MsGraph, r.node_guid()), + }) + }) + .collect() +} diff --git a/crates/ogar-az/tests/fixtures/users_page.json b/crates/ogar-az/tests/fixtures/users_page.json new file mode 100644 index 0000000..1f205f3 --- /dev/null +++ b/crates/ogar-az/tests/fixtures/users_page.json @@ -0,0 +1,39 @@ +{ + "@odata.context": "https://graph.microsoft.com/v1.0/$metadata#users(id,userPrincipalName)", + "@odata.nextLink": "https://graph.microsoft.com/v1.0/users?$select=id&$top=2&$skiptoken=X", + "value": [ + { + "id": "7a3c9e11-2b44-4c6d-9e8f-0123456789ab", + "userPrincipalName": "erika.mueller@example.de", + "displayName": "Erika Müller", + "givenName": "Erika", + "surname": "Müller", + "mail": "erika.mueller@example.de", + "mailNickname": "emueller", + "proxyAddresses": ["SMTP:erika.mueller@example.de", "smtp:emueller@example.mail.onmicrosoft.com"], + "accountEnabled": true, + "employeeId": null, + "department": "Infrastructure", + "companyName": null, + "officeLocation": "Stuttgart", + "userType": "Member", + "onPremisesSyncEnabled": true, + "onPremisesDistinguishedName": "CN=Erika Müller,OU=Exchange,OU=Infrastructure,OU=Stuttgart,DC=example,DC=de", + "onPremisesSamAccountName": "emueller", + "onPremisesDomainName": "example.de", + "onPremisesImmutableId": "4AQlP4lP0xGaDAMF6CwzAQ==", + "onPremisesSecurityIdentifier": "S-1-5-21-1-2-3-1104", + "onPremisesLastSyncDateTime": "2026-10-01T12:00:00Z", + "someNewGraphProperty": {"nested": true} + }, + { + "id": "00000000-0000-0001-0123-456789abcdef", + "userPrincipalName": "cloud.only@example.onmicrosoft.com", + "accountEnabled": false, + "onPremisesSyncEnabled": null, + "onPremisesDistinguishedName": null, + "onPremisesImmutableId": null, + "proxyAddresses": [] + } + ] +} diff --git a/crates/ogar-az/tests/main.rs b/crates/ogar-az/tests/main.rs new file mode 100644 index 0000000..eddd3ec --- /dev/null +++ b/crates/ogar-az/tests/main.rs @@ -0,0 +1,122 @@ +use ogar_az::{SCHEMA, SCHEMA_V1, attr_str, ingest_page, select_query, sync_edges, users_url}; +use ogar_dir_core::edge::{EdgeEvidence, EdgeKind}; +use ogar_dir_core::{Dn, Guid128, OuDictionary, SchemaFamily, ValuePool, schema}; +use std::collections::HashSet; + +const TENANT: &str = "c0ffee00-1234-4abc-8def-000000000042"; +const PAGE: &str = include_str!("fixtures/users_page.json"); +/// The AD objectGUID that `onPremisesImmutableId` in the fixture encodes. +const AD_GUID: &str = "3f2504e0-4f89-11d3-9a0c-0305e82c3301"; + +fn tenant() -> Guid128 { + Guid128::parse(TENANT).unwrap() +} + +#[test] +fn schema_table_is_well_formed() { + schema::validate(SCHEMA_V1).unwrap(); +} + +#[test] +fn select_is_derived_from_the_schema() { + let q = select_query(1); + let names: Vec<&str> = q.split(',').collect(); + assert_eq!(names[0], "id"); + assert_eq!(names.len(), 1 + SCHEMA_V1.len()); + for d in SCHEMA_V1 { + assert!(names.contains(&d.name), "{} missing from $select", d.name); + } + // a version before any attribute existed selects only the identity + assert_eq!(select_query(0), "id"); + assert!(users_url(1, 5000).ends_with("&$top=999")); + assert!(!users_url(1, 100).contains(' ')); +} + +#[test] +fn graph_page_ingests() { + let (mut dict, mut pool) = (OuDictionary::new(), ValuePool::new()); + let page = ingest_page(PAGE, tenant(), &mut dict, &mut pool, 99).unwrap(); + assert_eq!(page.records.len(), 2); + assert!(page.next_link.as_deref().unwrap().contains("$skiptoken")); + assert_eq!(page.ignored, ["someNewGraphProperty"]); + + let synced = &page.records[0]; + assert_eq!(synced.schema(), SCHEMA); + assert_eq!( + synced.node_guid().to_string(), + "7a3c9e11-2b44-4c6d-9e8f-0123456789ab" + ); + assert_eq!(synced.scope_guid(), tenant()); + assert_eq!(attr_str(synced, &pool, "displayName"), Some("Erika Müller")); + assert_eq!( + attr_str(synced, &pool, "employeeId"), + None, + "null is absence" + ); + assert_eq!(synced.num(0), Some(1)); // accountEnabled + assert_eq!(synced.num(1), Some(1)); // onPremisesSyncEnabled + let ou = synced.ou_hhtl().unwrap(); + assert_eq!( + dict.explain(&ou).unwrap(), + ["Stuttgart", "Infrastructure", "Exchange"] + ); + + let cloud = &page.records[1]; + assert_eq!(cloud.num(0), Some(0)); // accountEnabled = false is present + assert_eq!(cloud.num(1), None); // syncEnabled null = unknown, not false + assert_eq!(cloud.ou_hhtl(), None); + assert_eq!(attr_str(cloud, &pool, "onPremisesImmutableId"), None); +} + +// 9. AD and AZ nodes with different GUIDs relate without identity collapse. +#[test] +fn inv09_sync_edge_relates_two_distinct_nodes() { + let (mut dict, mut pool) = (OuDictionary::new(), ValuePool::new()); + let page = ingest_page(PAGE, tenant(), &mut dict, &mut pool, 99).unwrap(); + let ad = Guid128::parse(AD_GUID).unwrap(); + + // not observed in AD -> no dangling edge + assert!(sync_edges(&page.records, &pool, &HashSet::new()).is_empty()); + + let edges = sync_edges(&page.records, &pool, &HashSet::from([ad])); + assert_eq!(edges.len(), 1); + let e = edges[0]; + assert_eq!(e.kind, EdgeKind::SynchronizesTo); + assert_eq!(e.evidence, EdgeEvidence::ImmutableIdIsObjectGuid); + assert_eq!(e.src, (SchemaFamily::AdDs, ad)); + assert_eq!(e.dst, (SchemaFamily::MsGraph, page.records[0].node_guid())); + assert_ne!(e.src.1, e.dst.1, "two nodes, never merged"); + // The AZ record still carries its own id, not the AD one. + assert_ne!(page.records[0].node_guid(), ad); +} + +// With the on-prem domain's dictionary shared, the AZ evidence and the AD +// observation of one OU produce the identical HHTL. +#[test] +fn shared_onprem_dictionary_aligns_ad_and_az_hhtl() { + let mut dict = OuDictionary::new(); + let ad_dn = + Dn::parse("CN=Someone Else,OU=Exchange,OU=Infrastructure,OU=Stuttgart,DC=example,DC=de") + .unwrap(); + let ad_h = dict.intern(&ad_dn.ou_path_root_first()).unwrap(); + let mut pool = ValuePool::new(); + let page = ingest_page(PAGE, tenant(), &mut dict, &mut pool, 99).unwrap(); + assert_eq!(page.records[0].ou_hhtl(), Some(ad_h)); +} + +#[test] +fn malformed_pages_are_refused() { + let (mut d, mut p) = (OuDictionary::new(), ValuePool::new()); + assert!(ingest_page("{}", tenant(), &mut d, &mut p, 0).is_err()); + assert!(ingest_page(r#"{"value":[{"id":"nope"}]}"#, tenant(), &mut d, &mut p, 0).is_err()); + assert!( + ingest_page( + r#"{"value":[{"id":"7a3c9e11-2b44-4c6d-9e8f-0123456789ab","accountEnabled":"yes"}]}"#, + tenant(), + &mut d, + &mut p, + 0 + ) + .is_err() + ); +} diff --git a/crates/ogar-dir-core/Cargo.toml b/crates/ogar-dir-core/Cargo.toml new file mode 100644 index 0000000..b37434b --- /dev/null +++ b/crates/ogar-dir-core/Cargo.toml @@ -0,0 +1,11 @@ +[package] +name = "ogar-dir-core" +version.workspace = true +edition.workspace = true +license.workspace = true +repository.workspace = true +authors.workspace = true +rust-version.workspace = true +description = "Shared ABI for OGAR directory observations (ogar-ad / ogar-az): 128-bit source GUIDs, an 8x16-bit OU HHTL with an explicit per-parent dictionary, a fixed 512-byte NodeRow-shaped record, an out-of-line value pool, and a minimal GUID-pair edge record. Zero dependencies; no classid mint; no IAM semantics." + +[dependencies] diff --git a/crates/ogar-dir-core/src/base64.rs b/crates/ogar-dir-core/src/base64.rs new file mode 100644 index 0000000..6bc1e92 --- /dev/null +++ b/crates/ogar-dir-core/src/base64.rs @@ -0,0 +1,51 @@ +//! Minimal standard-alphabet base64 decoder (LDIF `::` values and Graph +//! `onPremisesImmutableId`). Decode only; padding optional; whitespace rejected. + +/// Decode standard base64 (`A-Z a-z 0-9 + /`, optional `=` padding). +/// Returns `None` on any invalid character or impossible length. +pub fn decode(s: &str) -> Option> { + let s = s.trim_end_matches('='); + if s.len() % 4 == 1 { + return None; + } + let mut out = Vec::with_capacity(s.len() * 3 / 4); + let mut acc = 0u32; + let mut bits = 0u32; + for c in s.bytes() { + let v = match c { + b'A'..=b'Z' => c - b'A', + b'a'..=b'z' => c - b'a' + 26, + b'0'..=b'9' => c - b'0' + 52, + b'+' => 62, + b'/' => 63, + _ => return None, + } as u32; + acc = (acc << 6) | v; + bits += 6; + if bits >= 8 { + bits -= 8; + out.push((acc >> bits) as u8); + } + } + Some(out) +} + +#[cfg(test)] +mod tests { + #[test] + fn decodes_rfc4648_vectors() { + for (enc, dec) in [ + ("", ""), + ("Zg==", "f"), + ("Zm8=", "fo"), + ("Zm9v", "foo"), + ("Zm9vYg==", "foob"), + ("Zm9vYmE=", "fooba"), + ("Zm9vYmFy", "foobar"), + ] { + assert_eq!(super::decode(enc).unwrap(), dec.as_bytes(), "{enc}"); + } + assert!(super::decode("Zm9v!").is_none()); + assert!(super::decode("Z").is_none()); + } +} diff --git a/crates/ogar-dir-core/src/dn.rs b/crates/ogar-dir-core/src/dn.rs new file mode 100644 index 0000000..bd902cc --- /dev/null +++ b/crates/ogar-dir-core/src/dn.rs @@ -0,0 +1,231 @@ +//! RFC 4514 Distinguished Name parsing — just enough to split a DN into its +//! leaf, its location (OU chain) and its naming context (DC chain). +//! +//! A DN is never identity here. It is *evidence about location*: the OU +//! components feed the [`OuHhtl`](crate::hhtl::OuHhtl); the leaf `CN=` and the +//! `DC=` components never do. +//! +//! Supported: `\,` style escapes, `\HH` hex escapes (multi-byte UTF-8 via +//! consecutive pairs, e.g. `H\C3\BCbener`), optional spaces around `,` and `=`. +//! Rejected (returned as [`DnError`], never guessed): multi-valued RDNs (`+`), +//! empty components, dangling escapes, non-UTF-8 hex payloads. + +use core::fmt; + +/// One `type=value` component, value unescaped. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Rdn { + /// Attribute type as written (`OU`, `ou`, `CN`, `DC`, …). + pub attr: String, + /// Unescaped value. + pub value: String, +} + +impl Rdn { + /// Case-insensitive attribute-type test. + pub fn is(&self, attr: &str) -> bool { + self.attr.eq_ignore_ascii_case(attr) + } +} + +/// A parsed DN, components in written order (leaf first). +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Dn { + /// Components, leaf first — exactly as written. + pub rdns: Vec, +} + +/// Why a DN could not be parsed. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum DnError { + /// Empty input. + Empty, + /// A component had no `=`. + MissingEquals(usize), + /// A component had an empty type or value. + EmptyComponent(usize), + /// `+` multi-valued RDNs are out of PoC scope. + MultiValuedRdn(usize), + /// Backslash at end, or an invalid hex escape. + BadEscape(usize), + /// Hex escapes did not form valid UTF-8. + NotUtf8(usize), +} + +impl fmt::Display for DnError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "invalid distinguished name: {self:?}") + } +} +impl std::error::Error for DnError {} + +fn hex(c: u8) -> Option { + match c { + b'0'..=b'9' => Some(c - b'0'), + b'a'..=b'f' => Some(c - b'a' + 10), + b'A'..=b'F' => Some(c - b'A' + 10), + _ => None, + } +} + +impl Dn { + /// Parse an RFC 4514 DN string. + pub fn parse(s: &str) -> Result { + if s.trim().is_empty() { + return Err(DnError::Empty); + } + let b = s.as_bytes(); + let mut rdns = Vec::new(); + let mut i = 0usize; + while i <= b.len() { + let comp = rdns.len(); + // attribute type: up to '=' + let start = i; + while i < b.len() && b[i] != b'=' && b[i] != b',' { + i += 1; + } + if i >= b.len() || b[i] != b'=' { + return Err(DnError::MissingEquals(comp)); + } + let attr = s[start..i].trim().to_string(); + i += 1; // '=' + while i < b.len() && b[i] == b' ' { + i += 1; + } + // value: until unescaped ',' + let mut val: Vec = Vec::new(); + let mut trailing_unescaped_spaces = 0usize; + while i < b.len() && b[i] != b',' { + match b[i] { + b'\\' => { + let n = *b.get(i + 1).ok_or(DnError::BadEscape(comp))?; + if let Some(h) = hex(n) { + let l = b + .get(i + 2) + .copied() + .and_then(hex) + .ok_or(DnError::BadEscape(comp))?; + val.push(h << 4 | l); + i += 3; + } else { + val.push(n); + i += 2; + } + trailing_unescaped_spaces = 0; + } + b'+' => return Err(DnError::MultiValuedRdn(comp)), + c => { + val.push(c); + trailing_unescaped_spaces = if c == b' ' { + trailing_unescaped_spaces + 1 + } else { + 0 + }; + i += 1; + } + } + } + val.truncate(val.len() - trailing_unescaped_spaces); + let value = String::from_utf8(val).map_err(|_| DnError::NotUtf8(comp))?; + if attr.is_empty() || value.is_empty() { + return Err(DnError::EmptyComponent(comp)); + } + rdns.push(Rdn { attr, value }); + if i >= b.len() { + break; + } + i += 1; // ',' + } + Ok(Self { rdns }) + } + + /// The leaf component (first written). Never part of the HHTL. + pub fn leaf(&self) -> &Rdn { + &self.rdns[0] + } + + /// OU values, **root first** (the reverse of written order), excluding the + /// leaf even if the leaf itself is an `OU=` (an OU object's own name is its + /// leaf, not its location). + pub fn ou_path_root_first(&self) -> Vec<&str> { + self.rdns[1..] + .iter() + .rev() + .filter(|r| r.is("OU")) + .map(|r| r.value.as_str()) + .collect() + } + + /// `DC=` values in written order (`["example", "de"]`). + pub fn dc_components(&self) -> Vec<&str> { + self.rdns + .iter() + .filter(|r| r.is("DC")) + .map(|r| r.value.as_str()) + .collect() + } + + /// True when the location (all non-leaf components) contains something that + /// is neither `OU=` nor `DC=` — e.g. the `CN=Users` container. Such objects + /// have a location the OU-HHTL does not express; the record flags it rather + /// than conflating it with "directly under the domain root". + pub fn has_non_ou_container(&self) -> bool { + self.rdns[1..].iter().any(|r| !r.is("OU") && !r.is("DC")) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn splits_leaf_ou_and_dc() { + let dn = + Dn::parse("CN=Jan Hübener,OU=Exchange,OU=Infrastructure,OU=Stuttgart,DC=example,DC=de") + .unwrap(); + assert_eq!(dn.leaf().value, "Jan Hübener"); + assert_eq!( + dn.ou_path_root_first(), + ["Stuttgart", "Infrastructure", "Exchange"] + ); + assert_eq!(dn.dc_components(), ["example", "de"]); + assert!(!dn.has_non_ou_container()); + } + + #[test] + fn handles_escapes_and_spaces() { + let dn = + Dn::parse(r"CN=Doe\, John , OU = Sales\2C EMEA, OU=H\C3\BCbener\20, DC=x").unwrap(); + assert_eq!(dn.leaf().value, "Doe, John"); + assert_eq!(dn.ou_path_root_first(), ["Hübener ", "Sales, EMEA"]); + } + + #[test] + fn flags_cn_containers() { + let dn = Dn::parse("CN=Jan,CN=Users,DC=example,DC=de").unwrap(); + assert!(dn.ou_path_root_first().is_empty()); + assert!(dn.has_non_ou_container()); + } + + #[test] + fn rejects_out_of_scope_or_broken_input() { + assert_eq!(Dn::parse(""), Err(DnError::Empty)); + assert!(matches!( + Dn::parse("CN=a+UID=b,DC=x"), + Err(DnError::MultiValuedRdn(0)) + )); + assert!(matches!( + Dn::parse("CN=a,OU,DC=x"), + Err(DnError::MissingEquals(1)) + )); + assert!(matches!(Dn::parse(r"CN=a\"), Err(DnError::BadEscape(0)))); + assert!(matches!( + Dn::parse("CN=,DC=x"), + Err(DnError::EmptyComponent(0)) + )); + assert!(matches!( + Dn::parse(r"CN=\FF\FE,DC=x"), + Err(DnError::NotUtf8(0)) + )); + } +} diff --git a/crates/ogar-dir-core/src/edge.rs b/crates/ogar-dir-core/src/edge.rs new file mode 100644 index 0000000..5e73cdb --- /dev/null +++ b/crates/ogar-dir-core/src/edge.rs @@ -0,0 +1,99 @@ +//! Relationships are records of their own, never inline lists in the 512-byte +//! node. One fixed 64-byte edge: +//! +//! ```text +//! 0x00 kind u16 EdgeKind +//! 0x02 evidence u16 EdgeEvidence — WHY the edge was asserted +//! 0x04 src_family u16 SchemaFamily of src +//! 0x06 dst_family u16 SchemaFamily of dst +//! 0x08 reserved [8] zero +//! 0x10 src_guid [16] textual byte order +//! 0x20 dst_guid [16] +//! 0x30 reserved [16] zero +//! ``` +//! +//! Endpoints are `(family, guid)` pairs. An edge between an AD object and an +//! Entra object relates two **distinct nodes**; it never merges them. +//! +//! The vocabulary is deliberately minimal: only [`EdgeKind::SynchronizesTo`] +//! exists. `MEMBER_OF` / `MANAGER` are expected later and get the next codes; +//! they are not defined until a slice actually emits them. +//! +//! lance-graph's `CausalEdge64` was checked and does not fit: it packs a +//! reasoning edge into 8 bytes and cannot carry two 128-bit endpoints. + +use crate::guid::Guid128; +use crate::schema::SchemaFamily; + +/// Edge record size. +pub const EDGE_BYTES: usize = 64; + +/// What the edge asserts. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u16)] +pub enum EdgeKind { + /// On-premises source object is synchronized to a cloud object (the + /// SAME_AS candidate). Direction: AD → Entra. + SynchronizesTo = 1, +} + +/// Which observation supports the edge. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u16)] +pub enum EdgeEvidence { + /// Graph `onPremisesImmutableId` base64-decodes to 16 bytes equal (as a + /// Microsoft mixed-endian GUID) to an observed AD `objectGUID`. + ImmutableIdIsObjectGuid = 1, +} + +/// One edge. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct DirEdge { + /// Kind. + pub kind: EdgeKind, + /// Evidence. + pub evidence: EdgeEvidence, + /// Source endpoint. + pub src: (SchemaFamily, Guid128), + /// Destination endpoint. + pub dst: (SchemaFamily, Guid128), +} + +impl DirEdge { + /// 64 wire bytes. + pub fn to_bytes(&self) -> [u8; EDGE_BYTES] { + let mut b = [0u8; EDGE_BYTES]; + b[0..2].copy_from_slice(&(self.kind as u16).to_le_bytes()); + b[2..4].copy_from_slice(&(self.evidence as u16).to_le_bytes()); + b[4..6].copy_from_slice(&(self.src.0 as u16).to_le_bytes()); + b[6..8].copy_from_slice(&(self.dst.0 as u16).to_le_bytes()); + b[0x10..0x20].copy_from_slice(&self.src.1.0); + b[0x20..0x30].copy_from_slice(&self.dst.1.0); + b + } + + /// Decode; `None` on unknown codes. + pub fn from_bytes(b: &[u8; EDGE_BYTES]) -> Option { + let u = |o: usize| u16::from_le_bytes([b[o], b[o + 1]]); + let kind = match u(0) { + 1 => EdgeKind::SynchronizesTo, + _ => return None, + }; + let evidence = match u(2) { + 1 => EdgeEvidence::ImmutableIdIsObjectGuid, + _ => return None, + }; + Some(Self { + kind, + evidence, + src: ( + SchemaFamily::from_u16(u(4))?, + Guid128(b[0x10..0x20].try_into().ok()?), + ), + dst: ( + SchemaFamily::from_u16(u(6))?, + Guid128(b[0x20..0x30].try_into().ok()?), + ), + }) + } +} diff --git a/crates/ogar-dir-core/src/guid.rs b/crates/ogar-dir-core/src/guid.rs new file mode 100644 index 0000000..9910932 --- /dev/null +++ b/crates/ogar-dir-core/src/guid.rs @@ -0,0 +1,170 @@ +//! 128-bit source identity. +//! +//! **Byte order is fixed, not inferred.** [`Guid128`] stores the 16 bytes in +//! *textual order* — the order the 32 hex digits appear in +//! `xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx` (RFC 9562 network order). Two inputs +//! differ: +//! +//! * Microsoft Graph returns ids as that string → [`Guid128::parse`]. +//! * LDAP returns `objectGUID` as 16 raw bytes in Microsoft *mixed-endian* +//! (`Data1` u32 LE, `Data2` u16 LE, `Data3` u16 LE, `Data4` 8 bytes) — the +//! same bytes .NET `Guid.ToByteArray()` and `onPremisesImmutableId` (base64) +//! carry → [`Guid128::from_ms_bytes`]. +//! +//! Getting this wrong would silently produce a different, equally valid-looking +//! GUID; both conversions are tested against a known pair. +//! +//! There is no `u128` in the ABI: the record carries `[u8; 16]` so no Rust +//! integer layout or host endianness is involved. No truncation API exists. + +use core::fmt; + +/// A 128-bit identifier in textual byte order. All-zero means "absent". +#[derive(Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord, Default)] +#[repr(transparent)] +pub struct Guid128(pub [u8; 16]); + +/// Why a GUID string or byte slice was rejected. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum GuidParseError { + /// Not the 36-char hyphenated (or 38-char braced) form. + BadLength(usize), + /// A hyphen is missing or a digit is not hex. + BadDigit(usize), + /// Raw input was not exactly 16 bytes. + BadRawLength(usize), +} + +impl fmt::Display for GuidParseError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::BadLength(n) => write!(f, "guid string has length {n}, expected 36"), + Self::BadDigit(i) => write!(f, "guid string invalid at position {i}"), + Self::BadRawLength(n) => write!(f, "raw guid has {n} bytes, expected 16"), + } + } +} +impl std::error::Error for GuidParseError {} + +/// Positions of the four hyphens in the 36-char form. +const HYPHENS: [usize; 4] = [8, 13, 18, 23]; + +impl Guid128 { + /// The all-zero GUID (never a valid directory object id). + pub const NIL: Self = Self([0; 16]); + + /// True for [`Self::NIL`]. + pub fn is_nil(&self) -> bool { + self.0 == [0; 16] + } + + /// Parse `xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx`, optionally `{…}`-braced, + /// case-insensitive. + pub fn parse(s: &str) -> Result { + let s = s + .strip_prefix('{') + .and_then(|t| t.strip_suffix('}')) + .unwrap_or(s); + let b = s.as_bytes(); + if b.len() != 36 { + return Err(GuidParseError::BadLength(b.len())); + } + let mut out = [0u8; 16]; + let mut nib = 0usize; + for (i, &c) in b.iter().enumerate() { + if HYPHENS.contains(&i) { + if c != b'-' { + return Err(GuidParseError::BadDigit(i)); + } + continue; + } + let v = match c { + b'0'..=b'9' => c - b'0', + b'a'..=b'f' => c - b'a' + 10, + b'A'..=b'F' => c - b'A' + 10, + _ => return Err(GuidParseError::BadDigit(i)), + }; + out[nib / 2] |= if nib.is_multiple_of(2) { v << 4 } else { v }; + nib += 1; + } + Ok(Self(out)) + } + + /// From Microsoft mixed-endian raw bytes (LDAP `objectGUID`, + /// `Guid.ToByteArray()`, decoded `onPremisesImmutableId`). + pub fn from_ms_bytes(raw: &[u8]) -> Result { + let r: [u8; 16] = raw + .try_into() + .map_err(|_| GuidParseError::BadRawLength(raw.len()))?; + Ok(Self([ + r[3], r[2], r[1], r[0], r[5], r[4], r[7], r[6], r[8], r[9], r[10], r[11], r[12], r[13], + r[14], r[15], + ])) + } + + /// Back to Microsoft mixed-endian raw bytes (inverse of [`Self::from_ms_bytes`]). + pub fn to_ms_bytes(&self) -> [u8; 16] { + let g = self.0; + [ + g[3], g[2], g[1], g[0], g[5], g[4], g[7], g[6], g[8], g[9], g[10], g[11], g[12], g[13], + g[14], g[15], + ] + } +} + +impl fmt::Display for Guid128 { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + for (i, b) in self.0.iter().enumerate() { + if matches!(i, 4 | 6 | 8 | 10) { + f.write_str("-")?; + } + write!(f, "{b:02x}")?; + } + Ok(()) + } +} + +impl fmt::Debug for Guid128 { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "Guid128({self})") + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn string_round_trip_is_exact() { + let s = "3f2504e0-4f89-11d3-9a0c-0305e82c3301"; + let g = Guid128::parse(s).unwrap(); + assert_eq!(g.0[0], 0x3f); + assert_eq!(g.0[15], 0x01); + assert_eq!(g.to_string(), s); + assert_eq!(Guid128::parse(&s.to_uppercase()).unwrap(), g); + assert_eq!(Guid128::parse(&format!("{{{s}}}")).unwrap(), g); + } + + /// Known pair: .NET `new Guid("3f2504e0-4f89-11d3-9a0c-0305e82c3301").ToByteArray()` + /// = e0 04 25 3f 89 4f d3 11 9a 0c 03 05 e8 2c 33 01. + #[test] + fn ms_mixed_endian_matches_dotnet_byte_order() { + let raw = [ + 0xe0, 0x04, 0x25, 0x3f, 0x89, 0x4f, 0xd3, 0x11, 0x9a, 0x0c, 0x03, 0x05, 0xe8, 0x2c, + 0x33, 0x01, + ]; + let g = Guid128::from_ms_bytes(&raw).unwrap(); + assert_eq!(g.to_string(), "3f2504e0-4f89-11d3-9a0c-0305e82c3301"); + assert_eq!(g.to_ms_bytes(), raw); + // The two orders really differ — a byte-order mistake is not a no-op. + assert_ne!(g.0, raw); + } + + #[test] + fn rejects_malformed_input() { + assert!(Guid128::parse("3f2504e0-4f89-11d3-9a0c-0305e82c330").is_err()); + assert!(Guid128::parse("3f2504e0x4f89-11d3-9a0c-0305e82c3301").is_err()); + assert!(Guid128::parse("3f2504e0-4f89-11d3-9a0c-0305e82c33zz").is_err()); + assert!(Guid128::from_ms_bytes(&[0u8; 15]).is_err()); + } +} diff --git a/crates/ogar-dir-core/src/hhtl.rs b/crates/ogar-dir-core/src/hhtl.rs new file mode 100644 index 0000000..539f460 --- /dev/null +++ b/crates/ogar-dir-core/src/hhtl.rs @@ -0,0 +1,239 @@ +//! The OU tree as a 128-bit HHTL: 8 levels × 16-bit segment ids, root first. +//! +//! ```text +//! CN=Jan Hübener,OU=Exchange,OU=Infrastructure,OU=Stuttgart,DC=example,DC=de +//! ─── DC: scope, not a level +//! level: 0 1 2 3 4 5 6 7 +//! segment: Stuttgart Infrastructure Exchange 0 0 0 0 0 +//! ``` +//! +//! * Only `OU=` components are levels. The leaf and every `DC=` are not. +//! * Segment id `0` means "no level here". Ids `1..=65535` are allocated. +//! * Depth is the number of leading non-zero levels; a non-zero level after a +//! zero one is malformed and rejected on decode. +//! * More than 8 OU levels is an error, never a silent truncation. +//! +//! ## Segment assignment: explicit per-parent dictionary, not a hash +//! +//! Hashing an OU name into 16 bits collides by the birthday bound after ~300 +//! siblings and is not reversible. Instead an [`OuDictionary`] (one per +//! directory scope — one per AD domain / Entra tenant) allocates ids +//! **sequentially per parent path** on first sight, starting at 1: +//! +//! * collisions are impossible by construction (a `(parent, name)` key maps to +//! exactly one id; an id under a parent maps back to exactly one name); +//! * an id is only meaningful under its parent, so the same id under two +//! parents is not a collision (65535 children per OU, not per directory); +//! * reconstruction is exact: [`OuDictionary::explain`] returns the original +//! (first-seen) spelling; +//! * determinism: ids depend on first-seen order, so the dictionary is +//! *state* and must be persisted alongside the records +//! ([`OuDictionary::entries`] / [`OuDictionary::from_entries`]). Re-deriving +//! it from a differently ordered crawl would assign different ids — this is +//! the documented price of reversibility. +//! +//! Name matching is case-insensitive (AD compares RDN values that way) using +//! Unicode lowercase as the fold. This is an approximation of AD's own +//! collation and is flagged as such; the original spelling is what `explain` +//! returns. + +use std::collections::HashMap; + +/// Number of HHTL levels. +pub const OU_LEVELS: usize = 8; + +/// 8 × u16 OU path, root first. Bytes on the wire: each level little-endian. +#[derive(Clone, Copy, PartialEq, Eq, Hash, Default, Debug, PartialOrd, Ord)] +pub struct OuHhtl(pub [u16; OU_LEVELS]); + +/// HHTL / dictionary failure. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum HhtlError { + /// More than [`OU_LEVELS`] OU components. + TooDeep(usize), + /// A parent already has 65535 children. + ParentExhausted, + /// A non-zero level follows a zero level. + Gap(usize), + /// An OU name was empty. + EmptyName, + /// Persisted entries map two names to one id, or one name to two ids. + Collision, +} + +impl std::fmt::Display for HhtlError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "ou hhtl: {self:?}") + } +} +impl std::error::Error for HhtlError {} + +impl OuHhtl { + /// The empty path (object directly under the naming context). + pub const ROOT: Self = Self([0; OU_LEVELS]); + + /// Number of populated levels. + pub fn depth(&self) -> usize { + self.0.iter().take_while(|&&s| s != 0).count() + } + + /// The first `depth` levels, rest zero. + pub fn prefix(&self, depth: usize) -> Self { + let mut p = Self::ROOT; + p.0[..depth].copy_from_slice(&self.0[..depth]); + p + } + + /// Child path with `seg` appended at the next free level. + fn child(&self, seg: u16) -> Result { + let d = self.depth(); + if d >= OU_LEVELS { + return Err(HhtlError::TooDeep(d + 1)); + } + let mut c = *self; + c.0[d] = seg; + Ok(c) + } + + /// True if `self` is a (non-strict) ancestor of `other`. + pub fn is_ancestor_of(&self, other: &Self) -> bool { + let d = self.depth(); + d <= other.depth() && self.0[..d] == other.0[..d] + } + + /// 16 wire bytes, each level little-endian. + pub fn to_le_bytes(&self) -> [u8; 16] { + let mut b = [0u8; 16]; + for (i, s) in self.0.iter().enumerate() { + b[i * 2..i * 2 + 2].copy_from_slice(&s.to_le_bytes()); + } + b + } + + /// Decode 16 wire bytes, rejecting gaps. + pub fn from_le_bytes(b: &[u8; 16]) -> Result { + let mut h = Self::ROOT; + for i in 0..OU_LEVELS { + h.0[i] = u16::from_le_bytes([b[i * 2], b[i * 2 + 1]]); + } + let d = h.depth(); + if let Some(i) = (d..OU_LEVELS).find(|&i| h.0[i] != 0) { + return Err(HhtlError::Gap(i)); + } + Ok(h) + } +} + +/// Per-scope OU segment registry. See the module docs. +#[derive(Debug, Default, Clone)] +pub struct OuDictionary { + forward: HashMap<(OuHhtl, String), u16>, + reverse: HashMap<(OuHhtl, u16), String>, + next: HashMap, +} + +fn fold(name: &str) -> String { + name.to_lowercase() +} + +impl OuDictionary { + /// Empty dictionary. + pub fn new() -> Self { + Self::default() + } + + /// Resolve a root-first OU path, allocating ids for unseen segments. + pub fn intern>(&mut self, root_first: &[S]) -> Result { + if root_first.len() > OU_LEVELS { + return Err(HhtlError::TooDeep(root_first.len())); + } + let mut path = OuHhtl::ROOT; + for name in root_first { + let name = name.as_ref(); + if name.is_empty() { + return Err(HhtlError::EmptyName); + } + let key = (path, fold(name)); + let seg = match self.forward.get(&key) { + Some(&s) => s, + None => { + let n = self.next.entry(path).or_insert(1); + if *n == 0 { + return Err(HhtlError::ParentExhausted); + } + let s = *n; + *n = n.wrapping_add(1); // 65535 -> 0 marks exhaustion + self.forward.insert(key, s); + self.reverse.insert((path, s), name.to_string()); + s + } + }; + path = path.child(seg)?; + } + Ok(path) + } + + /// Resolve without allocating; `None` if any segment is unknown. + pub fn resolve>(&self, root_first: &[S]) -> Option { + if root_first.len() > OU_LEVELS { + return None; + } + let mut path = OuHhtl::ROOT; + for name in root_first { + let s = *self.forward.get(&(path, fold(name.as_ref())))?; + path = path.child(s).ok()?; + } + Some(path) + } + + /// Reconstruct the root-first OU names (original spelling). + pub fn explain(&self, h: &OuHhtl) -> Option> { + (0..h.depth()) + .map(|d| self.reverse.get(&(h.prefix(d), h.0[d])).cloned()) + .collect() + } + + /// Persistable entries `(parent, segment, original name)`, sorted. + pub fn entries(&self) -> Vec<(OuHhtl, u16, String)> { + let mut v: Vec<_> = self + .reverse + .iter() + .map(|((p, s), n)| (*p, *s, n.clone())) + .collect(); + v.sort(); + v + } + + /// Rebuild from [`Self::entries`]. Rejects a duplicate `(parent, segment)` + /// or a duplicate `(parent, folded name)` — either would be a collision. + pub fn from_entries(entries: &[(OuHhtl, u16, String)]) -> Result { + let mut d = Self::new(); + for (p, s, n) in entries { + if *s == 0 || n.is_empty() { + return Err(HhtlError::EmptyName); + } + if d.reverse.insert((*p, *s), n.clone()).is_some() + || d.forward.insert((*p, fold(n)), *s).is_some() + { + return Err(HhtlError::Collision); + } + let next = d.next.entry(*p).or_insert(1); + if *s == u16::MAX { + *next = 0; // parent exhausted + } else if *next != 0 && s + 1 > *next { + *next = s + 1; + } + } + Ok(d) + } + + /// Number of allocated segments. + pub fn len(&self) -> usize { + self.reverse.len() + } + + /// True if nothing is allocated. + pub fn is_empty(&self) -> bool { + self.reverse.is_empty() + } +} diff --git a/crates/ogar-dir-core/src/lib.rs b/crates/ogar-dir-core/src/lib.rs new file mode 100644 index 0000000..d02ad2b --- /dev/null +++ b/crates/ogar-dir-core/src/lib.rs @@ -0,0 +1,44 @@ +//! # ogar-dir-core — observed directory reality, as bytes +//! +//! Shared substrate for `ogar-ad` (Active Directory / LDAP) and `ogar-az` +//! (Entra ID / Microsoft Graph). It answers exactly one question: *how is one +//! observation of one directory object laid out?* It is not an IAM, does not +//! provision, reconcile or apply business rules. +//! +//! The four axes are kept orthogonal: +//! +//! | axis | carrier | module | +//! |-------|-------------------------------------------|-------------| +//! | WHO | [`Guid128`] source object id | [`guid`] | +//! | WHERE | scope [`Guid128`] + [`OuHhtl`] | [`hhtl`] | +//! | WHAT | schema-defined attribute slots + [`ValuePool`] | [`record`], [`pool`], [`schema`] | +//! | LINKS | [`DirEdge`] records, never inline lists | [`edge`] | +//! | WHEN | `observed_at_ms` only (provenance hook) | [`record`] | +//! +//! ## Relation to the OGAR canonical node +//! +//! [`DirRecord`] is exactly 512 bytes and keeps the canonical +//! `key(16) | edges(16) | value(480)` split of lance-graph-contract `NodeRow`. +//! The 16-byte canonical key is a MINTED address (`classid` + 12-byte facet) +//! and is therefore **not** the source GUID: a 128-bit external GUID cannot be +//! placed in the key without truncation. In this PoC the key and edge facet +//! are left zero (the documented zero-fallback "default class, dormant"), and +//! the authoritative source GUID lives in the value slab. See `docs/` in the +//! workspace for the conflict note. + +pub mod base64; +pub mod dn; +pub mod edge; +pub mod guid; +pub mod hhtl; +pub mod pool; +pub mod record; +pub mod schema; + +pub use dn::{Dn, DnError, Rdn}; +pub use edge::{DirEdge, EdgeEvidence, EdgeKind}; +pub use guid::{Guid128, GuidParseError}; +pub use hhtl::{HhtlError, OU_LEVELS, OuDictionary, OuHhtl}; +pub use pool::{PoolError, StrRef, ValuePool}; +pub use record::{ABI_MAJOR, ABI_MINOR, DirRecord, RECORD_BYTES, RecordError}; +pub use schema::{AttrDef, AttrKind, SchemaFamily, SchemaId}; diff --git a/crates/ogar-dir-core/src/pool.rs b/crates/ogar-dir-core/src/pool.rs new file mode 100644 index 0000000..26a8e00 --- /dev/null +++ b/crates/ogar-dir-core/src/pool.rs @@ -0,0 +1,126 @@ +//! Out-of-line value storage. Strings, binary values (`objectSid`) and +//! multi-valued attributes (`proxyAddresses`) live here; the fixed record +//! carries only a [`StrRef`] per attribute slot. +//! +//! A pool belongs to a batch of records (one column chunk). A [`StrRef`] is +//! meaningless without the pool it was issued by. Values are stored byte-exact +//! as observed — no case folding, trimming or normalisation. +//! +//! Multi-valued layout inside the pool: `(u32 LE length, bytes)*`, in source +//! order. Whether a slot is single- or multi-valued is a property of the +//! schema ([`AttrKind`](crate::schema::AttrKind)), not of the bytes. + +/// `(offset, length)` into a [`ValuePool`], both u32 LE on the wire. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Default)] +pub struct StrRef { + /// Byte offset into the pool. + pub off: u32, + /// Byte length. + pub len: u32, +} + +impl StrRef { + /// 8 wire bytes. + pub fn to_le_bytes(self) -> [u8; 8] { + let mut b = [0u8; 8]; + b[..4].copy_from_slice(&self.off.to_le_bytes()); + b[4..].copy_from_slice(&self.len.to_le_bytes()); + b + } + /// From 8 wire bytes. + pub fn from_le_bytes(b: [u8; 8]) -> Self { + Self { + off: u32::from_le_bytes([b[0], b[1], b[2], b[3]]), + len: u32::from_le_bytes([b[4], b[5], b[6], b[7]]), + } + } +} + +/// Pool failure. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum PoolError { + /// The pool would exceed 4 GiB (u32 offsets). + Overflow, +} + +/// Append-only byte arena. +#[derive(Debug, Default, Clone)] +pub struct ValuePool { + bytes: Vec, +} + +impl ValuePool { + /// Empty pool. + pub fn new() -> Self { + Self::default() + } + + fn reserve_ref(&self, len: usize) -> Result { + let off = u32::try_from(self.bytes.len()).map_err(|_| PoolError::Overflow)?; + let len32 = u32::try_from(len).map_err(|_| PoolError::Overflow)?; + off.checked_add(len32).ok_or(PoolError::Overflow)?; + Ok(StrRef { off, len: len32 }) + } + + /// Store one value. + pub fn push(&mut self, v: &[u8]) -> Result { + let r = self.reserve_ref(v.len())?; + self.bytes.extend_from_slice(v); + Ok(r) + } + + /// Store a multi-valued attribute. + pub fn push_multi>(&mut self, vs: &[V]) -> Result { + let total: usize = vs.iter().map(|v| 4 + v.as_ref().len()).sum(); + let r = self.reserve_ref(total)?; + for v in vs { + let v = v.as_ref(); + self.bytes + .extend_from_slice(&(v.len() as u32).to_le_bytes()); + self.bytes.extend_from_slice(v); + } + Ok(r) + } + + /// The bytes behind a ref, `None` if out of range. + pub fn get(&self, r: StrRef) -> Option<&[u8]> { + let s = r.off as usize; + self.bytes.get(s..s.checked_add(r.len as usize)?) + } + + /// Decode a multi-valued ref, `None` if malformed or out of range. + pub fn get_multi(&self, r: StrRef) -> Option> { + let mut b = self.get(r)?; + let mut out = Vec::new(); + while !b.is_empty() { + let n = u32::from_le_bytes(b.get(..4)?.try_into().ok()?) as usize; + out.push(b.get(4..4 + n)?); + b = &b[4 + n..]; + } + Some(out) + } + + /// Raw pool bytes (the column payload). + pub fn as_bytes(&self) -> &[u8] { + &self.bytes + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn single_and_multi_round_trip() { + let mut p = ValuePool::new(); + let a = p.push("SMTP:jan@example.de".as_bytes()).unwrap(); + let m = p.push_multi(&["SMTP:a@x", "smtp:b@x", ""]).unwrap(); + assert_eq!(p.get(a).unwrap(), b"SMTP:jan@example.de"); + assert_eq!( + p.get_multi(m).unwrap(), + vec![&b"SMTP:a@x"[..], b"smtp:b@x", b""] + ); + assert_eq!(StrRef::from_le_bytes(m.to_le_bytes()), m); + assert!(p.get(StrRef { off: 1000, len: 1 }).is_none()); + } +} diff --git a/crates/ogar-dir-core/src/record.rs b/crates/ogar-dir-core/src/record.rs new file mode 100644 index 0000000..35c27cc --- /dev/null +++ b/crates/ogar-dir-core/src/record.rs @@ -0,0 +1,320 @@ +//! The fixed 512-byte directory observation record. +//! +//! ## Exact layout (all multi-byte integers little-endian; no padding exists +//! because the record is one `[u8; 512]` addressed by the offsets below) +//! +//! ```text +//! 0x000..0x010 canonical_key [16] NodeRow key slot. ZERO in this PoC +//! (= dormant default class). NOT the source GUID. +//! 0x010..0x020 canonical_edges [16] NodeRow edge-facet slot. ZERO; relations are DirEdge. +//! ---- 0x020..0x200 = the 480-byte NodeRow value slab ------------------------- +//! 0x020..0x024 magic [4] b"OGDR" +//! 0x024..0x026 abi_major u16 ABI_MAJOR (exact match required) +//! 0x026..0x028 abi_minor u16 ABI_MINOR (additive; readers ignore newer) +//! 0x028..0x02A schema_family u16 SchemaFamily (1 = AD_DS, 2 = MS_GRAPH) +//! 0x02A..0x02C schema_version u16 encoder's schema version +//! 0x02C..0x02E object_kind u16 family-scoped kind code +//! 0x02E..0x030 flags u16 FLAG_* below +//! 0x030..0x040 node_guid [16] WHO source object id, textual byte order +//! 0x040..0x050 scope_guid [16] WHERE AD domain GUID / Entra tenant id +//! 0x050..0x060 ou_hhtl [16] WHERE 8 × u16 LE, root first +//! 0x060..0x068 observed_at_ms i64 WHEN unix ms the observation was taken (0 = unknown) +//! 0x068..0x070 presence u64 bit s = pooled slot s present (0..32); +//! bit 32+n = numeric slot n present (n < 4) +//! 0x070..0x080 num 4×u32 numeric slots (meaning per schema) +//! 0x080..0x180 str_refs 32×(u32 off, u32 len) into the batch ValuePool +//! 0x180..0x200 reserved [128] zero; writers MUST zero, readers MUST ignore +//! ``` +//! +//! GUIDs are stored in textual byte order (see [`crate::guid`]). Strings never +//! live in the record; an absent slot has its presence bit clear (a present +//! empty string is distinguishable from absence). +//! +//! The canonical `key|edges|value` split mirrors lance-graph-contract +//! `NodeRow` (16|16|480, stride 512). That mirror is **unguarded** in this +//! crate (no dependency on the contract); see the PoC conflict notes. + +use crate::guid::Guid128; +use crate::hhtl::{HhtlError, OuHhtl}; +use crate::pool::StrRef; +use crate::schema::{SchemaFamily, SchemaId}; + +/// Record size in bytes. +pub const RECORD_BYTES: usize = 512; +/// Layout major: bump only when an existing offset changes meaning. +pub const ABI_MAJOR: u16 = 1; +/// Layout minor: bump when reserved bytes gain meaning (additive). +pub const ABI_MINOR: u16 = 0; +/// Magic. +pub const MAGIC: [u8; 4] = *b"OGDR"; +/// Pooled slot count. +pub const STR_SLOTS: usize = 32; +/// Numeric slot count. +pub const NUM_SLOTS: usize = 4; + +/// Offsets (public so a non-Rust reader can be generated from them). +pub mod off { + #![allow(missing_docs)] + pub const CANONICAL_KEY: usize = 0x000; + pub const CANONICAL_EDGES: usize = 0x010; + pub const VALUE: usize = 0x020; + pub const MAGIC: usize = 0x020; + pub const ABI_MAJOR: usize = 0x024; + pub const ABI_MINOR: usize = 0x026; + pub const SCHEMA_FAMILY: usize = 0x028; + pub const SCHEMA_VERSION: usize = 0x02A; + pub const OBJECT_KIND: usize = 0x02C; + pub const FLAGS: usize = 0x02E; + pub const NODE_GUID: usize = 0x030; + pub const SCOPE_GUID: usize = 0x040; + pub const OU_HHTL: usize = 0x050; + pub const OBSERVED_AT: usize = 0x060; + pub const PRESENCE: usize = 0x068; + pub const NUM: usize = 0x070; + pub const STR_REFS: usize = 0x080; + pub const RESERVED: usize = 0x180; + pub const END: usize = 0x200; +} + +/// `ou_hhtl` was derived from a parsed DN (else it is zero and meaningless). +pub const FLAG_OU_PRESENT: u16 = 1 << 0; +/// The location contains a non-OU, non-DC container (e.g. `CN=Users`). +pub const FLAG_NON_OU_CONTAINER: u16 = 1 << 1; +/// A DN was observed but could not be encoded (unparseable or > 8 OUs); the +/// raw DN is still in its slot. +pub const FLAG_DN_UNENCODED: u16 = 1 << 2; + +// Layout lock. +const _: () = assert!(off::END == RECORD_BYTES); +const _: () = assert!(off::STR_REFS + STR_SLOTS * 8 == off::RESERVED); +const _: () = assert!(off::NUM + NUM_SLOTS * 4 == off::STR_REFS); +const _: () = assert!(off::VALUE + 480 == RECORD_BYTES); +const _: () = assert!(core::mem::size_of::() == RECORD_BYTES); +const _: () = assert!(core::mem::align_of::() == 64); + +/// One observation. A thin typed view over exactly 512 bytes. +#[derive(Clone, Copy, PartialEq, Eq)] +#[repr(C, align(64))] +pub struct DirRecord { + bytes: [u8; RECORD_BYTES], +} + +/// Why bytes were not accepted as a record. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum RecordError { + /// Magic mismatch. + BadMagic, + /// Layout major differs. + AbiMajor(u16), + /// Unknown schema family. + UnknownFamily(u16), + /// Malformed HHTL. + Hhtl(HhtlError), + /// Slot index out of range. + Slot(usize), +} + +impl std::fmt::Display for RecordError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "dir record: {self:?}") + } +} +impl std::error::Error for RecordError {} + +impl std::fmt::Debug for DirRecord { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("DirRecord") + .field("schema", &self.schema()) + .field("kind", &self.object_kind()) + .field("node", &self.node_guid()) + .field("scope", &self.scope_guid()) + .field("ou_hhtl", &self.ou_hhtl()) + .field("flags", &self.flags()) + .finish() + } +} + +impl DirRecord { + fn u16_at(&self, o: usize) -> u16 { + u16::from_le_bytes([self.bytes[o], self.bytes[o + 1]]) + } + fn put_u16(&mut self, o: usize, v: u16) { + self.bytes[o..o + 2].copy_from_slice(&v.to_le_bytes()); + } + fn guid_at(&self, o: usize) -> Guid128 { + Guid128(self.bytes[o..o + 16].try_into().expect("16")) + } + + /// A fresh record: header, identity and scope set; no attributes. + pub fn new( + schema: SchemaId, + kind: u16, + node: Guid128, + scope: Guid128, + observed_at_ms: i64, + ) -> Self { + let mut r = Self { + bytes: [0; RECORD_BYTES], + }; + r.bytes[off::MAGIC..off::MAGIC + 4].copy_from_slice(&MAGIC); + r.put_u16(off::ABI_MAJOR, ABI_MAJOR); + r.put_u16(off::ABI_MINOR, ABI_MINOR); + r.put_u16(off::SCHEMA_FAMILY, schema.family as u16); + r.put_u16(off::SCHEMA_VERSION, schema.version); + r.put_u16(off::OBJECT_KIND, kind); + r.bytes[off::NODE_GUID..off::NODE_GUID + 16].copy_from_slice(&node.0); + r.bytes[off::SCOPE_GUID..off::SCOPE_GUID + 16].copy_from_slice(&scope.0); + r.bytes[off::OBSERVED_AT..off::OBSERVED_AT + 8] + .copy_from_slice(&observed_at_ms.to_le_bytes()); + r + } + + /// Validate and wrap raw bytes. + pub fn from_bytes(b: &[u8; RECORD_BYTES]) -> Result { + let r = Self { bytes: *b }; + if r.bytes[off::MAGIC..off::MAGIC + 4] != MAGIC { + return Err(RecordError::BadMagic); + } + let major = r.u16_at(off::ABI_MAJOR); + if major != ABI_MAJOR { + return Err(RecordError::AbiMajor(major)); + } + let fam = r.u16_at(off::SCHEMA_FAMILY); + SchemaFamily::from_u16(fam).ok_or(RecordError::UnknownFamily(fam))?; + let h: [u8; 16] = r.bytes[off::OU_HHTL..off::OU_HHTL + 16] + .try_into() + .expect("16"); + OuHhtl::from_le_bytes(&h).map_err(RecordError::Hhtl)?; + Ok(r) + } + + /// The 512 wire bytes. + pub fn as_bytes(&self) -> &[u8; RECORD_BYTES] { + &self.bytes + } + + /// `(major, minor)`. + pub fn abi(&self) -> (u16, u16) { + (self.u16_at(off::ABI_MAJOR), self.u16_at(off::ABI_MINOR)) + } + + /// Schema family + version. + pub fn schema(&self) -> SchemaId { + SchemaId { + family: SchemaFamily::from_u16(self.u16_at(off::SCHEMA_FAMILY)).expect("validated"), + version: self.u16_at(off::SCHEMA_VERSION), + } + } + + /// Family-scoped kind code. + pub fn object_kind(&self) -> u16 { + self.u16_at(off::OBJECT_KIND) + } + /// Flags. + pub fn flags(&self) -> u16 { + self.u16_at(off::FLAGS) + } + /// Source object id. + pub fn node_guid(&self) -> Guid128 { + self.guid_at(off::NODE_GUID) + } + /// Domain / tenant id. + pub fn scope_guid(&self) -> Guid128 { + self.guid_at(off::SCOPE_GUID) + } + /// Observation time. + pub fn observed_at_ms(&self) -> i64 { + i64::from_le_bytes( + self.bytes[off::OBSERVED_AT..off::OBSERVED_AT + 8] + .try_into() + .expect("8"), + ) + } + + /// OU path; `None` unless [`FLAG_OU_PRESENT`]. + pub fn ou_hhtl(&self) -> Option { + if self.flags() & FLAG_OU_PRESENT == 0 { + return None; + } + let h: [u8; 16] = self.bytes[off::OU_HHTL..off::OU_HHTL + 16] + .try_into() + .expect("16"); + OuHhtl::from_le_bytes(&h).ok() + } + + /// Set the OU path (sets [`FLAG_OU_PRESENT`]). + pub fn set_ou_hhtl(&mut self, h: OuHhtl) { + self.bytes[off::OU_HHTL..off::OU_HHTL + 16].copy_from_slice(&h.to_le_bytes()); + self.add_flags(FLAG_OU_PRESENT); + } + + /// OR flags in. + pub fn add_flags(&mut self, f: u16) { + let v = self.flags() | f; + self.put_u16(off::FLAGS, v); + } + + fn presence(&self) -> u64 { + u64::from_le_bytes( + self.bytes[off::PRESENCE..off::PRESENCE + 8] + .try_into() + .expect("8"), + ) + } + fn set_presence_bit(&mut self, bit: usize) { + let v = self.presence() | (1u64 << bit); + self.bytes[off::PRESENCE..off::PRESENCE + 8].copy_from_slice(&v.to_le_bytes()); + } + + /// Set a pooled slot. + pub fn set_str(&mut self, slot: usize, r: StrRef) -> Result<(), RecordError> { + if slot >= STR_SLOTS { + return Err(RecordError::Slot(slot)); + } + let o = off::STR_REFS + slot * 8; + self.bytes[o..o + 8].copy_from_slice(&r.to_le_bytes()); + self.set_presence_bit(slot); + Ok(()) + } + + /// Read a pooled slot. + pub fn str_ref(&self, slot: usize) -> Option { + if slot >= STR_SLOTS || self.presence() & (1 << slot) == 0 { + return None; + } + let o = off::STR_REFS + slot * 8; + Some(StrRef::from_le_bytes( + self.bytes[o..o + 8].try_into().expect("8"), + )) + } + + /// Set a numeric slot. + pub fn set_num(&mut self, slot: usize, v: u32) -> Result<(), RecordError> { + if slot >= NUM_SLOTS { + return Err(RecordError::Slot(slot)); + } + let o = off::NUM + slot * 4; + self.bytes[o..o + 4].copy_from_slice(&v.to_le_bytes()); + self.set_presence_bit(STR_SLOTS + slot); + Ok(()) + } + + /// Read a numeric slot. + pub fn num(&self, slot: usize) -> Option { + if slot >= NUM_SLOTS || self.presence() & (1 << (STR_SLOTS + slot)) == 0 { + return None; + } + let o = off::NUM + slot * 4; + Some(u32::from_le_bytes( + self.bytes[o..o + 4].try_into().expect("4"), + )) + } + + /// Reserved bytes are zero (writer obligation). + pub fn reserved_is_zero(&self) -> bool { + self.bytes[off::CANONICAL_KEY..off::VALUE] + .iter() + .all(|&b| b == 0) + && self.bytes[off::RESERVED..off::END].iter().all(|&b| b == 0) + } +} diff --git a/crates/ogar-dir-core/src/schema.rs b/crates/ogar-dir-core/src/schema.rs new file mode 100644 index 0000000..356e9ac --- /dev/null +++ b/crates/ogar-dir-core/src/schema.rs @@ -0,0 +1,110 @@ +//! Schema identity, kept apart from ABI identity. +//! +//! * **ABI version** ([`crate::record::ABI_MAJOR`]/`ABI_MINOR`): how the 512 +//! bytes are laid out. Changes only when a byte offset changes meaning. +//! * **Schema family** ([`SchemaFamily`]): which source system's semantics +//! (`AD_DS` vs `MS_GRAPH`). Never flattened into one generic identity schema. +//! * **Schema version**: which attribute set the *encoder* understood. +//! +//! Schema evolution is append-only within a family: a new attribute takes the +//! next free slot and a higher `since` version; existing slots never change +//! meaning. A vN record therefore reads correctly under vN+1 (the new slots are +//! simply absent), and adding an attribute never touches the ABI. A slot whose +//! meaning must change is retired, not reused. +//! +//! A `SchemaGuid` was considered. OGAR has no schema-GUID convention today and +//! the (family, version) pair is enough to discriminate, so a GUID would be an +//! unanchored new identifier; it can be added later in reserved bytes. + +/// Source schema family. `u16` LE on the wire. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +#[repr(u16)] +pub enum SchemaFamily { + /// Active Directory Domain Services (LDAP). + AdDs = 1, + /// Microsoft Graph / Entra ID. + MsGraph = 2, +} + +impl SchemaFamily { + /// Decode, `None` for unknown families. + pub fn from_u16(v: u16) -> Option { + match v { + 1 => Some(Self::AdDs), + 2 => Some(Self::MsGraph), + _ => None, + } + } +} + +/// Family + encoder schema version. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct SchemaId { + /// Which source semantics. + pub family: SchemaFamily, + /// Which attribute set the encoder understood. + pub version: u16, +} + +/// How an attribute's value is stored. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum AttrKind { + /// UTF-8 string, pool slot. + Str, + /// Raw binary, pool slot. + Bytes, + /// Multi-valued UTF-8, pool slot. + MultiStr, + /// Unsigned 32-bit, numeric slot. + U32, + /// Boolean (0/1), numeric slot. Absence (presence bit clear) = unknown/null. + Bool, +} + +impl AttrKind { + /// True for pool-backed kinds (string slot space), false for numeric. + pub fn is_pooled(self) -> bool { + matches!(self, Self::Str | Self::Bytes | Self::MultiStr) + } +} + +/// One attribute the encoder understands. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct AttrDef { + /// Source-native attribute name, exactly as the source spells it + /// (`sAMAccountName`, `onPremisesDistinguishedName`). + pub name: &'static str, + /// Slot index within its slot space (pooled 0..32, numeric 0..4). + pub slot: u8, + /// Storage kind. + pub kind: AttrKind, + /// First schema version that defines this attribute. + pub since: u16, +} + +/// Structural check every schema table must pass: slots in range, no slot or +/// name used twice, `since` ≥ 1. Returns the first problem found. +pub fn validate(attrs: &[AttrDef]) -> Result<(), String> { + for (i, a) in attrs.iter().enumerate() { + let limit = if a.kind.is_pooled() { + crate::record::STR_SLOTS + } else { + crate::record::NUM_SLOTS + }; + if a.slot as usize >= limit { + return Err(format!("{}: slot {} out of range", a.name, a.slot)); + } + if a.since == 0 { + return Err(format!("{}: since must be >= 1", a.name)); + } + for b in &attrs[..i] { + if b.name == a.name { + return Err(format!("{}: duplicate name", a.name)); + } + if b.slot == a.slot && b.kind.is_pooled() == a.kind.is_pooled() { + return Err(format!("{} and {}: same slot {}", b.name, a.name, a.slot)); + } + } + } + Ok(()) +} diff --git a/crates/ogar-dir-core/tests/main.rs b/crates/ogar-dir-core/tests/main.rs new file mode 100644 index 0000000..327d99f --- /dev/null +++ b/crates/ogar-dir-core/tests/main.rs @@ -0,0 +1,224 @@ +//! Invariants 1–8, 10–12 of the directory-observation PoC (core side). + +use ogar_dir_core::record::{FLAG_OU_PRESENT, RECORD_BYTES, off}; +use ogar_dir_core::*; + +fn g(s: &str) -> Guid128 { + Guid128::parse(s).unwrap() +} + +const SCHEMA_AD1: SchemaId = SchemaId { + family: SchemaFamily::AdDs, + version: 1, +}; + +// 1. GUIDs survive exact 128-bit round trips — through the record bytes, not +// just through the string form. +#[test] +fn inv01_guid_round_trips_through_the_record() { + let node = g("0f1e2d3c-4b5a-6978-8796-a5b4c3d2e1f0"); + let scope = g("ffffffff-0000-ffff-0000-ffffffffffff"); + let r = DirRecord::new(SCHEMA_AD1, 1, node, scope, 42); + let back = DirRecord::from_bytes(r.as_bytes()).unwrap(); + assert_eq!(back.node_guid(), node); + assert_eq!(back.scope_guid(), scope); + assert_eq!(back.node_guid().0, node.0); + assert_eq!(&r.as_bytes()[off::NODE_GUID..off::NODE_GUID + 16], &node.0); +} + +// 2. Two GUIDs that differ only in their upper 64 bits stay distinct +// everywhere (no u64 truncation anywhere in the path). +#[test] +fn inv02_upper_64_bits_are_not_truncated() { + let a = g("00000000-0000-0001-0123-456789abcdef"); + let b = g("00000000-0000-0002-0123-456789abcdef"); + assert_eq!(a.0[8..], b.0[8..], "fixture: lower 64 bits identical"); + assert_ne!(a, b); + let ra = DirRecord::new(SCHEMA_AD1, 1, a, Guid128::NIL, 0); + let rb = DirRecord::new(SCHEMA_AD1, 1, b, Guid128::NIL, 0); + assert_ne!(ra.node_guid(), rb.node_guid()); + assert_ne!(ra.as_bytes(), rb.as_bytes()); + let set: std::collections::HashSet<_> = [ra.node_guid(), rb.node_guid()].into(); + assert_eq!(set.len(), 2); +} + +fn ou_dn(ous_leaf_first: &[&str]) -> String { + let mut s = String::from("CN=Erika Mustermann"); + for o in ous_leaf_first { + s.push_str(",OU="); + s.push_str(o); + } + s + ",DC=example,DC=de" +} + +// 3. A DN with 1–8 OU levels round-trips through HHTL + dictionary. +#[test] +fn inv03_one_to_eight_levels_round_trip() { + let names = [ + "Stuttgart", + "Infrastructure", + "Exchange", + "L4", + "L5", + "L6", + "L7", + "L8", + ]; + let mut dict = OuDictionary::new(); + for depth in 1..=8 { + let root_first: Vec<&str> = names[..depth].to_vec(); + let leaf_first: Vec<&str> = root_first.iter().rev().copied().collect(); + let dn = Dn::parse(&ou_dn(&leaf_first)).unwrap(); + assert_eq!(dn.ou_path_root_first(), root_first); + let h = dict.intern(&dn.ou_path_root_first()).unwrap(); + assert_eq!(h.depth(), depth); + assert_eq!(OuHhtl::from_le_bytes(&h.to_le_bytes()).unwrap(), h); + assert_eq!(dict.explain(&h).unwrap(), root_first); + assert_eq!(dict.resolve(&root_first), Some(h)); + } + // persisted dictionary reconstructs identically + let reloaded = OuDictionary::from_entries(&dict.entries()).unwrap(); + let h = dict.resolve(&names).unwrap(); + assert_eq!(reloaded.explain(&h).unwrap(), names); + // and keeps allocating after the reload without reusing an id + let mut reloaded = reloaded; + let sib = reloaded.intern(&["Stuttgart", "Finance"]).unwrap(); + assert_ne!(sib, dict.resolve(&["Stuttgart", "Infrastructure"]).unwrap()); + // nine is refused, never truncated + let nine: Vec<&str> = names.iter().copied().chain(["L9"]).collect(); + assert_eq!(dict.intern(&nine), Err(HhtlError::TooDeep(9))); +} + +// 4. DC= components never consume levels. 5. The leaf CN never does. +#[test] +fn inv04_05_dc_and_leaf_are_not_levels() { + let mut dict = OuDictionary::new(); + let a = Dn::parse("CN=X,OU=Sales,DC=a,DC=b,DC=c,DC=d").unwrap(); + let b = Dn::parse("CN=Y,OU=Sales,DC=z").unwrap(); + let ha = dict.intern(&a.ou_path_root_first()).unwrap(); + let hb = dict.intern(&b.ou_path_root_first()).unwrap(); + assert_eq!(ha.depth(), 1); + assert_eq!(ha, hb, "neither the DC count nor the leaf changes the HHTL"); + let none = Dn::parse("CN=Z,DC=example,DC=de").unwrap(); + assert_eq!( + dict.intern(&none.ou_path_root_first()).unwrap(), + OuHhtl::ROOT + ); + // An OU object's own name is its leaf, not a level of its location. + let ou_obj = Dn::parse("OU=Exchange,OU=Infrastructure,DC=x").unwrap(); + assert_eq!(ou_obj.ou_path_root_first(), ["Infrastructure"]); +} + +// 8. Segment allocation cannot collide. +#[test] +fn inv08_allocation_is_collision_free() { + let mut dict = OuDictionary::new(); + // 2000 siblings under one parent: a 16-bit hash would collide here + // (birthday bound ~300); the explicit dictionary never does. + let mut seen = std::collections::HashSet::new(); + for i in 0..2000 { + let h = dict.intern(&["Root", &format!("ou-{i}")]).unwrap(); + assert!(seen.insert(h), "collision at {i}"); + } + // case-insensitive: same OU under different spelling is the same segment + assert_eq!( + dict.intern(&["root", "OU-7"]).unwrap(), + dict.resolve(&["Root", "ou-7"]).unwrap() + ); + // same name under different parents: different paths + let x = dict.intern(&["A", "Exchange"]).unwrap(); + let y = dict.intern(&["B", "Exchange"]).unwrap(); + assert_ne!(x, y); + // corrupted persisted state (two names on one id) is refused + let mut e = dict.entries(); + let dup = (e[0].0, e[0].1, "Intruder".to_string()); + e.push(dup); + assert_eq!( + OuDictionary::from_entries(&e).unwrap_err(), + HhtlError::Collision + ); + // exhaustion is an error, never a wrap-around onto id 0 / id 1 + let mut d = OuDictionary::new(); + for i in 0..u16::MAX as u32 { + d.intern(&[format!("n{i}")]).unwrap(); + } + assert_eq!(d.intern(&["one-too-many"]), Err(HhtlError::ParentExhausted)); +} + +// 10. The fixed DTO is exactly 512 bytes, layout locked. +#[test] +fn inv10_record_is_512_bytes() { + assert_eq!(std::mem::size_of::(), 512); + assert_eq!(RECORD_BYTES, 0x200); + let r = DirRecord::new(SCHEMA_AD1, 1, Guid128::NIL, Guid128::NIL, 0); + assert_eq!(r.as_bytes().len(), 512); + assert_eq!(&r.as_bytes()[0x20..0x24], b"OGDR"); + assert!(r.reserved_is_zero()); + // canonical NodeRow split: key 16 | edges 16 | value 480 + assert_eq!( + (off::CANONICAL_EDGES, off::VALUE, 512 - off::VALUE), + (16, 32, 480) + ); +} + +// 11. A schema version change is not an ABI version change. +#[test] +fn inv11_schema_version_is_independent_of_abi() { + let v1 = DirRecord::new(SCHEMA_AD1, 1, Guid128::NIL, Guid128::NIL, 0); + let v2 = DirRecord::new( + SchemaId { + version: 2, + ..SCHEMA_AD1 + }, + 1, + Guid128::NIL, + Guid128::NIL, + 0, + ); + assert_eq!(v1.abi(), v2.abi()); + assert_ne!(v1.schema(), v2.schema()); + // only the schema_version bytes differ + let diff: Vec = (0..512) + .filter(|&i| v1.as_bytes()[i] != v2.as_bytes()[i]) + .collect(); + assert_eq!(diff, vec![off::SCHEMA_VERSION]); + // a foreign ABI major is rejected, a newer minor is accepted + let mut b = *v1.as_bytes(); + b[off::ABI_MAJOR] = 2; + assert_eq!(DirRecord::from_bytes(&b), Err(RecordError::AbiMajor(2))); + let mut b = *v1.as_bytes(); + b[off::ABI_MINOR] = 9; + assert!(DirRecord::from_bytes(&b).is_ok()); +} + +// The decoder refuses malformed HHTL bytes rather than misreading them. +#[test] +fn hhtl_gap_is_rejected_on_decode() { + let mut r = DirRecord::new(SCHEMA_AD1, 1, Guid128::NIL, Guid128::NIL, 0); + r.set_ou_hhtl(OuHhtl([1, 0, 3, 0, 0, 0, 0, 0])); + assert_eq!(r.flags() & FLAG_OU_PRESENT, FLAG_OU_PRESENT); + assert!(matches!( + DirRecord::from_bytes(r.as_bytes()), + Err(RecordError::Hhtl(HhtlError::Gap(2))) + )); +} + +#[test] +fn edge_record_round_trips_and_keeps_endpoints_distinct() { + let e = DirEdge { + kind: EdgeKind::SynchronizesTo, + evidence: EdgeEvidence::ImmutableIdIsObjectGuid, + src: ( + SchemaFamily::AdDs, + g("11111111-1111-1111-1111-111111111111"), + ), + dst: ( + SchemaFamily::MsGraph, + g("22222222-2222-2222-2222-222222222222"), + ), + }; + let b = e.to_bytes(); + assert_eq!(b.len(), 64); + assert_eq!(DirEdge::from_bytes(&b), Some(e)); + assert_ne!(e.src.1, e.dst.1); +} diff --git a/docs/DIRECTORY-ADAPTERS-POC.md b/docs/DIRECTORY-ADAPTERS-POC.md new file mode 100644 index 0000000..4e728ed --- /dev/null +++ b/docs/DIRECTORY-ADAPTERS-POC.md @@ -0,0 +1,177 @@ +# Directory adapters PoC — `ogar-ad` / `ogar-az` + +Status: **PoC, 2026-10-03.** Read-only encoding of observed Active Directory +and Entra ID objects into versioned, source-native, fixed-size records. Not an +IAM, not provisioning, not reconciliation, no business rules. + +Crates: `ogar-dir-core` (shared ABI, zero deps) · `ogar-ad` (AD DS / LDIF) · +`ogar-az` (Microsoft Graph). Same footing as `ogar-doc-ir`: neutral tissue, +**no classid mint, no canon dependency.** + +## 1. Existing machinery found, and what was reused + +| Existing | Where | Used? | +|---|---|---| +| `NodeRow` 512 B = key 16 / edges 16 / value 480 | lance-graph-contract `canonical_node.rs` | **Shape reused**: `DirRecord` keeps the identical split; the directory payload lives in the 480-byte value slab. | +| `NodeGuid` 16 B key (`classid · HEEL · HIP · TWIG · tail`) | same | **Not reused for identity** — see conflict C1. | +| `NiblePath` (16ⁿ nibble router, u64) / HEEL·HIP·TWIG (3×u16) | `hhtl.rs`, `canonical_node.rs` | Not reused — see C2. | +| `ogar-auth` `AuthBinding(provider, subject)` | OGAR `ogar-auth/src/user.rs` | Not touched. It answers "which login maps to which local user"; this PoC answers "what does the directory say". A later bridge may emit `AuthBinding`s from AZ records. | +| `ogar-vocab` Auth domain `0x0B` | OGAR `ogar-vocab` | No mint (C3). | +| `CausalEdge64` | lance-graph-contract | Does not fit: 8 bytes, no 128-bit endpoints. | +| GUID-pair edge record, schema-version machinery, LDAP/Graph code | — | **None exists**; defined minimally here. | + +## 2. Crate boundaries + +``` +ogar-dir-core Guid128, Dn, OuHhtl + OuDictionary, ValuePool/StrRef, + SchemaFamily/SchemaId/AttrDef, DirRecord (512 B), DirEdge (64 B) + ├── ogar-ad AD SCHEMA_V1, AdKind, LDIF reader, encode(entry) + └── ogar-az Graph SCHEMA_V1, $select derivation, ingest_page, sync_edges +``` + +`ogar-ad` and `ogar-az` do not depend on each other. The AD↔AZ relation is +computed from records (`ogar_az::sync_edges` takes a set of observed AD GUIDs). + +## 3. `DirRecord` — exact 512-byte layout + +All multi-byte integers little-endian. The record is one `[u8; 512]` +(`#[repr(C, align(64))]`) addressed by constant offsets, so there is no +compiler padding to reason about. Size/alignment/offset sums are `const` +asserts. + +``` +0x000..0x010 canonical_key [16] NodeRow key slot — ZERO in the PoC (dormant default class) +0x010..0x020 canonical_edges [16] NodeRow edge-facet slot — ZERO; relations are DirEdge +---- NodeRow value slab (480 B) ---- +0x020..0x024 magic [4] "OGDR" +0x024..0x026 abi_major u16 1 (exact match required) +0x026..0x028 abi_minor u16 0 (additive; readers ignore newer) +0x028..0x02A schema_family u16 1 = AD_DS, 2 = MS_GRAPH +0x02A..0x02C schema_version u16 encoder's attribute-set version +0x02C..0x02E object_kind u16 family-scoped (AD: user/group/computer/contact/OU; AZ: user) +0x02E..0x030 flags u16 OU_PRESENT | NON_OU_CONTAINER | DN_UNENCODED +0x030..0x040 node_guid [16] WHO objectGUID / Graph id, textual byte order +0x040..0x050 scope_guid [16] WHERE AD domain GUID / Entra tenant id +0x050..0x060 ou_hhtl [16] WHERE 8 × u16 LE, root first +0x060..0x068 observed_at_ms i64 WHEN observation time (provenance hook) +0x068..0x070 presence u64 bits 0..31 pooled slots, 32..35 numeric slots +0x070..0x080 num 4×u32 numeric/bool slots +0x080..0x180 str_refs 32 × (u32 off, u32 len) into the batch ValuePool +0x180..0x200 reserved [128] writers zero, readers ignore +``` + +**GUID byte order:** textual order (the order of the 32 hex digits, RFC 9562 +network order). LDAP `objectGUID` and `onPremisesImmutableId` are Microsoft +mixed-endian (`Data1..3` little-endian); `Guid128::from_ms_bytes` converts, +tested against the .NET `ToByteArray()` pair. No `u128`, no truncation API. + +**Strings** never live in the record. A `StrRef` points into a per-batch +`ValuePool` (the column payload). Multi-valued slots are `(u32 len, bytes)*`. +Values are raw as observed; nothing is folded or normalised. Derived values +(enabled-from-UAC, canonical SMTP) are functions over records, never stored. + +**Stability:** byte offsets, endianness, GUID order and padding are fixed and +asserted. Not yet claimed: a non-Rust reader (offsets are exported in +`record::off` for generating one) and the NodeRow mirror guard (C4). + +## 4. OU-HHTL and the OU dictionary + +`OuHhtl = [u16; 8]`, root first; `0` = no level; depth = leading non-zero +levels; a gap is rejected on decode; > 8 OUs is an error and the record gets +`DN_UNENCODED` (never a truncated path). Only `OU=` components are levels — +not the leaf, not `DC=`. An OU object's own `OU=` is its leaf. + +Segment ids come from an explicit **per-parent dictionary** (`OuDictionary`, +one per AD domain), allocated sequentially from 1 on first sight. No hashing: +collisions are impossible by construction, exhaustion (65535 children of one +OU) is an error, and `explain` reconstructs the original spelling. Matching is +case-insensitive via Unicode lowercase (an approximation of AD collation). + +Price of reversibility: ids depend on first-seen order, so the dictionary is +state persisted next to the records (`entries` / `from_entries`, which +rejects duplicate ids or names). Containers that are not OUs (`CN=Users`) do +not enter the HHTL; the record is flagged `NON_OU_CONTAINER` so it is not +confused with "directly under the domain root". **Open question:** whether +such containers should become levels. + +For AZ, `onPremisesDistinguishedName` is parsed the same way into the +*on-premises domain's* dictionary. With a shared dictionary the AD record and +the AZ evidence of one OU carry the identical HHTL (tested). For an AZ record +that HHTL is evidence about the on-prem location, not a Graph location. + +## 5. Schema / version strategy + +Three independent numbers: **ABI version** (how bytes are laid out), +**schema family** (whose semantics), **schema version** (which attribute set +the encoder understood). Each adapter embeds `SCHEMA_V1: &[AttrDef]` +(`name, slot, kind, since`). Evolution is append-only: a new attribute takes +the next free slot with a higher `since`; a slot never changes meaning (retire, +don't reuse). Old records read under a newer schema with the new slots absent; +no ABI change. Graph `$select` is derived from the table (`select_query`). +A `SchemaGuid` was considered and not adopted: OGAR has no such convention and +(family, version) discriminates; reserved bytes allow adding one later. + +## 6. Minimal attribute sets + +**AD v1:** `distinguishedName, objectSid (bytes), sAMAccountName, +userPrincipalName, objectClass (multi), displayName, givenName, sn, mail, +mailNickname, proxyAddresses (multi), targetAddress, whenCreated, whenChanged`, +numeric `userAccountControl`. Identity `objectGUID`; scope = domain GUID. +Membership attributes are relations (future edges), `msExch*` deferred. + +**Graph v1:** `userPrincipalName, displayName, givenName, surname, mail, +mailNickname, proxyAddresses, employeeId, department, companyName, +officeLocation, userType, onPremisesDistinguishedName, +onPremisesSamAccountName, onPremisesDomainName, onPremisesImmutableId, +onPremisesSecurityIdentifier, onPremisesLastSyncDateTime`, bools +`accountEnabled, onPremisesSyncEnabled` (null = absent = unknown, never +false). Identity `id`; scope = tenant id (supplied by the caller — it is not a +user property). Licences, sign-in activity and groups are out of scope. + +Unknown attributes are reported (`ignored`) and never stored: they cannot +reach the record or the pool (tested with 200 synthetic unknowns). + +## 7. Edges + +`DirEdge`, 64 bytes: `kind u16, evidence u16, src_family u16, dst_family u16, +src_guid[16], dst_guid[16]`, rest zero. One kind: `SynchronizesTo` (AD → AZ), +one evidence: `ImmutableIdIsObjectGuid` (the Graph immutable id base64-decodes +to an **observed** AD objectGUID). The two nodes stay distinct; non-GUID +anchors and unobserved targets yield no edge. `MEMBER_OF` / `MANAGER` get +codes when a slice emits them. + +## 8. Conflicts with existing OGAR / HHTL contracts + +* **C1 — "NodeGuid" means two things.** In OGAR canon the 16-byte `NodeGuid` + is a *minted address* (`classid` + HHTL tiers + facet) that prerenders a node + from the key. A directory `objectGUID` is an opaque external identity and + cannot be placed there without dropping bits. Resolution here: the source id + is `Guid128` in the value slab; the canonical key slot is reserved zero. A + future mint of a canonical key for directory nodes (classid + a 12-byte + facet derived *from* the record) is an operator decision and must not be a + truncated GUID. +* **C2 — HHTL width.** Canon HHTL is 3 tiers × 16 bits in the key, and the + contract `NiblePath` is a 16-ary nibble path. The requested OU-HHTL is + 8 × 16-bit levels with per-parent dictionary ids — a different, wider tree + living in the value slab. It is named `OuHhtl` to keep it from being + mistaken for the canonical HEEL/HIP/TWIG path. +* **C3 — No classid.** Directory concepts would naturally sit in the Auth + domain (`0x0B`) of `ogar-vocab`. Minting is a codebook change and was not + done; `object_kind` is a family-scoped source-native code. +* **C4 — Unguarded mirror.** `DirRecord`'s 16/16/480 split matches `NodeRow` + by documentation and local asserts only; the crate deliberately does not + depend on lance-graph-contract. A feature-gated `From for + NodeRow` with a size assert would turn it into a real guard. + +## 9. Ingest path for a lab tenant (read-only) + +```sh +URL=$(cargo run -q -p ogar-az --example az_ingest -- --url) # schema-derived $select +curl -s -H "Authorization: Bearer $TOKEN" "$URL" > page1.json # User.Read.All +cargo run -p ogar-az --example az_ingest -- page1.json +``` + +The crate performs no HTTP and no writes; follow `next_link` for more pages. +**Not yet run against a real tenant** (no credentials in the build +environment); verified against a synthetic Graph page fixture only. AD side: +`ldapsearch -LLL … objectGUID …` / `ldifde` output through `ogar_ad::ldif::parse`. From 8f8158a1d72a413b4c8a3d9e321f1c33aa168600 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 01:20:56 +0000 Subject: [PATCH 2/2] ogar-ad/ogar-az: accept ldifde changetype: add; attr_str refuses multi-values Bugbot on #313: - ldifde -f writes 'changetype: add' after every dn; it is the export marker, not a change record. Accept it; still refuse modify/delete/modrdn. - attr_str matched any pooled kind, so a MultiStr slot came back as its length-prefixed blob. Restrict it to AttrKind::Str and add attr_multi. Both covered by new tests that failed before the fix. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G22yT6htkcdyXsihxxXdrg --- crates/ogar-ad/src/ldif.rs | 12 ++++++++++-- crates/ogar-ad/tests/main.rs | 29 +++++++++++++++++++++++++++++ crates/ogar-az/src/lib.rs | 18 ++++++++++++++++-- crates/ogar-az/tests/main.rs | 21 +++++++++++++++++++++ 4 files changed, 76 insertions(+), 4 deletions(-) diff --git a/crates/ogar-ad/src/ldif.rs b/crates/ogar-ad/src/ldif.rs index 575526d..36faee4 100644 --- a/crates/ogar-ad/src/ldif.rs +++ b/crates/ogar-ad/src/ldif.rs @@ -4,8 +4,9 @@ //! Supported: `#` comments, a leading `version:` line, folded lines (a line //! starting with one space continues the previous one), `attr: value`, //! `attr:: base64` (binary or non-ASCII values — this is how `objectGUID` -//! arrives), blank-line entry separation. Rejected: `attr:< URL` and -//! change records (`changetype:`) — this is an observation reader. +//! arrives), blank-line entry separation, and `changetype: add` (which +//! `ldifde -f` emits on every entry). Rejected: `attr:< URL` and real change +//! records (`modify` / `delete` / `modrdn`) — this is an observation reader. use crate::AdEntry; use ogar_dir_core::base64; @@ -64,6 +65,13 @@ pub fn parse(text: &str) -> Result, LdifError> { continue; } if name.eq_ignore_ascii_case("changetype") { + // `ldifde -f` writes `changetype: add` after every dn: that is the + // export marker, i.e. an observation. Real change records are not. + if cur.is_some() + && std::str::from_utf8(&value).is_ok_and(|v| v.trim().eq_ignore_ascii_case("add")) + { + continue; + } return Err(LdifError::Unsupported(ln)); } match cur.as_mut() { diff --git a/crates/ogar-ad/tests/main.rs b/crates/ogar-ad/tests/main.rs index 582c521..7160beb 100644 --- a/crates/ogar-ad/tests/main.rs +++ b/crates/ogar-ad/tests/main.rs @@ -196,3 +196,32 @@ fn too_deep_dn_is_flagged_not_truncated() { assert_ne!(r.flags() & ogar_dir_core::record::FLAG_DN_UNENCODED, 0); assert!(d.is_empty(), "nothing interned for a refused path"); } + +// Bugbot #313: `ldifde -f` emits `changetype: add` after every dn. That is the +// export marker, not a change record — it must be accepted. Real change +// records (modify/delete/modrdn) are still refused. +#[test] +fn ldifde_changetype_add_is_accepted_other_changetypes_refused() { + let ldifde = "dn: CN=E,OU=Exchange,DC=x\nchangetype: add\nobjectGUID:: 4AQlP4lP0xGaDAMF6CwzAQ==\nobjectClass: user\n"; + let e = ldif::parse(ldifde).unwrap(); + assert_eq!(e.len(), 1); + assert!( + e[0].attrs + .iter() + .all(|(n, _)| !n.eq_ignore_ascii_case("changetype")) + ); + let (mut d, mut p) = (OuDictionary::new(), ValuePool::new()); + assert!( + encode(&e[0], domain(), &mut d, &mut p, 0) + .unwrap() + .ignored + .is_empty() + ); + for ct in ["modify", "delete", "modrdn", "moddn"] { + let text = format!("dn: CN=E,DC=x\nchangetype: {ct}\n"); + assert!( + matches!(ldif::parse(&text), Err(ldif::LdifError::Unsupported(2))), + "{ct}" + ); + } +} diff --git a/crates/ogar-az/src/lib.rs b/crates/ogar-az/src/lib.rs index fd7a43d..60523e3 100644 --- a/crates/ogar-az/src/lib.rs +++ b/crates/ogar-az/src/lib.rs @@ -343,14 +343,28 @@ pub fn ingest_page( }) } -/// Read a string slot by attribute name. +/// Read a single-valued string slot by attribute name. `None` for absent +/// values and for attributes that are not [`AttrKind::Str`] (a multi-valued +/// slot is a length-prefixed blob, never a string — use [`attr_multi`]). pub fn attr_str<'p>(rec: &DirRecord, pool: &'p ValuePool, name: &str) -> Option<&'p str> { let def = SCHEMA_V1 .iter() - .find(|d| d.name == name && d.kind.is_pooled())?; + .find(|d| d.name == name && d.kind == AttrKind::Str)?; std::str::from_utf8(pool.get(rec.str_ref(def.slot as usize)?)?).ok() } +/// Read a multi-valued string slot by attribute name. `None` for absent +/// values and for attributes that are not [`AttrKind::MultiStr`]. +pub fn attr_multi<'p>(rec: &DirRecord, pool: &'p ValuePool, name: &str) -> Option> { + let def = SCHEMA_V1 + .iter() + .find(|d| d.name == name && d.kind == AttrKind::MultiStr)?; + pool.get_multi(rec.str_ref(def.slot as usize)?)? + .into_iter() + .map(|v| std::str::from_utf8(v).ok()) + .collect() +} + /// `SynchronizesTo` edges for AZ records whose `onPremisesImmutableId` /// base64-decodes to the `objectGUID` of an **observed** AD object. /// diff --git a/crates/ogar-az/tests/main.rs b/crates/ogar-az/tests/main.rs index eddd3ec..08b9b29 100644 --- a/crates/ogar-az/tests/main.rs +++ b/crates/ogar-az/tests/main.rs @@ -120,3 +120,24 @@ fn malformed_pages_are_refused() { .is_err() ); } + +// Bugbot #313: `attr_str` must not hand back the length-prefixed blob of a +// multi-valued slot as if it were a string. +#[test] +fn attr_str_refuses_multi_values_and_attr_multi_reads_them() { + let (mut dict, mut pool) = (OuDictionary::new(), ValuePool::new()); + let page = ingest_page(PAGE, tenant(), &mut dict, &mut pool, 99).unwrap(); + assert_eq!(attr_str(&page.records[0], &pool, "proxyAddresses"), None); + assert_eq!( + ogar_az::attr_multi(&page.records[0], &pool, "proxyAddresses").unwrap(), + [ + "SMTP:erika.mueller@example.de", + "smtp:emueller@example.mail.onmicrosoft.com" + ] + ); + assert_eq!( + ogar_az::attr_multi(&page.records[1], &pool, "proxyAddresses").unwrap(), + Vec::<&str>::new() + ); + assert_eq!(ogar_az::attr_multi(&page.records[0], &pool, "mail"), None); +}