From 1c43faa5755daa07ed9bd91a48a28f7e7a294326 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 8 Oct 2026 16:25:32 +0000 Subject: [PATCH 1/4] =?UTF-8?q?quack:=20fold=5Fjoin=5Fprobe=20(D-RPF-9)=20?= =?UTF-8?q?=E2=80=94=20which=20lowering=20already=20avoids=20the=20interme?= =?UTF-8?q?diate?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A probe, not a primitive. Every arm runs the shipped mask-risc executor or quack's lowering into it, against a per-row oracle, over 64K rows: resident masks folded directly, predicate chains (ungated and gated), a resident mask gating a predicate, the Boolean variants, quack's lower/lower_fused, and CE64 fields read in place from ValueTenant::MaterializedEdges against the canonical CausalEdge64 accessors. Edge cases, a demanded bitmap, a reused intermediate and a merge-law refusal are asserted; timings are printed only. causal-edge becomes an example-only dev-dependency of quack, so the CE64 arm is diffed against the real accessors rather than a restated layout. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01EFw2WdKr1oxvaKCJC2ua2R --- crates/lance-graph-quack/Cargo.toml | 4 + .../examples/fold_join_probe.rs | 1094 +++++++++++++++++ 2 files changed, 1098 insertions(+) create mode 100644 crates/lance-graph-quack/examples/fold_join_probe.rs diff --git a/crates/lance-graph-quack/Cargo.toml b/crates/lance-graph-quack/Cargo.toml index 8ed232759..590bb4178 100644 --- a/crates/lance-graph-quack/Cargo.toml +++ b/crates/lance-graph-quack/Cargo.toml @@ -21,3 +21,7 @@ lance-graph-contract = { path = "../lance-graph-contract" } # Round 6 (`tests/spofc_cycle.rs`): the SPOFC truth projection is the shipped # `arm_to_truth_u8`, reused rather than restated. Zero-dep, test-only. lance-graph-arm-discovery = { path = "../lance-graph-arm-discovery" } +# D-RPF-9 (`examples/fold_join_probe.rs`): the CE64 arm is diffed against the +# canonical `CausalEdge64` accessors, never against a re-derived bit layout. +# Zero-dep, example-only. +causal-edge = { path = "../causal-edge" } diff --git a/crates/lance-graph-quack/examples/fold_join_probe.rs b/crates/lance-graph-quack/examples/fold_join_probe.rs new file mode 100644 index 000000000..7a37864b8 --- /dev/null +++ b/crates/lance-graph-quack/examples/fold_join_probe.rs @@ -0,0 +1,1094 @@ +//! D-RPF-9 — Fold-Join deforestation probe: when two address-aligned +//! membership sources meet in one terminal, which existing lowering already +//! avoids the intermediate, and what does each arm actually write? +//! +//! Nothing here is a production primitive. Every arm runs the SHIPPED +//! `lance-graph-mask-risc` executor (or `lance-graph-quack`'s lowering into +//! it). The one probe-local loop (`H`) is the row-at-a-time oracle every arm +//! is diffed against; its timing is printed as a scalar reference, never as +//! evidence about a backend. +//! +//! Arms, all answering `COUNT(A ∧ B)` (or the named Boolean variant): +//! +//! | arm | execution | +//! |---|---| +//! | A | `Keep(A)` and `Keep(B)` into caller masks, then `And → Count` over them | +//! | B | two RESIDENT masks, `And → Count` (`Lowering::Ternlog`, no slot) | +//! | C | `Pred A`, `Pred B`, `And`, `Count` in one program (tiled) | +//! | D | `Pred A`, then `Pred B under A`, `Count` (tiled, accumulator gate) | +//! | Dp | RESIDENT mask A gates `Pred B` (`under: Plane`), `Count` | +//! | E | `AndNot` / `Xor` / three-input `Ternlog` over resident masks | +//! | F | `lance-graph-quack` `lower` / `lower_fused` of `A AND B` | +//! | H | per-row oracle (no `ndarray`, no mask) | +//! +//! The CE64 section runs C and D over `ValueTenant::MaterializedEdges` read in +//! place through `Pred::MatchFacet16Strided` and diffs them against the +//! canonical `causal_edge::CausalEdge64` accessors. +//! +//! ```text +//! CARGO_PROFILE_RELEASE_DEBUG=0 cargo run --release -p lance-graph-quack --example fold_join_probe +//! ``` +//! +//! Timings are printed, never asserted. Every count is asserted. + +use std::alloc::{GlobalAlloc, Layout, System}; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::time::Instant; + +use causal_edge::CausalEdge64; +use lance_graph_contract::canonical_node::{ValueTenant, NODE_ROW_STRIDE, VALUE_SLAB_ROW_OFFSET}; +use lance_graph_mask_risc::exec::{execute_extent, execute_into, Scratch}; +use lance_graph_mask_risc::{ + tile_words_for, words_for, ExecError, Foreign, LaneRef, Lowering, MaskOp, Operand, Out, Planes, + Pred, Program, StridedRef, Terminal, Value, TILE_WORDS, +}; +use lance_graph_quack::{lower, lower_fused, Agg, Cmp, Col, Filter, Query}; + +// ───────────────────────── allocation counter ───────────────────────── + +struct Counting; +static ALLOCS: AtomicUsize = AtomicUsize::new(0); +static BYTES: AtomicUsize = AtomicUsize::new(0); + +// SAFETY: a pure pass-through to `System`; the counters are the only addition. +unsafe impl GlobalAlloc for Counting { + unsafe fn alloc(&self, layout: Layout) -> *mut u8 { + ALLOCS.fetch_add(1, Ordering::Relaxed); + BYTES.fetch_add(layout.size(), Ordering::Relaxed); + // SAFETY: same layout, same contract as the caller's. + unsafe { System.alloc(layout) } + } + unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) { + // SAFETY: `ptr` came from `alloc` above with this `layout`. + unsafe { System.dealloc(ptr, layout) } + } +} + +#[global_allocator] +static GLOBAL: Counting = Counting; + +fn alloc_snapshot() -> (usize, usize) { + ( + ALLOCS.load(Ordering::Relaxed), + BYTES.load(Ordering::Relaxed), + ) +} + +// ───────────────────────── data ───────────────────────── + +/// SplitMix64. An LCG's low bytes are periodic, and the first version of +/// this probe used them: every field came out uniform to the row, so a care +/// shifted by one bit matched exactly as often as the right one and the +/// can-fire check could not fire. +struct Rng(u64); +impl Rng { + fn next(&mut self) -> u64 { + self.0 = self.0.wrapping_add(0x9E37_79B9_7F4A_7C15); + let mut z = self.0; + z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9); + z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB); + z ^ (z >> 31) + } +} + +const SCALE: i32 = 1_000_000; + +/// Two `i32` lanes. `x < sel_a·SCALE` is predicate A, `y < sel_b·SCALE` is B. +/// `clustered` sorts `x` so A's survivors form one run (dead words become dead +/// tiles); otherwise survivors are scattered. +fn lanes(n: usize, seed: u64, clustered: bool) -> (Vec, Vec) { + let mut r = Rng(seed); + let mut x: Vec = (0..n).map(|_| (r.next() % SCALE as u64) as i32).collect(); + let y: Vec = (0..n).map(|_| (r.next() % SCALE as u64) as i32).collect(); + if clustered { + x.sort_unstable(); + } + (x, y) +} + +fn thresh(sel: f64) -> i32 { + (sel * f64::from(SCALE)) as i32 +} + +/// The oracle: per row, no mask, no `ndarray`. +fn oracle(x: &[i32], y: &[i32], ta: i32, tb: i32) -> usize { + x.iter() + .zip(y) + .filter(|(a, b)| **a < ta && **b < tb) + .count() +} + +fn bits_of(lane: &[i32], t: i32) -> Vec { + let mut m = vec![0u64; words_for(lane.len())]; + for (i, v) in lane.iter().enumerate() { + if *v < t { + m[i / 64] |= 1 << (i % 64); + } + } + m +} + +fn dead_fractions(m: &[u64]) -> (f64, f64) { + let dead_words = m.iter().filter(|w| **w == 0).count(); + let tiles: Vec<&[u64]> = m.chunks(TILE_WORDS).collect(); + let dead_tiles = tiles.iter().filter(|t| t.iter().all(|w| *w == 0)).count(); + ( + dead_words as f64 / m.len().max(1) as f64, + dead_tiles as f64 / tiles.len().max(1) as f64, + ) +} + +// ───────────────────────── one arm ───────────────────────── + +struct Report { + name: &'static str, + lowering: String, + predicates: usize, + mask_passes: usize, + scratch_bytes: usize, + population_bytes: usize, + allocs_per_exec: f64, + ns: f64, +} + +fn lowering_name(p: &Program) -> String { + match p.compile().lowering() { + Lowering::Range(_) => "Range".into(), + Lowering::Ternlog(_) => "Ternlog(no slot)".into(), + Lowering::TernlogKeep(_) => "TernlogKeep".into(), + Lowering::Tern2(_) => "Tern2".into(), + Lowering::Tern3(_) => "Tern3".into(), + Lowering::Tiled => "Tiled".into(), + } +} + +fn count(v: Result) -> usize { + match v { + Ok(Value::Count(c)) => c, + other => panic!("expected a Count, got {other:?}"), + } +} + +/// Run `p` `reps` times over `planes` with a pre-built scratch; return the +/// count and a [`Report`] (scratch is sized BEFORE the timer and printed). +fn run_arm( + name: &'static str, + p: &Program, + planes: &Planes<'_>, + reps: usize, + population_bytes: usize, +) -> (usize, Report) { + let mut scratch = Scratch::for_program(p, planes.n_rows).expect("scratch"); + let scratch_bytes = p.scratch_slots as usize * tile_words_for(planes.n_rows) * 8; + let first = count(execute_into( + p, + planes, + &Foreign::NONE, + &mut scratch, + Out::None, + )); + let (a0, _) = alloc_snapshot(); + let t = Instant::now(); + let mut c = 0; + for _ in 0..reps { + c = count(execute_into( + p, + planes, + &Foreign::NONE, + &mut scratch, + Out::None, + )); + } + let ns = t.elapsed().as_nanos() as f64 / reps as f64; + let (a1, _) = alloc_snapshot(); + assert_eq!(c, first); + let h = p.op_histogram(); + ( + c, + Report { + name, + lowering: lowering_name(p), + predicates: h.predicates, + mask_passes: h.mask_passes(), + scratch_bytes, + population_bytes, + allocs_per_exec: (a1 - a0) as f64 / reps as f64, + ns, + }, + ) +} + +fn print(r: &Report) { + println!( + " {:<4} {:<17} preds {} passes {} scratch {:>6} B population-intermediate {:>6} B allocs/exec {:.1} {:>9.0} ns", + r.name, + r.lowering, + r.predicates, + r.mask_passes, + r.scratch_bytes, + r.population_bytes, + r.allocs_per_exec, + r.ns + ); +} + +// ───────────────────────── programs ───────────────────────── + +fn pred_a(ta: i32, under: Option, dst: u16) -> MaskOp { + MaskOp::Pred { + pred: Pred::LtI32 { lane: 0, t: ta }, + under, + dst, + } +} +fn pred_b(tb: i32, under: Option, dst: u16) -> MaskOp { + MaskOp::Pred { + pred: Pred::LtI32 { lane: 1, t: tb }, + under, + dst, + } +} +fn count_of(m: Operand) -> Terminal { + Terminal::Count { mask: m } +} +fn keep_of(m: Operand) -> Terminal { + Terminal::Keep { mask: m } +} + +/// Arm A, all three steps: two `Keep`s into caller masks, then the fold. +/// Returns the count; `ma`/`mb` are the materialised population masks. +/// The three programs arm A runs, built once outside the timer. +struct ArmA { + keep_a: Program, + keep_b: Program, + fold: Program, +} + +impl ArmA { + fn new(ta: i32, tb: i32) -> Self { + Self { + keep_a: Program::new(vec![pred_a(ta, None, 0)], keep_of(Operand::Scratch(0))), + keep_b: Program::new(vec![pred_b(tb, None, 0)], keep_of(Operand::Scratch(0))), + fold: Program::new( + vec![MaskOp::And { + a: Operand::Plane(0), + b: Operand::Plane(1), + dst: 0, + }], + count_of(Operand::Scratch(0)), + ), + } + } + + /// All three steps: two `Keep`s into the caller's masks, then the fold. + #[allow(clippy::too_many_arguments)] + fn run( + &self, + lanes: &[LaneRef<'_>], + n: usize, + ma: &mut [u64], + mb: &mut [u64], + sa: &mut Scratch<'_>, + sb: &mut Scratch<'_>, + sf: &mut Scratch<'_>, + ) -> usize { + let none: [&[u64]; 0] = []; + let planes = Planes { + n_rows: n, + masks: &none, + lanes, + }; + execute_into(&self.keep_a, &planes, &Foreign::NONE, sa, Out::Mask(ma)).expect("keep A"); + execute_into(&self.keep_b, &planes, &Foreign::NONE, sb, Out::Mask(mb)).expect("keep B"); + let both: [&[u64]; 2] = [ma, mb]; + let fold_planes = Planes { + n_rows: n, + masks: &both, + lanes: &[], + }; + count(execute_into( + &self.fold, + &fold_planes, + &Foreign::NONE, + sf, + Out::None, + )) + } +} + +// ───────────────────────── sections ───────────────────────── + +/// The selectivity sweep: every arm, sparse to dense, scattered and clustered. +fn sweep(n: usize, reps: usize) { + println!("\n== sweep: COUNT(x < a AND y < b), n = {n}, B selectivity 0.5 =="); + let tb = thresh(0.5); + for clustered in [false, true] { + let (x, y) = lanes(n, 0xF01D, clustered); + let lanes_ = [LaneRef::I32(&x), LaneRef::I32(&y)]; + for sel in [0.001, 0.01, 0.1, 0.5, 0.9] { + let ta = thresh(sel); + let want = oracle(&x, &y, ta, tb); + let ma = bits_of(&x, ta); + let mb = bits_of(&y, tb); + let (dw, dt) = dead_fractions(&ma); + println!( + "\n -- {} A sel {sel}: oracle {want}, A dead-word {:.3}, dead-tile {:.3}", + if clustered { "clustered" } else { "scattered" }, + dw, + dt + ); + let words = words_for(n); + let none: [&[u64]; 0] = []; + let lane_planes = Planes { + n_rows: n, + masks: &none, + lanes: &lanes_, + }; + let resident: [&[u64]; 2] = [&ma, &mb]; + let res_planes = Planes { + n_rows: n, + masks: &resident, + lanes: &lanes_, + }; + + // A — materialise both masks, then fold. + { + let arm = ArmA::new(ta, tb); + let mut sa = Scratch::for_program(&arm.keep_a, n).unwrap(); + let mut sb = Scratch::for_program(&arm.keep_b, n).unwrap(); + let mut sf = Scratch::for_program(&arm.fold, n).unwrap(); + let mut bufa = vec![0u64; words]; + let mut bufb = vec![0u64; words]; + let c0 = arm.run(&lanes_, n, &mut bufa, &mut bufb, &mut sa, &mut sb, &mut sf); + let (a0, _) = alloc_snapshot(); + let t = Instant::now(); + let mut c = 0; + for _ in 0..reps { + c = arm.run(&lanes_, n, &mut bufa, &mut bufb, &mut sa, &mut sb, &mut sf); + } + let ns = t.elapsed().as_nanos() as f64 / reps as f64; + let (a1, _) = alloc_snapshot(); + assert_eq!(c, want); + assert_eq!(c0, want); + print(&Report { + name: "A", + lowering: "Keep,Keep,Ternlog".into(), + predicates: 2, + mask_passes: 1, + scratch_bytes: 2 * tile_words_for(n) * 8, + population_bytes: 2 * words * 8, + allocs_per_exec: (a1 - a0) as f64 / reps as f64, + ns, + }); + } + // B — two resident masks, one fused fold. + let pb = Program::new( + vec![MaskOp::And { + a: Operand::Plane(0), + b: Operand::Plane(1), + dst: 0, + }], + count_of(Operand::Scratch(0)), + ); + let (cb, rb) = run_arm("B", &pb, &res_planes, reps, 0); + assert_eq!(cb, want); + assert_eq!(rb.lowering, "Ternlog(no slot)"); + print(&rb); + // C — two predicates, And, Count. + let pc = Program::new( + vec![ + pred_a(ta, None, 0), + pred_b(tb, None, 1), + MaskOp::And { + a: Operand::Scratch(0), + b: Operand::Scratch(1), + dst: 2, + }, + ], + count_of(Operand::Scratch(2)), + ); + let (cc, rc) = run_arm("C", &pc, &lane_planes, reps, 0); + assert_eq!(cc, want); + print(&rc); + // D — B under the accumulated A. + let pd = Program::new( + vec![ + pred_a(ta, None, 0), + pred_b(tb, Some(Operand::Scratch(0)), 1), + ], + count_of(Operand::Scratch(1)), + ); + let (cd, rd) = run_arm("D", &pd, &lane_planes, reps, 0); + assert_eq!(cd, want); + print(&rd); + // Dp — resident mask A gates predicate B. + let pdp = Program::new( + vec![pred_b(tb, Some(Operand::Plane(0)), 0)], + count_of(Operand::Scratch(0)), + ); + let (cdp, rdp) = run_arm("Dp", &pdp, &res_planes, reps, 0); + assert_eq!(cdp, want); + print(&rdp); + // F — quack's own lowering of `x < a AND y < b`. + let q = Query { + filter: Filter::and([ + Filter::cmp(Col(0), Cmp::LtI32(ta)), + Filter::cmp(Col(1), Cmp::LtI32(tb)), + ]), + agg: Agg::Count, + }; + let pf = lower(&q).expect("lower"); + let (cf, rf) = run_arm("F", &pf, &lane_planes, reps, 0); + assert_eq!(cf, want); + print(&rf); + let pff = lower_fused(&q).expect("lower_fused"); + let (cff, rff) = run_arm("Ff", &pff, &lane_planes, reps, 0); + assert_eq!(cff, want); + print(&rff); + // H — the scalar oracle, timed for scale only. + let t = Instant::now(); + let mut h = 0; + for _ in 0..reps { + h = oracle(std::hint::black_box(&x), &y, ta, tb); + } + let ns = t.elapsed().as_nanos() as f64 / reps as f64; + assert_eq!(h, want); + println!(" H scalar oracle (reference only) {ns:>9.0} ns"); + } + } +} + +/// E — the other two- and three-input folds over resident masks, each against +/// a per-word reference, each confirmed to take the no-slot lowering. +fn boolean_variants(n: usize) { + println!("\n== E: AndNot / Xor / Or / Ternlog(maj) over resident masks =="); + let (x, y) = lanes(n, 0xE, false); + let ma = bits_of(&x, thresh(0.3)); + let mb = bits_of(&y, thresh(0.6)); + let mc: Vec = bits_of(&x, thresh(0.7)) + .iter() + .zip(&bits_of(&y, thresh(0.2))) + .map(|(a, b)| a ^ b) + .collect(); + let planes_data: [&[u64]; 3] = [&ma, &mb, &mc]; + let planes = Planes { + n_rows: n, + masks: &planes_data, + lanes: &[], + }; + let pc = |f: fn(u64, u64, u64) -> u64| -> usize { + ma.iter() + .zip(&mb) + .zip(&mc) + .map(|((a, b), c)| f(*a, *b, *c).count_ones() as usize) + .sum() + }; + let p = |o: MaskOp| Program::new(vec![o], count_of(Operand::Scratch(0))); + let (pa, pb, pcc) = (Operand::Plane(0), Operand::Plane(1), Operand::Plane(2)); + let cases: [(&str, Program, usize); 4] = [ + ( + "a & !b", + p(MaskOp::AndNot { + a: pa, + b: pb, + dst: 0, + }), + pc(|a, b, _| a & !b), + ), + ( + "a ^ b", + p(MaskOp::Xor { + a: pa, + b: pb, + dst: 0, + }), + pc(|a, b, _| a ^ b), + ), + ( + "a | b", + p(MaskOp::Or { + a: pa, + b: pb, + dst: 0, + }), + pc(|a, b, _| a | b), + ), + ( + "maj(a,b,c)", + p(MaskOp::Ternlog { + imm: 0xE8, + a: pa, + b: pb, + c: pcc, + dst: 0, + }), + pc(|a, b, c| (a & b) | (a & c) | (b & c)), + ), + ]; + for (name, prog, want) in cases { + let (got, r) = run_arm("E", &prog, &planes, 200, 0); + assert_eq!(got, want, "{name}"); + assert_eq!(r.lowering, "Ternlog(no slot)", "{name}"); + println!(" {name:<11} = {got:>6} {} {:>7.0} ns", r.lowering, r.ns); + } + // anti-vacuity: the four answers are pairwise distinct on this fixture + let a = pc(|a, b, _| a & !b); + let b = pc(|a, b, _| a ^ b); + let c = pc(|a, b, _| a | b); + assert!( + a != b && b != c && a != c, + "the Boolean variants must differ" + ); +} + +/// `(x, y)` for row `i` of an edge-case fixture. +type RowGen = Box (i32, i32)>; + +/// Edge cases: every arm must agree with the oracle. +fn edge_cases() { + println!("\n== edge cases (every arm equals the oracle) =="); + let cases: Vec<(&str, usize, RowGen)> = vec![ + ("all zero", 4096, Box::new(|_| (SCALE, SCALE))), + ("all one", 4096, Box::new(|_| (0, 0))), + ( + "disjoint", + 4096, + Box::new(|i| if i % 2 == 0 { (0, SCALE) } else { (SCALE, 0) }), + ), + ( + "identical", + 4096, + Box::new(|i| if i % 3 == 0 { (0, 0) } else { (SCALE, SCALE) }), + ), + ( + "one live bit", + 4096, + Box::new(|i| if i == 1777 { (0, 0) } else { (0, SCALE) }), + ), + ( + "partial word n=100", + 100, + Box::new(|i| if i % 5 == 0 { (0, 0) } else { (0, SCALE) }), + ), + ( + "n = 65536+37", + 65_573, + Box::new(|i| { + if i % 7 < 3 { + (0, i as i32 % 2) + } else { + (SCALE, 0) + } + }), + ), + ]; + let (ta, tb) = (1, 1); + for (name, n, f) in cases { + let (x, y): (Vec, Vec) = (0..n).map(&f).unzip(); + let want = oracle(&x, &y, ta, tb); + let lanes_ = [LaneRef::I32(&x), LaneRef::I32(&y)]; + let ma = bits_of(&x, ta); + let mb = bits_of(&y, tb); + let resident: [&[u64]; 2] = [&ma, &mb]; + let planes = Planes { + n_rows: n, + masks: &resident, + lanes: &lanes_, + }; + let progs = [ + Program::new( + vec![MaskOp::And { + a: Operand::Plane(0), + b: Operand::Plane(1), + dst: 0, + }], + count_of(Operand::Scratch(0)), + ), + Program::new( + vec![ + pred_a(ta, None, 0), + pred_b(tb, None, 1), + MaskOp::And { + a: Operand::Scratch(0), + b: Operand::Scratch(1), + dst: 2, + }, + ], + count_of(Operand::Scratch(2)), + ), + Program::new( + vec![ + pred_a(ta, None, 0), + pred_b(tb, Some(Operand::Scratch(0)), 1), + ], + count_of(Operand::Scratch(1)), + ), + Program::new( + vec![pred_b(tb, Some(Operand::Plane(0)), 0)], + count_of(Operand::Scratch(0)), + ), + ]; + let got: Vec = progs + .iter() + .map(|p| { + let mut s = Scratch::for_program(p, n).unwrap(); + count(execute_into(p, &planes, &Foreign::NONE, &mut s, Out::None)) + }) + .collect(); + assert!(got.iter().all(|g| *g == want), "{name}: {got:?} vs {want}"); + println!(" {name:<20} n {n:>6} count {want:>6} arms B,C,D,Dp agree"); + } +} + +/// The intermediates that must NOT be eliminated, and the refusals. +fn must_not_eliminate(n: usize) { + println!("\n== reuse, demanded bitmap, refusal =="); + let (x, y) = lanes(n, 0xBEEF, false); + let (ta, tb) = (thresh(0.2), thresh(0.5)); + let lanes_ = [LaneRef::I32(&x), LaneRef::I32(&y)]; + let none: [&[u64]; 0] = []; + let planes = Planes { + n_rows: n, + masks: &none, + lanes: &lanes_, + }; + // 1. A terminal that DEMANDS the bitmap writes it, and it is the right one. + let keep = Program::new( + vec![ + pred_a(ta, None, 0), + pred_b(tb, Some(Operand::Scratch(0)), 1), + ], + keep_of(Operand::Scratch(1)), + ); + let mut s = Scratch::for_program(&keep, n).unwrap(); + let mut kept = vec![0u64; words_for(n)]; + execute_into(&keep, &planes, &Foreign::NONE, &mut s, Out::Mask(&mut kept)).unwrap(); + let want: Vec = bits_of(&x, ta) + .iter() + .zip(&bits_of(&y, tb)) + .map(|(a, b)| a & b) + .collect(); + assert_eq!(kept, want, "a demanded bitmap must be written in full"); + println!( + " Keep(A∧B) writes the demanded {} B bitmap: equal to the oracle", + kept.len() * 8 + ); + // 2. A reused intermediate: one kept A feeds two folds; their sum is |A|. + let ma = bits_of(&x, ta); + let mb = bits_of(&y, tb); + let both: [&[u64]; 2] = [&ma, &mb]; + let rp = Planes { + n_rows: n, + masks: &both, + lanes: &[], + }; + let and = Program::new( + vec![MaskOp::And { + a: Operand::Plane(0), + b: Operand::Plane(1), + dst: 0, + }], + count_of(Operand::Scratch(0)), + ); + let andnot = Program::new( + vec![MaskOp::AndNot { + a: Operand::Plane(0), + b: Operand::Plane(1), + dst: 0, + }], + count_of(Operand::Scratch(0)), + ); + let mut s0 = Scratch::new(0, 0); + let c_and = count(execute_into(&and, &rp, &Foreign::NONE, &mut s0, Out::None)); + let c_andnot = count(execute_into( + &andnot, + &rp, + &Foreign::NONE, + &mut s0, + Out::None, + )); + let pop_a: usize = ma.iter().map(|w| w.count_ones() as usize).sum(); + assert_eq!(c_and + c_andnot, pop_a); + assert!( + c_and > 0 && c_andnot > 0, + "both consumers must see survivors" + ); + println!(" A reused by two folds: |A∧B| {c_and} + |A∧¬B| {c_andnot} = |A| {pop_a}"); + // 3. A terminal with no merge law is refused on a partial extent. + let blend = Program::new( + vec![pred_a(ta, None, 0)], + Terminal::BlendI32 { + mask: Operand::Scratch(0), + then: 0, + els: 1, + }, + ); + let mut sb = Scratch::for_program(&blend, n).unwrap(); + let mut out = vec![0i32; n]; + let r = execute_extent( + &blend, + &planes, + &Foreign::NONE, + &mut sb, + Out::I32(&mut out), + 0..n / 2, + ); + assert!( + matches!(r, Err(ExecError::ExtentUnsupported { what: "BlendI32" })), + "{r:?}" + ); + println!(" BlendI32 over a partial extent: refused (no merge law)"); +} + +// ───────────────────────── CE64 over MaterializedEdges ───────────────────────── + +/// Bit layout of the two fields under test (`causal-edge` v2 layout). +const PEARL_SHIFT: u32 = 40; // bits 40..42 +const EPI5_SHIFT: u32 = 59; // bits 59..63 + +/// A 16-byte `(pattern, care)` matching `value` in the `width`-bit field at +/// `shift` of the CE64 in half `half` (0 = first eight bytes) of the window. +fn ce64_pattern(half: usize, shift: u32, width: u32, value: u64) -> ([u8; 16], [u8; 16]) { + let care = ((1u64 << width) - 1) << shift; + let pat = (value << shift) & care; + let mut p = [0u8; 16]; + let mut c = [0u8; 16]; + p[half * 8..half * 8 + 8].copy_from_slice(&pat.to_le_bytes()); + c[half * 8..half * 8 + 8].copy_from_slice(&care.to_le_bytes()); + (p, c) +} + +fn edge(rows: &[u8], r: usize, k: usize) -> CausalEdge64 { + let off = r * NODE_ROW_STRIDE + + VALUE_SLAB_ROW_OFFSET + + ValueTenant::MaterializedEdges.value_offset() + + 8 * k; + CausalEdge64(u64::from_le_bytes(rows[off..off + 8].try_into().unwrap())) +} + +fn ce64(n: usize, reps: usize) { + println!("\n== CE64 in place: COUNT(edge0.pearl == p AND edge2.epi5 == q), n = {n} =="); + let vo = VALUE_SLAB_ROW_OFFSET + ValueTenant::MaterializedEdges.value_offset(); + assert_eq!( + ValueTenant::MaterializedEdges.byte_len(), + 32, + "the tenant is 4 × u64" + ); + let mut r = Rng(0xCE64); + let mut rows = vec![0u8; n * NODE_ROW_STRIDE]; + for b in rows.iter_mut() { + *b = r.next() as u8; + } + let (p, q) = (5u64, 9u64); + // Window 0 = edges 0|1, window 1 = edges 2|3: both inside the 32-byte tenant. + let win0 = StridedRef { + bytes: &rows, + first_offset: vo, + stride: NODE_ROW_STRIDE, + records: n, + }; + let win1 = StridedRef { + bytes: &rows, + first_offset: vo + 16, + stride: NODE_ROW_STRIDE, + records: n, + }; + // No window leaves the tenant: window 1 ends exactly where the tenant does. + let tenant_end = vo + ValueTenant::MaterializedEdges.byte_len(); + assert_eq!(win1.first_offset + 16, tenant_end); + assert_eq!(win0.first_offset, vo); + let lanes_ = [LaneRef::Strided(win0), LaneRef::Strided(win1)]; + let none: [&[u64]; 0] = []; + let planes = Planes { + n_rows: n, + masks: &none, + lanes: &lanes_, + }; + let (pa, ca) = ce64_pattern(0, PEARL_SHIFT, 3, p); // edge 0, half 0 of window 0 + let (pb, cb) = ce64_pattern(0, EPI5_SHIFT, 5, q); // edge 2, half 0 of window 1 + let ma = |under, dst| MaskOp::Pred { + pred: Pred::MatchFacet16Strided { + lane: 0, + pattern: pa, + care: ca, + }, + under, + dst, + }; + let mb = |under, dst| MaskOp::Pred { + pred: Pred::MatchFacet16Strided { + lane: 1, + pattern: pb, + care: cb, + }, + under, + dst, + }; + let want = (0..n) + .filter(|&i| { + edge(&rows, i, 0).causal_mask() as u8 as u64 == p + && u64::from(edge(&rows, i, 2).epistemic_raw5()) == q + }) + .count(); + let only_a = (0..n) + .filter(|&i| edge(&rows, i, 0).causal_mask() as u8 as u64 == p) + .count(); + assert!(want > 0 && want * 3 < n, "anti-vacuity: {want} of {n}"); + println!(" accessor oracle: {want} (A alone {only_a})"); + // One strided predicate on its own: what each of C's two passes costs. + let p_one = Program::new(vec![ma(None, 0)], count_of(Operand::Scratch(0))); + let (c1, r1) = run_arm("A1", &p_one, &planes, reps, 0); + assert_eq!(c1, only_a); + print(&r1); + let pc = Program::new( + vec![ + ma(None, 0), + mb(None, 1), + MaskOp::And { + a: Operand::Scratch(0), + b: Operand::Scratch(1), + dst: 2, + }, + ], + count_of(Operand::Scratch(2)), + ); + let (cc, rc) = run_arm("C", &pc, &planes, reps, 0); + assert_eq!(cc, want, "C vs canonical accessors"); + print(&rc); + let pd = Program::new( + vec![ma(None, 0), mb(Some(Operand::Scratch(0)), 1)], + count_of(Operand::Scratch(1)), + ); + let (cd, rd) = run_arm("D", &pd, &planes, reps, 0); + assert_eq!(cd, want, "D vs canonical accessors"); + print(&rd); + // The skip a strided `_under` kernel would give: B's bytes are loaded only + // on A's survivors. Scalar, probe-local, a REFERENCE for what the missing + // primitive could save — not a backend arm and not a proposal for one. + let gated_scalar = |rows: &[u8]| -> usize { + let mut c = 0; + for i in 0..n { + let base = i * NODE_ROW_STRIDE + vo; + let e0 = u64::from_le_bytes(rows[base..base + 8].try_into().unwrap()); + if (e0 >> PEARL_SHIFT) & 0b111 == p { + let e2 = u64::from_le_bytes(rows[base + 16..base + 24].try_into().unwrap()); + if (e2 >> EPI5_SHIFT) & 0b11111 == q { + c += 1; + } + } + } + c + }; + let both_scalar = |rows: &[u8]| -> usize { + let mut c = 0; + for i in 0..n { + let base = i * NODE_ROW_STRIDE + vo; + let e0 = u64::from_le_bytes(rows[base..base + 8].try_into().unwrap()); + let e2 = u64::from_le_bytes(rows[base + 16..base + 24].try_into().unwrap()); + c += usize::from((e0 >> PEARL_SHIFT) & 0b111 == p && (e2 >> EPI5_SHIFT) & 0b11111 == q); + } + c + }; + for (name, f) in [ + ("H-gated", &gated_scalar as &dyn Fn(&[u8]) -> usize), + ("H-both", &both_scalar), + ] { + let t = Instant::now(); + let mut c = 0; + for _ in 0..reps { + c = f(std::hint::black_box(&rows)); + } + let ns = t.elapsed().as_nanos() as f64 / reps as f64; + assert_eq!(c, want); + println!( + " {name:<8} scalar reference (A's row read first; B read {}) {ns:>12.0} ns", + if name == "H-gated" { + "only on A's survivors" + } else { + "on every row" + } + ); + } + let line = |b: usize| b / 64; + println!( + " bytes: resident {} B; tenant at row bytes {vo}..{}; named per row 32 B (two 16 B windows); \ + 64 B lines per row: window 0 {:?}, window 1 {:?}", + rows.len(), + vo + 32, + line(vo)..=line(vo + 15), + line(vo + 16)..=line(vo + 31) + ); + + // Pattern merge: two field predicates inside ONE 16-byte window are one + // ternary match — `match(p1, c1) ∧ match(p2, c2) = match(p1 | p2, c1 | c2)` + // when the patterns agree on `c1 & c2` (here the cares are disjoint). + let (pe1, ce1) = ce64_pattern(1, EPI5_SHIFT, 5, q); // edge 1, half 1 of window 0 + let merged_p: [u8; 16] = core::array::from_fn(|k| pa[k] | pe1[k]); + let merged_c: [u8; 16] = core::array::from_fn(|k| ca[k] | ce1[k]); + assert!((0..16).all(|k| ca[k] & ce1[k] == 0), "disjoint cares"); + let want_w = (0..n) + .filter(|&i| { + edge(&rows, i, 0).causal_mask() as u8 as u64 == p + && u64::from(edge(&rows, i, 1).epistemic_raw5()) == q + }) + .count(); + assert!( + want_w > 0 && want_w * 3 < n, + "anti-vacuity: {want_w} of {n}" + ); + let two = Program::new( + vec![ + ma(None, 0), + MaskOp::Pred { + pred: Pred::MatchFacet16Strided { + lane: 0, + pattern: pe1, + care: ce1, + }, + under: None, + dst: 1, + }, + MaskOp::And { + a: Operand::Scratch(0), + b: Operand::Scratch(1), + dst: 2, + }, + ], + count_of(Operand::Scratch(2)), + ); + let one = Program::new( + vec![MaskOp::Pred { + pred: Pred::MatchFacet16Strided { + lane: 0, + pattern: merged_p, + care: merged_c, + }, + under: None, + dst: 0, + }], + count_of(Operand::Scratch(0)), + ); + let (c2, r2) = run_arm("C2", &two, &planes, reps, 0); + let (c1m, r1m) = run_arm("M1", &one, &planes, reps, 0); + assert_eq!(c2, want_w); + assert_eq!(c1m, want_w); + println!(" same window, edge0.pearl AND edge1.epi5 (oracle {want_w}):"); + print(&r2); + print(&r1m); + + // The precondition is load-bearing: two patterns that DISAGREE on a shared + // care bit have an empty conjunction, and an unconditional `p1 | p2` merge + // would answer something else. + let (px, cx) = ce64_pattern(0, PEARL_SHIFT, 3, p ^ 1); // same field, other value + let conflict = Program::new( + vec![ + ma(None, 0), + MaskOp::Pred { + pred: Pred::MatchFacet16Strided { + lane: 0, + pattern: px, + care: cx, + }, + under: None, + dst: 1, + }, + MaskOp::And { + a: Operand::Scratch(0), + b: Operand::Scratch(1), + dst: 2, + }, + ], + count_of(Operand::Scratch(2)), + ); + let naive_p: [u8; 16] = core::array::from_fn(|k| pa[k] | px[k]); + let naive_c: [u8; 16] = core::array::from_fn(|k| ca[k] | cx[k]); + let naive = Program::new( + vec![MaskOp::Pred { + pred: Pred::MatchFacet16Strided { + lane: 0, + pattern: naive_p, + care: naive_c, + }, + under: None, + dst: 0, + }], + count_of(Operand::Scratch(0)), + ); + let (cc0, _) = run_arm("Cx", &conflict, &planes, 1, 0); + let (cn, _) = run_arm("Mx", &naive, &planes, 1, 0); + assert_eq!(cc0, 0, "conflicting patterns have no common row"); + assert_ne!( + cn, 0, + "an unconditional OR-merge answers a different question" + ); + println!(" conflicting cares: two predicates {cc0}, unconditional OR-merge {cn} (refuse or constant-false, never merge)"); + + // can fire: the wrong bit range must disagree + let (pw, cw) = ce64_pattern(0, PEARL_SHIFT + 1, 3, p); + let pwrong = Program::new( + vec![ + MaskOp::Pred { + pred: Pred::MatchFacet16Strided { + lane: 0, + pattern: pw, + care: cw, + }, + under: None, + dst: 0, + }, + mb(Some(Operand::Scratch(0)), 1), + ], + count_of(Operand::Scratch(1)), + ); + let (cw_, _) = run_arm("Cw", &pwrong, &planes, 1, 0); + assert_ne!(cw_, want, "a care on the wrong bits must disagree"); + // stay silent: vary everything OUTSIDE care — edge1, edge3, the other + // fields of edges 0 and 2, and every byte outside the tenant. + let mut rows2 = rows.clone(); + for i in 0..n { + let base = i * NODE_ROW_STRIDE; + for (j, b) in rows2[base..base + NODE_ROW_STRIDE].iter_mut().enumerate() { + let in_tenant = j >= vo && j < vo + 32; + let k = (j.wrapping_sub(vo)) / 8; + let bit_safe = if in_tenant && (k == 0 || k == 2) { + let byte = (j - vo) % 8; + let field_mask: u64 = if k == 0 { + 0b111 << PEARL_SHIFT + } else { + 0b11111 << EPI5_SHIFT + }; + ((field_mask >> (8 * byte)) & 0xFF) as u8 + } else { + 0 + }; + *b = (*b & bit_safe) | (!bit_safe & (r.next() as u8)); + } + } + let win0b = StridedRef { + bytes: &rows2, + ..win0 + }; + let win1b = StridedRef { + bytes: &rows2, + ..win1 + }; + let lanes2 = [LaneRef::Strided(win0b), LaneRef::Strided(win1b)]; + let planes2 = Planes { + n_rows: n, + masks: &none, + lanes: &lanes2, + }; + let (cd2, _) = run_arm("D", &pd, &planes2, 1, 0); + assert_eq!(cd2, want, "bits outside care must not change the answer"); + println!(" can-fire (care shifted by one bit): {cw_} ≠ {want}; silent (all non-care bits rewritten): {cd2} = {want}"); +} + +fn main() { + let n = 65_536; + edge_cases(); + must_not_eliminate(n); + boolean_variants(n); + ce64(n, 200); + sweep(n, 400); + println!("\nall counts agree with the oracle"); +} From f145f0b92f19d82dd36da3f198b0c458b8f03019 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 8 Oct 2026 16:27:36 +0000 Subject: [PATCH 2/4] plan(D-RPF): three deforestation stages; D-RPF-9 measured Board entry for the fold_join_probe results and the capability inventory; the resident-projection plan gains the three stages (projection, mask, fold-join), the D-RPF-0 equality status, and D-RPF-9 with its proposed rewrite contract (R2 strided pattern merge, R3 gate selection) and the three named T1 gaps. Nothing is implemented beyond the probe. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01EFw2WdKr1oxvaKCJC2ua2R --- .claude/board/STATUS_BOARD.md | 3 +- ...026-10-08-fold-join-deforestation-probe.md | 174 ++++++++++++++++++ .claude/board/entries/README.md | 3 +- ...-10-08-resident-projection-fold-mask-v1.md | 48 +++++ 4 files changed, 226 insertions(+), 2 deletions(-) create mode 100644 .claude/board/entries/2026-10-08-fold-join-deforestation-probe.md diff --git a/.claude/board/STATUS_BOARD.md b/.claude/board/STATUS_BOARD.md index 0887a931b..74d79b1db 100644 --- a/.claude/board/STATUS_BOARD.md +++ b/.claude/board/STATUS_BOARD.md @@ -4,7 +4,7 @@ Plan: `.claude/plans/2026-10-08-resident-projection-fold-mask-v1.md`. Mask = adm | D-id | scope | status | gate / falsifier | |---|---|---|---| -| **D-RPF-0** | Pearl3 / Epi5 / mantissa predicate over `MaterializedEdges` via `MatchFacet16Strided`, threshold as ≤ n+1 patterns | Queued | equals CE64 accessors; wrong-bit care disagrees; unnamed fields silent; both windows stay inside the 32 B tenant | +| **D-RPF-0** | Pearl3 / Epi5 / mantissa predicate over `MaterializedEdges` via `MatchFacet16Strided`, threshold as ≤ n+1 patterns | In progress (equality green via D-RPF-9; threshold union open) | equals CE64 accessors; wrong-bit care disagrees; unnamed fields silent; both windows stay inside the 32 B tenant | | **D-RPF-1** | Law tables → per-recipe ternary pattern set (≤ 24 × 8), run under an admission plane (class, rail, generation, provenance) | Queued | equals `measure_declared`; dropping one code loses exactly its rows; removing the admission plane admits a v1 row | | **D-RPF-2** | Census of eligibility and read-set classes on a real population | Queued | classes/rows reported; near 1 drops D-RPF-3; no population-sized seen-set | | **D-RPF-3** | Exact read-set dedup per CE64 instruction, key incl. handle context | Queued (after D-RPF-2) | removing cohort from key changes a result; dedup-then-fold equals per-row for Count/Sum/Avg | @@ -13,6 +13,7 @@ Plan: `.claude/plans/2026-10-08-resident-projection-fold-mask-v1.md`. Mask = adm | **D-RPF-6** | Tile skip on zero gate measured on resident-row predicates (capstone pt 3) | Queued | identical answers; dead-tile beside dead-word fraction | | **D-RPF-7** | Rank spectrum of the 256×256 palette distance LUTs | Queued (measurement only) | reported; never a fold | | **D-RPF-8** | Boundary: mask-risc never executes CE64 instructions; no VSA on this path | Queued (doc) | review-enforced | +| **D-RPF-9** | Fold-Join deforestation probe: resident ∧ resident, predicate chains gated/ungated, CE64 in place, strided pattern merge | Shipped (probe; R2/R3/G1–G3 proposed) | every arm equals the oracle; care-shift fires, non-care rewrite silent; merge precondition refuses conflicting cares | ## D-CTX — Palette texture proof ladder: fieldless first, rendering only if needed (2026-10-06) diff --git a/.claude/board/entries/2026-10-08-fold-join-deforestation-probe.md b/.claude/board/entries/2026-10-08-fold-join-deforestation-probe.md new file mode 100644 index 000000000..0eece764c --- /dev/null +++ b/.claude/board/entries/2026-10-08-fold-join-deforestation-probe.md @@ -0,0 +1,174 @@ +# 2026-10-08 — Fold-Join deforestation probe (D-RPF-9) + +**Status:** MEASURED. Probe only, no production primitive. Branch +`ccr-b2e415d9-4jfvyk-fold-join`, commit `1c43faa5`: +`crates/lance-graph-quack/examples/fold_join_probe.rs`. + +```text +CARGO_PROFILE_RELEASE_DEBUG=0 cargo run --release -p lance-graph-quack --example fold_join_probe +``` + +The question: when two address-aligned membership sources meet in one +terminal, which shipped lowering already avoids the intermediate, and what does +each arm actually write? Every arm runs the shipped mask-risc executor or +quack's lowering into it. Each one is checked against a per-row oracle on +65,536 rows. + +## Capability inventory (VERIFIED-IN-CODE) + +| shape | where it lives | intermediate | +|---|---|---| +| Boolean chain over ≤ 6 resident planes → `Count`/`Any` | `Program::fused_ternlog` / `fused_tern2` / `fused_tern3` (`mask-risc/src/ir.rs`) → `mask_ternlog_popcount` | none: no slot, no bitmap | +| the same chain → `Keep` | `fused_keep`, `Tern2/Tern3` with `Out::Mask` | the demanded bitmap only | +| any `Pred` or `Gather` in the chain | `Lowering::Tiled` (`exec.rs`, `TILE_WORDS = 256`) | one 2 KB tile per slot; no population mask (tiled law, 2026-09-21) | +| `Pred B under A` | `MaskOp::Pred { under }`; quack `lower` gates every conjunct on the accumulator (`emit_gated`) | as tiled | +| strided predicates under a gate | `exec.rs` `run_pred`: no `_under` twin, so the full kernel runs and is then ANDed | as tiled, and B is evaluated on rejected rows | +| `lower_fused` | one slot per comparison, ternlog skeleton, never gated | as tiled | + +What the Cognitive Shader Driver has is a different operation: +- `Quad8::fold_product` folds the Cartesian product of ONE object's occupancy + coordinates without building it. That is a precedent for not materialising a + product, but it is not an aligned intersection of two populations. +- `MergeMode::{Xor, Bundle, Superposition}` are commit semantics for deltas. +- The rotated-XOR fingerprint in `driver.rs` is a lossy aggregate. +- `thinking-engine` superposition multiplies f32 amplitudes, which A2 forbids + for masks. + +No CSD arm was run (arm G): none of these computes `popcount(A ∧ B)`. + +## Results + +### Correctness + +Every arm equals the oracle: +- **Edge cases:** all-zero, all-one, disjoint, identical, one live bit, n = 100, + n = 65,573. +- **A bitmap the terminal asks for** (`Keep(A ∧ B)`) is written in full and + equals the oracle. +- **One intermediate feeding two folds:** `|A∧B| + |A∧¬B| = |A|`. +- **A terminal without a merge law** (`BlendI32` over a partial extent) is + refused. + +### Timings + +Timings are ns per execution on this host, 64K rows. They are noisy run to run +(about ±10 %); two runs were taken and agree on every ordering below. + +| arm | what | scattered A sel 0.001 / 0.01 / 0.1 / 0.5 / 0.9 | +|---|---|---| +| B | resident A ∧ B → Count (no slot) | 445 / 387 / 382 / 384 / 385 | +| A | Keep A, Keep B (16 KB population masks), then fold | ~18.5 k at every selectivity | +| C | Pred A, Pred B, And, Count | 18.7 k / 17.9 k / 18.2 k / 18.0 k / 18.0 k | +| D | Pred A, Pred B under A | 13.5 k / 19.6 k / 25.2 k / 25.1 k / 25.9 k | +| Dp | resident mask A gates Pred B | 3.9 k / 10.0 k / 15.8 k / 15.7 k / 16.2 k | +| H | per-row scalar oracle (reference only) | 14.6 k – 18.3 k | + +- **Clustered A** (survivors in one run; dead-tile fraction 0.75 at sel ≤ 0.1): + D 12.6 k / 12.3 k / 13.7 k / 19.8 k / 25.0 k against C 23.1 k / 18.2 k / + 17.6 k / 19.3 k / 18.8 k. +- **Quack:** `lower` tracks D and `lower_fused` tracks C at every selectivity. +- **E (`AndNot` / `Xor` / `Or` / majority over resident masks):** all take the + no-slot lowering, 362–1213 ns. + +### CE64 read in place from `MaterializedEdges` + +`MatchFacet16Strided` on 16-byte windows equals the `CausalEdge64` accessors. +- **Can fire:** a care shifted by one bit disagrees (227 vs 282). +- **Can stay silent:** rewriting every bit outside care leaves the answer + unchanged. +- **Anti-vacuity:** 282 of 65,536 rows. +- **Disable run:** placing edge 2's pattern in the wrong half made C answer 263 + against the accessors' 282. + +The tenant sits at row bytes 48..80. Window 0 (edges 0 | 1) is in 64-byte +line 0 of the row and window 1 (edges 2 | 3) in line 1. Both windows stay +inside the tenant, so there is no RBAC widening. + +| arm | ns (64K rows, 33.5 MB resident) | +|---|---| +| one strided predicate | 1.03 M | +| C: edge0.pearl ∧ edge2.epi5, two passes | 2.11 M | +| D: the same, B under A (no `_under` kernel) | 2.24 M | +| scalar reference, both fields in one row visit | 0.97 M | +| scalar reference, B read only on A's survivors | 0.79 M | +| two fields in the SAME window, two predicates | 2.47 M | +| the same two fields as ONE merged pattern | 1.17 M | + +## Findings + +1. **Fold-Join over aligned resident masks is already deforested.** `popcount` + of any Boolean function of up to six resident planes runs with no slot and + no bitmap, at about 0.4 µs per 64K rows. No new primitive is needed for the + resident-mask ∧ resident-mask case. +2. **Mask deforestation at population scale is already the law.** The tiled + executor writes no population-sized mask; its intermediates are one 2 KB tile + per slot, and there were 0 allocations per execution in every arm. For i32 + predicates, the two-predicate chain (C) runs at about the per-row scalar + loop's speed. Reading the lanes costs more than writing the tile masks, so + fusing the predicates into the terminal has little left to save there. +3. **Gate or not, measured.** Gating B under A (D, and quack `lower`) beats + the ungated chain only when A's dead-word fraction is high: + - ~1.4× faster at dead-word ≥ 0.9; + - about even near 0.5; + - ~1.4× slower with no dead words. + + Quack picks one shape per entry point (`lower` always gates, `lower_fused` + never does). This is the capstone's fuse-or-gate switch, and the measured + crossover on this host is a dead-word fraction around 0.5. It is a pin for + this host, not physics. +4. **Strided predicates pay one pass per predicate.** With a 512-byte stride, + each predicate touches 64K separate cache lines, so cost scales with passes: + - two predicates take 2.1× one; + - **merging two field predicates in the same 16-byte window into one ternary + pattern** gives the same answer at 1.17 ms instead of 2.47 ms. This is a + lowering rewrite with no new primitive. + + The precondition is load-bearing: two patterns that disagree on a shared + care bit have an empty conjunction (0 rows), while an unconditional OR-merge + answered 8,273. +5. **Gating a strided predicate does not skip anything.** Without an `_under` + kernel, D costs at least as much as C. The scalar reference that reads B only + on A's survivors (12.6 % here) is about 20 % faster than reading both fields + on every row. That is an upper bound on the potential, not a measurement of a + kernel. + +## Proposed rewrite contract (not implemented) + +- **R1 (shipped, unchanged):** a Boolean chain over ≤ 6 resident planes feeding + `Count`/`Any`/`Keep` folds directly. +- **R2 strided pattern merge (lowering only):** + - **When:** `MatchFacet16Strided(L, p1, c1) ∧ MatchFacet16Strided(L, p2, c2)`, + on the SAME lane and the same declared reading, where neither intermediate + has another consumer. + - **Rewrite:** to `MatchFacet16Strided(L, p1 | p2, c1 | c2)` when + `(p1 ^ p2) & c1 & c2 == 0`, and to the empty mask otherwise. Never an + unconditional OR. + - **Exact for every terminal**, because it is a mask identity. +- **R3 gate selection:** gate a conjunct under the accumulator only when the + gate's dead-word fraction is known and high; otherwise run it ungated. This + needs the live count threaded with the aperture (seventeen rooms, room 6). + The crossover is a measured pin, re-measured per host class. +- **Gaps named, not built (T1, `ndarray::simd`):** + - G1 `ternary_match_strided16_to_mask_under`; + - G2 a two-window / 32-byte strided match (one row visit for both CE64 pairs); + - G3 tile skip in the executor (capstone point 3, now printed as a + dead-tile fraction). + +## Boundaries kept + +- No CE64 instruction or NARS revision is folded; the strided predicates read + raw bits. +- Pattern merge applies only within one lane under one declared reading. Two + signed-register lanes bound under different `RegisterLaw`s never meet in a + fold: `bind_signed_register` refuses `RegisterLawMismatch` (#1410) before any + mask exists. +- A Fold-Join is an aligned-address intersection. A hop that resolves a + different address is `Gather` / `ScatterOrU32`, not a Fold-Join. + +## OPEN + +- R2, R3 and G1–G3 are proposals; none is implemented. +- The D-RPF-0 threshold-as-union-of-patterns claim is not exercised by this + probe. +- Timings come from one host class. The crossover in finding 3 must be + re-measured before it becomes a planner constant. diff --git a/.claude/board/entries/README.md b/.claude/board/entries/README.md index d1a137647..fef0459c6 100644 --- a/.claude/board/entries/README.md +++ b/.claude/board/entries/README.md @@ -25,7 +25,7 @@ index row, (3) no duplicate entry id. Checks 1 and 2 are deliberately opposite directions; the stranding this convention prevents shows up in exactly one of them, never both. -271 entries, 2026-08-06 .. 2026-10-08. +272 entries, 2026-08-06 .. 2026-10-08. | date | entry id | finding | file | |---|---|---|---| @@ -36,6 +36,7 @@ exactly one of them, never both. | 2026-10-08 | `moore-nars16-isa-visible-representation` | | [2026-10-08-moore-nars16-isa-visible-representation.md](2026-10-08-moore-nars16-isa-visible-representation.md) | | 2026-10-08 | `D-MOORE-NARS-0` | | [2026-10-08-moore-nars-0-recipe-learning-gomoku.md](2026-10-08-moore-nars-0-recipe-learning-gomoku.md) | | 2026-10-08 | `hhtl-nars-moore-value-tenants` | | [2026-10-08-hhtl-nars-moore-value-tenants.md](2026-10-08-hhtl-nars-moore-value-tenants.md) | +| 2026-10-08 | `D-RPF-9` | | [2026-10-08-fold-join-deforestation-probe.md](2026-10-08-fold-join-deforestation-probe.md) | | 2026-10-08 | `coresearch-ce64-moore-masking-wiring` | | [2026-10-08-coresearch-ce64-moore-masking-wiring.md](2026-10-08-coresearch-ce64-moore-masking-wiring.md) | | 2026-10-08 | `ce64-isa-register-contract` | | [2026-10-08-ce64-isa-register-contract.md](2026-10-08-ce64-isa-register-contract.md) | | 2026-10-07 | `tinker-janus-fold-harvest` | TinkerPop bulk = K (GroupReduce), ONE_BULK = S, GValue pinning = bundle invalidation; JanusGraph slice = OrderedLaneWitness→Range (30–38×); no new V4 op | [2026-10-07-tinker-janus-fold-harvest.md](2026-10-07-tinker-janus-fold-harvest.md) | diff --git a/.claude/plans/2026-10-08-resident-projection-fold-mask-v1.md b/.claude/plans/2026-10-08-resident-projection-fold-mask-v1.md index 422c979b3..d29ac7493 100644 --- a/.claude/plans/2026-10-08-resident-projection-fold-mask-v1.md +++ b/.claude/plans/2026-10-08-resident-projection-fold-mask-v1.md @@ -60,6 +60,26 @@ between them. Where a hop has no primitive, the phase stops and files the gap. - [ ] **D-RPF-6** — Tile skip on a zero gate (capstone point 3) measured on resident rows - [ ] **D-RPF-7** — Rank spectrum of the 256×256 palette distance LUTs (measurement only) - [ ] **D-RPF-8** — Boundary declaration: what mask-risc never executes +- [x] **D-RPF-9** — Fold-Join deforestation probe (measured 2026-10-08; rewrite contract proposed, not implemented) + +## Three stages of deforestation (added 2026-10-08, D-RPF-9) + +Each stage removes one kind of intermediate. They are distinct and must not be +conflated: + +| stage | what is never built | state on `main` | phases | +|---|---|---|---| +| **1. Projection** | an extracted CE64 lane | predicates read `MaterializedEdges` in place through `MatchFacet16Strided`; equal to the accessors (D-RPF-9) | D-RPF-0, D-RPF-1, D-RPF-5 | +| **2. Mask** | a population-sized bitmap between a predicate and a terminal that consumes it | already the tiled law: one 2 KB tile per slot, 0 allocations; remaining work is R2 (strided pattern merge), R3 (gate selection), G1 (strided `_under`), G3 (tile skip) | D-RPF-6, D-RPF-9 | +| **3. Fold-Join** | a pair relation, cross product, join result or intermediate aggregate when aligned resident addresses and the fold's law suffice | shipped for ≤ 6 resident planes (`fused_ternlog` / `tern2` / `tern3`, no slot); G2 (two-window strided match) is the open gap for CE64 pairs | D-RPF-9 | + +Stage 3 is an intersection at the SAME address. A hop that resolves a +different address (`Gather`, `ScatterOrU32`, multi-hop graph joins) is not a +Fold-Join and keeps its own lowering. Two signed-register lanes are only +combinable when both bind under the same `RegisterLaw`; a mismatch is refused +by `bind_signed_register` (`RegisterLawMismatch`, #1410) before any mask +exists, and `RelativeOffset`, `AxisPosition` and `Support` are never +interchangeable. ## Details @@ -91,6 +111,13 @@ so no RBAC question arises and no 8-byte strided primitive is needed. (An earlier draft placed one window per edge and flagged an overlap for `k = 3`; that placement was unnecessary — Codex review on #1411.) +**Status (2026-10-08, D-RPF-9).** The equality half is green: the probe +matches edge 0's Pearl3 and edge 2's Epi5 in place, equal to the +`CausalEdge64` accessors; a care shifted by one bit disagrees; rewriting every +non-care bit leaves the answer unchanged; the windows stay inside the tenant +(row bytes 48..80, one 64 B line each). The threshold-as-union-of-patterns half +is not yet exercised. + ### D-RPF-1 — Recipe eligibility as a mask **Grounding (VERIFIED-IN-CODE).** `affordance_measurement_probe.rs`: eligibility @@ -271,6 +298,27 @@ To be written down, then enforced by review: - No new opcode, field, or tenant is introduced by this plan; D-RPF-4 and D-RPF-0 may each file one T1 primitive, backend-first. +### D-RPF-9 — Fold-Join deforestation probe + +**Measured.** `crates/lance-graph-quack/examples/fold_join_probe.rs`; numbers +and the full inventory in +`entries/2026-10-08-fold-join-deforestation-probe.md`. Summary: + +- Resident ∧ resident → fold is already deforested (no slot, ~0.4 µs / 64K). +- Predicate → fold writes no population mask (tiled); for i32 lanes the two- + predicate chain runs at roughly the per-row scalar loop's speed. +- Gating pays only at a high dead-word fraction (crossover ≈ 0.5 on this host); + quack `lower` always gates and `lower_fused` never does. +- Strided predicates cost one pass over the rows each; merging two field + predicates in one 16 B window into one ternary pattern halved the time + (2.47 → 1.17 ms) with the same answer. + +**Proposed rewrite contract (not implemented; each its own focused PR):** +R2 strided pattern merge (`p1 | p2`, `c1 | c2` only when +`(p1 ^ p2) & c1 & c2 == 0`, else empty — never an unconditional OR); +R3 gate selection from a known dead-word fraction; gaps G1 strided `_under`, +G2 two-window strided match, G3 tile skip — backend-first in `ndarray::simd`. + ## Order D-RPF-0 → D-RPF-1 → D-RPF-2 (decides whether D-RPF-3 is worth doing) → From 7faefd78067c477418013f1ffbe434f858814c6c Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 8 Oct 2026 16:16:23 +0000 Subject: [PATCH 3/4] ci: optimize cognitive-shader-driver in member-tests (debug stays 0) member-tests hit its 30-minute limit on main (run 37766947128) and on #1410, both cut off in the hydrate step. The time went to one binary: the crossword_real_words_probe example tests (D-PUZZLE-0, added 2026-10-07) ran 810 s at opt-level 0. The shader-driver test step now passes --config 'profile.dev.package.cognitive-shader-driver.opt-level=3'. Only that package is optimized; debug info stays 0 from the manifest and every other crate keeps the shared opt-level-0 cache. Measured locally: 184 s for that binary, 4m17 for the whole step including the compile, all green. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01EFw2WdKr1oxvaKCJC2ua2R --- .github/workflows/rust-test.yml | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/.github/workflows/rust-test.yml b/.github/workflows/rust-test.yml index f7a375926..0e742356e 100644 --- a/.github/workflows/rust-test.yml +++ b/.github/workflows/rust-test.yml @@ -436,8 +436,16 @@ jobs: run: cargo test --manifest-path crates/sigma-tier-router/Cargo.toml - name: Run neural-debug tests (previously ungated) # 11 green run: cargo test --manifest-path crates/neural-debug/Cargo.toml + # opt-level 3 for THIS package only, debug stays 0 (manifest). The + # crossword_real_words_probe example tests (D-PUZZLE-0, 2026-10-07) run a + # solver search that takes 810 s at opt-level 0 and pushed this job past + # its 30-minute limit on main (run 37766947128, cut off in `hydrate`). + # Measured locally at opt-level 3: 183 s, 4m08 including the compile. + # `--config` keeps every other crate on the shared opt-level-0 cache. - name: Run shader-driver tests (previously ungated) # 107 + 2 green - run: cargo test --manifest-path crates/cognitive-shader-driver/Cargo.toml + run: >- + cargo test --manifest-path crates/cognitive-shader-driver/Cargo.toml + --config 'profile.dev.package.cognitive-shader-driver.opt-level=3' # The default run above skips these targets: `w2_differential` and # `mailbox_cutover` are `cfg(feature = "mailbox-thoughtspace")`, and the # StreamDto circuit probe needs `with-engine` for `StreamDto`. From 91d9784b11b4b1ebcc0e2cb9f7d5d6189646bd37 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 8 Oct 2026 17:44:51 +0000 Subject: [PATCH 4/4] =?UTF-8?q?fold=5Fjoin=5Fprobe:=20physical=20scratch;?= =?UTF-8?q?=20mask=20patterns=20before=20merging=20(Codex=20P2=20=C3=972)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The scratch column now reports what Scratch::for_program actually carved (slots × words), so the no-slot lowering prints 0 B instead of the program's logical slot count. The R2 pattern-merge law is now (p1 & c1) | (p2 & c2) under the conflict check: a pattern may carry bits outside its own care, which the matcher ignores. New falsifier: with junk bits outside edge 1's care, the two predicates and the masked merge both answer 258, an unmasked OR 275. Board entry and plan updated to the masked form. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01EFw2WdKr1oxvaKCJC2ua2R --- ...026-10-08-fold-join-deforestation-probe.md | 14 ++- ...-10-08-resident-projection-fold-mask-v1.md | 4 +- .../examples/fold_join_probe.rs | 93 ++++++++++++++++--- 3 files changed, 90 insertions(+), 21 deletions(-) diff --git a/.claude/board/entries/2026-10-08-fold-join-deforestation-probe.md b/.claude/board/entries/2026-10-08-fold-join-deforestation-probe.md index 2e78c7399..77a618ac5 100644 --- a/.claude/board/entries/2026-10-08-fold-join-deforestation-probe.md +++ b/.claude/board/entries/2026-10-08-fold-join-deforestation-probe.md @@ -124,8 +124,11 @@ inside the tenant, so there is no RBAC widening. lowering rewrite with no new primitive. The precondition is load-bearing: two patterns that disagree on a shared - care bit have an empty conjunction (0 rows), while an unconditional OR-merge - answered 8,273. + care bit have an empty conjunction (0 rows), while a union that skipped the + conflict check answered 8,273. So is the masking: a pattern may carry bits + outside its own care, which the matcher ignores; an OR of unmasked patterns + reads them as part of the other field's pattern and answered 275 against the + conjunction's 258 (Codex P2 on #1414). 5. **Gating a strided predicate does not skip anything.** Without an `_under` kernel, D costs at least as much as C. The scalar reference that reads B only on A's survivors (12.6 % here) is about 20 % faster than reading both fields @@ -140,9 +143,10 @@ inside the tenant, so there is no RBAC widening. - **When:** `MatchFacet16Strided(L, p1, c1) ∧ MatchFacet16Strided(L, p2, c2)`, on the SAME lane and the same declared reading, where neither intermediate has another consumer. - - **Rewrite:** to `MatchFacet16Strided(L, p1 | p2, c1 | c2)` when - `(p1 ^ p2) & c1 & c2 == 0`, and to the empty mask otherwise. Never an - unconditional OR. + - **Rewrite:** to + `MatchFacet16Strided(L, (p1 & c1) | (p2 & c2), c1 | c2)` when + `(p1 ^ p2) & c1 & c2 == 0`, and to the empty mask otherwise. Never an OR + of unmasked patterns, and never a union without the conflict check. - **Exact for every terminal**, because it is a mask identity. - **R3 gate selection:** gate a conjunct under the accumulator only when the gate's dead-word fraction is known and high; otherwise run it ungated. This diff --git a/.claude/plans/2026-10-08-resident-projection-fold-mask-v1.md b/.claude/plans/2026-10-08-resident-projection-fold-mask-v1.md index 5a4f80234..e29542109 100644 --- a/.claude/plans/2026-10-08-resident-projection-fold-mask-v1.md +++ b/.claude/plans/2026-10-08-resident-projection-fold-mask-v1.md @@ -313,8 +313,8 @@ and the full inventory in (2.47 → 1.17 ms) with the same answer. **Proposed rewrite contract (not implemented; each its own focused PR):** -R2 strided pattern merge (`p1 | p2`, `c1 | c2` only when -`(p1 ^ p2) & c1 & c2 == 0`, else empty — never an unconditional OR); +R2 strided pattern merge (`(p1 & c1) | (p2 & c2)`, `c1 | c2` only when +`(p1 ^ p2) & c1 & c2 == 0`, else empty — never an OR of unmasked patterns); R3 gate selection from a known dead-word fraction; gaps G1 strided `_under`, G2 two-window strided match, G3 tile skip — backend-first in `ndarray::simd`. diff --git a/crates/lance-graph-quack/examples/fold_join_probe.rs b/crates/lance-graph-quack/examples/fold_join_probe.rs index 7a37864b8..55395a915 100644 --- a/crates/lance-graph-quack/examples/fold_join_probe.rs +++ b/crates/lance-graph-quack/examples/fold_join_probe.rs @@ -39,8 +39,8 @@ use causal_edge::CausalEdge64; use lance_graph_contract::canonical_node::{ValueTenant, NODE_ROW_STRIDE, VALUE_SLAB_ROW_OFFSET}; use lance_graph_mask_risc::exec::{execute_extent, execute_into, Scratch}; use lance_graph_mask_risc::{ - tile_words_for, words_for, ExecError, Foreign, LaneRef, Lowering, MaskOp, Operand, Out, Planes, - Pred, Program, StridedRef, Terminal, Value, TILE_WORDS, + words_for, ExecError, Foreign, LaneRef, Lowering, MaskOp, Operand, Out, Planes, Pred, Program, + StridedRef, Terminal, Value, TILE_WORDS, }; use lance_graph_quack::{lower, lower_fused, Agg, Cmp, Col, Filter, Query}; @@ -179,7 +179,9 @@ fn run_arm( population_bytes: usize, ) -> (usize, Report) { let mut scratch = Scratch::for_program(p, planes.n_rows).expect("scratch"); - let scratch_bytes = p.scratch_slots as usize * tile_words_for(planes.n_rows) * 8; + // What `Scratch::for_program` actually carved, not the program's logical + // slot count: a fused lowering declares slots and allocates none. + let scratch_bytes = scratch.slots() * scratch.words() * 8; let first = count(execute_into( p, planes, @@ -375,7 +377,10 @@ fn sweep(n: usize, reps: usize) { lowering: "Keep,Keep,Ternlog".into(), predicates: 2, mask_passes: 1, - scratch_bytes: 2 * tile_words_for(n) * 8, + scratch_bytes: (sa.slots() * sa.words() + + sb.slots() * sb.words() + + sf.slots() * sf.words()) + * 8, population_bytes: 2 * words * 8, allocs_per_exec: (a1 - a0) as f64 / reps as f64, ns, @@ -920,11 +925,18 @@ fn ce64(n: usize, reps: usize) { ); // Pattern merge: two field predicates inside ONE 16-byte window are one - // ternary match — `match(p1, c1) ∧ match(p2, c2) = match(p1 | p2, c1 | c2)` - // when the patterns agree on `c1 & c2` (here the cares are disjoint). + // ternary match — + // `match(p1, c1) ∧ match(p2, c2) = match((p1 & c1) | (p2 & c2), c1 | c2)` + // when `(p1 ^ p2) & c1 & c2 == 0`. Pattern bits outside their own care are + // don't-care, so each is masked by its care before the union (Codex P2 on + // #1414); here the cares are disjoint. let (pe1, ce1) = ce64_pattern(1, EPI5_SHIFT, 5, q); // edge 1, half 1 of window 0 - let merged_p: [u8; 16] = core::array::from_fn(|k| pa[k] | pe1[k]); - let merged_c: [u8; 16] = core::array::from_fn(|k| ca[k] | ce1[k]); + let merge = |p1: &[u8; 16], c1: &[u8; 16], p2: &[u8; 16], c2: &[u8; 16]| { + let p: [u8; 16] = core::array::from_fn(|k| (p1[k] & c1[k]) | (p2[k] & c2[k])); + let c: [u8; 16] = core::array::from_fn(|k| c1[k] | c2[k]); + (p, c) + }; + let (merged_p, merged_c) = merge(&pa, &ca, &pe1, &ce1); assert!((0..16).all(|k| ca[k] & ce1[k] == 0), "disjoint cares"); let want_w = (0..n) .filter(|&i| { @@ -976,9 +988,62 @@ fn ce64(n: usize, reps: usize) { print(&r2); print(&r1m); + // Unmasked operands: give edge 1's pattern junk bits outside its own care + // but inside edge 0's. The matcher ignores them, so the two-predicate + // program is unchanged; an OR without masking would read them as part of + // edge 0's pattern and answer something else. + let junk: [u8; 16] = core::array::from_fn(|k| pe1[k] | (ca[k] & !ce1[k])); + assert!( + (0..16).any(|k| junk[k] & ca[k] & !pa[k] != 0), + "the junk must hit edge 0's care" + ); + let two_junk = Program::new( + vec![ + ma(None, 0), + MaskOp::Pred { + pred: Pred::MatchFacet16Strided { + lane: 0, + pattern: junk, + care: ce1, + }, + under: None, + dst: 1, + }, + MaskOp::And { + a: Operand::Scratch(0), + b: Operand::Scratch(1), + dst: 2, + }, + ], + count_of(Operand::Scratch(2)), + ); + let single = |p: [u8; 16], c: [u8; 16]| { + Program::new( + vec![MaskOp::Pred { + pred: Pred::MatchFacet16Strided { + lane: 0, + pattern: p, + care: c, + }, + under: None, + dst: 0, + }], + count_of(Operand::Scratch(0)), + ) + }; + let (mp, mc) = merge(&pa, &ca, &junk, &ce1); + let unmasked_p: [u8; 16] = core::array::from_fn(|k| pa[k] | junk[k]); + let (cj, _) = run_arm("Cj", &two_junk, &planes, 1, 0); + let (cm, _) = run_arm("Mj", &single(mp, mc), &planes, 1, 0); + let (cu, _) = run_arm("Uj", &single(unmasked_p, mc), &planes, 1, 0); + assert_eq!(cj, want_w, "bits outside care are ignored by the matcher"); + assert_eq!(cm, want_w, "the masked merge equals the conjunction"); + assert_ne!(cu, want_w, "an unmasked OR reads junk as pattern"); + println!(" unmasked operands: two predicates {cj}, masked merge {cm}, unmasked OR {cu}"); + // The precondition is load-bearing: two patterns that DISAGREE on a shared - // care bit have an empty conjunction, and an unconditional `p1 | p2` merge - // would answer something else. + // care bit have an empty conjunction, and a union that skips the conflict + // check would answer something else. let (px, cx) = ce64_pattern(0, PEARL_SHIFT, 3, p ^ 1); // same field, other value let conflict = Program::new( vec![ @@ -1000,8 +1065,8 @@ fn ce64(n: usize, reps: usize) { ], count_of(Operand::Scratch(2)), ); - let naive_p: [u8; 16] = core::array::from_fn(|k| pa[k] | px[k]); - let naive_c: [u8; 16] = core::array::from_fn(|k| ca[k] | cx[k]); + // Both operands are already masked; the conflict alone makes the union wrong. + let (naive_p, naive_c) = merge(&pa, &ca, &px, &cx); let naive = Program::new( vec![MaskOp::Pred { pred: Pred::MatchFacet16Strided { @@ -1019,9 +1084,9 @@ fn ce64(n: usize, reps: usize) { assert_eq!(cc0, 0, "conflicting patterns have no common row"); assert_ne!( cn, 0, - "an unconditional OR-merge answers a different question" + "a union without the conflict check answers a different question" ); - println!(" conflicting cares: two predicates {cc0}, unconditional OR-merge {cn} (refuse or constant-false, never merge)"); + println!(" conflicting cares: two predicates {cc0}, union without the conflict check {cn} (empty mask, never merge)"); // can fire: the wrong bit range must disagree let (pw, cw) = ce64_pattern(0, PEARL_SHIFT + 1, 3, p);