From 9982e9c4e48f08fc15fa392ab9a26b6a2b555c19 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 12:21:24 +0000 Subject: [PATCH 1/4] =?UTF-8?q?mask-risc:=20the=20terminal=20elects=20mate?= =?UTF-8?q?rialization=20=E2=80=94=20fused=20Range=20=E2=88=A9=20plane=20?= =?UTF-8?q?=E2=86=92=20Count/Any?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A mask expression denotes membership; whether it becomes bits is the terminal's decision. Program::fused_terminal() recognises Pred::Range{lo,hi} optionally gated by a RESIDENT plane and folded by Count or Any, and execute_into evaluates it over the plane's touched words plus two register-masked edge words: no scratch slot carved, no derived membership word written, no row id. Keep is never fused — it is the explicit election of a bitmap. - ir.rs: FusedTerminal / FusedFold, Program::fused_terminal, derived Program::requires_scratch, touched_words (the one span spelling) - exec.rs: fused dispatch after the slot-ceiling check, validation kept total via a local bookkeeping word; Scratch::for_program / over_for_program carve zero slots when !requires_scratch - tests/fused_terminal.rs: fold vs Keep vs scalar oracle over 4 row counts x 4 plane shapes x the named edge ranges; poisoned-arena gate (fold writes nothing, Keep's carve is visible); allocation twin; exact-shape guard; touched-word law; validation still refuses No ndarray primitive needed: the range's interior mask is all ones, so existing popcount_batch_u64 / mask_any over the borrowed plane suffice. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01GXUahz73MZxtxWcfpHp9dG --- crates/lance-graph-mask-risc/src/exec.rs | 82 +++- crates/lance-graph-mask-risc/src/ir.rs | 90 ++++ crates/lance-graph-mask-risc/src/lib.rs | 5 +- .../tests/fused_terminal.rs | 422 ++++++++++++++++++ 4 files changed, 590 insertions(+), 9 deletions(-) create mode 100644 crates/lance-graph-mask-risc/tests/fused_terminal.rs diff --git a/crates/lance-graph-mask-risc/src/exec.rs b/crates/lance-graph-mask-risc/src/exec.rs index 13940b677..474812c43 100644 --- a/crates/lance-graph-mask-risc/src/exec.rs +++ b/crates/lance-graph-mask-risc/src/exec.rs @@ -38,8 +38,8 @@ use ndarray::simd::{ }; use crate::ir::{ - Foreign, GroupFold, GroupKey, LaneRef, MaskOp, Operand, Planes, Pred, Program, Terminal, - MAX_SCRATCH_SLOTS, + touched_words, Foreign, FusedFold, FusedTerminal, GroupFold, GroupKey, LaneRef, MaskOp, + Operand, Planes, Pred, Program, Terminal, MAX_SCRATCH_SLOTS, }; use crate::reference::{out_shape, validate}; use crate::ternlog_dispatch::{ternlog_dispatch, ternlog_dispatch_assign}; @@ -193,10 +193,7 @@ impl Scratch<'static> { } // `scratch_slots` is a `u32` count; a program naming slot `u16::MAX` // needs 65,536 buffers, which fits `usize` on every supported target. - Ok(Self::new( - tile_words_for(n_rows), - program.scratch_slots as usize, - )) + Ok(Self::new(tile_words_for(n_rows), slots_needed(program))) } } @@ -252,7 +249,7 @@ impl<'a> Scratch<'a> { declared: program.scratch_slots, }); } - Self::over(buf, tile_words_for(n_rows), program.scratch_slots as usize) + Self::over(buf, tile_words_for(n_rows), slots_needed(program)) } /// Words per slot. @@ -650,6 +647,68 @@ pub fn execute( ) } +/// Slots [`Scratch::for_program`] / [`Scratch::over_for_program`] carve: none +/// for a program [`Program::requires_scratch`] says needs none. +fn slots_needed(program: &Program) -> usize { + if program.requires_scratch() { + program.scratch_slots as usize + } else { + 0 + } +} + +/// The bits of word `w` that fall inside `[lo, hi)`. +fn edge_mask(w: usize, lo: u32, hi: u32) -> u64 { + let base = w * 64; + let from = (lo as usize).saturating_sub(base).min(64); + let to = (hi as usize - base).min(64); + let upper = if to == 64 { u64::MAX } else { (1u64 << to) - 1 }; + upper & (u64::MAX << from) +} + +/// Evaluate a [`FusedTerminal`] over the resident plane's touched words. +/// +/// Nothing is written: the interior words are read from the borrowed plane +/// as they are (a range's interior mask is all ones), and the two edge words +/// are masked in a register. `validate` has already proven `lo <= hi <= +/// n_rows` and the plane index in range. +fn run_fused(f: FusedTerminal, planes: &Planes<'_>) -> Value { + let span = touched_words(f.lo, f.hi); + let Some(p) = f.plane else { + return match f.fold { + FusedFold::Count => Value::Count((f.hi - f.lo) as usize), + FusedFold::Any => Value::Bool(f.lo < f.hi), + }; + }; + if span.is_empty() { + return match f.fold { + FusedFold::Count => Value::Count(0), + FusedFold::Any => Value::Bool(false), + }; + } + let plane = planes.masks[usize::from(p)]; + let (first, last) = (span.start, span.end - 1); + let head = [plane[first] & edge_mask(first, f.lo, f.hi)]; + let tail = [plane[last] & edge_mask(last, f.lo, f.hi)]; + let interior = if last > first + 1 { + &plane[first + 1..last] + } else { + &[][..] + }; + match f.fold { + FusedFold::Count => { + let mut n = popcount_batch_u64(&head) + popcount_batch_u64(interior); + if last != first { + n += popcount_batch_u64(&tail); + } + Value::Count(n as usize) + } + FusedFold::Any => { + Value::Bool(mask_any(&head) || mask_any(interior) || (last != first && mask_any(&tail))) + } + } +} + /// Fold one tile's `Option` reduction into the running one. fn fold_opt(acc: Option, tile: Option, f: fn(i32, i32) -> i32) -> Option { match (acc, tile) { @@ -702,6 +761,15 @@ pub fn execute_into( declared: program.scratch_slots, }); } + // A fused program folds from its operands: it reads no slot and writes no + // membership bit, so the scratch capacity checks below do not apply to it. + // Validation stays total — the one declared slot is tracked in a local + // word of read-before-write bookkeeping, never in the caller's arena. + if let Some(f) = program.fused_terminal() { + let mut written = [0u64; 1]; + validate(program, planes, foreign, out_shape(&out), &mut written)?; + return Ok(run_fused(f, planes)); + } if scratch.slots() < program.scratch_slots as usize { return Err(ExecError::ScratchTooSmall { need: program.scratch_slots, diff --git a/crates/lance-graph-mask-risc/src/ir.rs b/crates/lance-graph-mask-risc/src/ir.rs index f931e5340..c18173367 100644 --- a/crates/lance-graph-mask-risc/src/ir.rs +++ b/crates/lance-graph-mask-risc/src/ir.rs @@ -563,6 +563,63 @@ impl Program { } } + /// The fused lowering of this program, if its terminal can fold straight + /// from its operands without any derived membership bits being written. + /// + /// A mask expression denotes membership; whether that membership ever + /// becomes a bitmap is the TERMINAL's decision, not the op's. The ops of + /// this IR are assignments (`dst = …`), so read literally every program + /// writes its intermediate membership into a scratch slot before the + /// terminal reads it back. For a demanded result that is a scalar — + /// `Count`, `Any` — that write is a materialization nobody asked for. + /// + /// The shape recognised here is `Range[lo, hi)` optionally gated by a + /// RESIDENT plane, folded by `Count` or `Any`: the range's interior words + /// have an all-ones mask, so the whole relation is the resident plane's own + /// words over the touched span plus two edge masks. No slot is needed. + /// + /// Everything else returns `None` and runs the ordinary tiled path. + /// `Keep` in particular is NEVER fused: it is the explicit election of a + /// bitmap, and the materialization is its whole point. + pub fn fused_terminal(&self) -> Option { + let [MaskOp::Pred { + pred: Pred::Range { lo, hi }, + under, + dst, + }] = self.ops.as_slice() + else { + return None; + }; + let plane = match under { + None => None, + Some(Operand::Plane(p)) => Some(*p), + // A scratch gate is itself derived membership; nothing here holds + // it, so the shape is not fusable. + Some(Operand::Scratch(_)) => return None, + }; + let fold = match self.terminal { + Terminal::Count { mask } if mask == Operand::Scratch(*dst) => FusedFold::Count, + Terminal::Any { mask } if mask == Operand::Scratch(*dst) => FusedFold::Any, + _ => return None, + }; + Some(FusedTerminal { + lo: *lo, + hi: *hi, + plane, + fold, + }) + } + + /// Whether executing this program needs any scratch slot at all. + /// + /// DERIVED, not declared: a flag is a claim, a derived predicate is a + /// proof. `false` exactly when [`Program::fused_terminal`] lowers the + /// program (or it names no slot), and [`crate::Scratch::for_program`] + /// carves zero slots for such a program. + pub fn requires_scratch(&self) -> bool { + self.scratch_slots > 0 && self.fused_terminal().is_none() + } + /// Count of ops of each physical kind — the "logical ops vs physical /// passes" bookkeeping the benchmark reports. pub fn op_histogram(&self) -> OpHistogram { @@ -607,6 +664,39 @@ impl Program { } } +/// A program the executor folds without writing membership bits: see +/// [`Program::fused_terminal`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct FusedTerminal { + /// First row of the range. + pub lo: u32, + /// One past the last row of the range. + pub hi: u32, + /// The resident plane gating the range, or `None` for a bare range. + pub plane: Option, + /// The scalar the terminal demands. + pub fold: FusedFold, +} + +/// The scalar folds that may consume membership without materializing it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum FusedFold { + /// Population of the relation. + Count, + /// Whether the relation is non-empty. + Any, +} + +/// The mask words a row range `[lo, hi)` touches: none when `lo == hi`, +/// otherwise `floor(lo / 64) ..= floor((hi - 1) / 64)`. One spelling, used by +/// the fused executor and by the tests that pin the touched-word law. +pub fn touched_words(lo: u32, hi: u32) -> core::ops::Range { + if lo >= hi { + return 0..0; + } + (lo as usize / 64)..((hi as usize - 1) / 64 + 1) +} + /// Per-kind op counts of a program. #[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] pub struct OpHistogram { diff --git a/crates/lance-graph-mask-risc/src/lib.rs b/crates/lance-graph-mask-risc/src/lib.rs index 1fc256061..8fd51bb2c 100644 --- a/crates/lance-graph-mask-risc/src/lib.rs +++ b/crates/lance-graph-mask-risc/src/lib.rs @@ -123,8 +123,9 @@ pub use exec::{ }; pub use fuse::{fuse, fuse_program, ternlog_imm, BoolExpr, FuseError, Fused}; pub use ir::{ - Foreign, ForeignPlane, GroupFold, GroupKey, LaneRef, MaskOp, Operand, Planes, Pred, Program, - Terminal, GROUP_SUM_SYM_MAX_ROWS, MASKED_SUM_I32_MAX_ROWS, MAX_SCRATCH_SLOTS, + touched_words, Foreign, ForeignPlane, FusedFold, FusedTerminal, GroupFold, GroupKey, LaneRef, + MaskOp, Operand, Planes, Pred, Program, Terminal, GROUP_SUM_SYM_MAX_ROWS, + MASKED_SUM_I32_MAX_ROWS, MAX_SCRATCH_SLOTS, }; pub use reference::{ reference_execute, reference_execute_into, reference_scratch, reference_scratch_with_foreign, diff --git a/crates/lance-graph-mask-risc/tests/fused_terminal.rs b/crates/lance-graph-mask-risc/tests/fused_terminal.rs new file mode 100644 index 000000000..901789e67 --- /dev/null +++ b/crates/lance-graph-mask-risc/tests/fused_terminal.rs @@ -0,0 +1,422 @@ +//! The terminal decides whether membership becomes bits. +//! +//! `Range[lo, hi) ∩ resident plane` has two legal physical endings over ONE +//! membership relation: +//! +//! - FOLD — `Count` / `Any`: the executor reads the resident plane's touched +//! words and two register-masked edge words. No scratch slot is carved, no +//! derived membership word is written, no row id exists. +//! - MATERIALIZE — `Keep`: the relation is written to the demanded +//! `Out::Mask`, because keeping the set is what `Keep` asks for. +//! +//! Every case below checks both arms against each other and against a scalar +//! oracle, and the gates check that the fold arm materializes NOTHING — not +//! "less than O(N)", nothing: a tile-sized scratch mask written only to be +//! counted is still a materialization. + +use std::alloc::{GlobalAlloc, Layout, System}; +use std::sync::atomic::{AtomicUsize, Ordering}; + +use lance_graph_mask_risc::exec::{execute_into, Scratch}; +use lance_graph_mask_risc::{ + touched_words, Foreign, FusedFold, MaskOp, Operand, Out, Planes, Pred, Program, Terminal, Value, +}; + +struct Counting; + +static BYTES: AtomicUsize = AtomicUsize::new(0); + +// SAFETY: a pure pass-through to `System`; the counter is the only addition. +unsafe impl GlobalAlloc for Counting { + unsafe fn alloc(&self, layout: Layout) -> *mut u8 { + BYTES.fetch_add(layout.size(), Ordering::Relaxed); + // SAFETY: same layout, same contract as the caller's. + unsafe { System.alloc(layout) } + } + unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) { + // SAFETY: `ptr` came from `alloc` above with this `layout`. + unsafe { System.dealloc(ptr, layout) } + } +} + +#[global_allocator] +static A: Counting = Counting; + +fn lcg(seed: &mut u64) -> u64 { + *seed = seed + .wrapping_mul(6364136223846793005) + .wrapping_add(1442695040888963407); + *seed >> 11 +} + +fn words_for(n: usize) -> usize { + n.div_ceil(64) +} + +/// A resident plane of `n` rows with `set(row)` bits, tail bits clear. +fn plane(n: usize, set: impl Fn(usize) -> bool) -> Vec { + let mut w = vec![0u64; words_for(n)]; + for r in 0..n { + if set(r) { + w[r / 64] |= 1u64 << (r % 64); + } + } + w +} + +fn range_program(lo: u32, hi: u32, gate: Option, terminal: Terminal) -> Program { + Program::new( + vec![MaskOp::Pred { + pred: Pred::Range { lo, hi }, + under: gate, + dst: 0, + }], + terminal, + ) +} + +fn oracle(p: &[u64], lo: usize, hi: usize) -> usize { + (lo..hi).filter(|&r| p[r / 64] >> (r % 64) & 1 == 1).count() +} + +/// Run the FOLD arm with a scratch of zero slots — a scratch that cannot +/// hold a single derived word, so a fold that needed one could not run. +fn fold(p: &Program, planes: &Planes<'_>) -> Value { + let mut s = Scratch::new(0, 0); + execute_into(p, planes, &Foreign::NONE, &mut s, Out::None).expect("fold arm") +} + +/// Run the MATERIALIZE arm and return the kept mask. +fn keep(lo: u32, hi: u32, planes: &Planes<'_>) -> Vec { + let p = range_program( + lo, + hi, + Some(Operand::Plane(0)), + Terminal::Keep { + mask: Operand::Scratch(0), + }, + ); + assert!(p.fused_terminal().is_none(), "Keep is never fused"); + let mut s = Scratch::for_program(&p, planes.n_rows).expect("scratch"); + let mut out = vec![0u64; words_for(planes.n_rows)]; + execute_into(&p, planes, &Foreign::NONE, &mut s, Out::Mask(&mut out)).expect("keep arm"); + out +} + +/// The flagship cases the capstone names, clipped to `n`. +fn ranges(n: u32) -> Vec<(u32, u32)> { + let mut v = vec![ + (0, 0), + (65.min(n), 65.min(n)), + (0, 1), + (n - 1, n), + (0, 64.min(n)), // aligned lo, aligned hi + (3, 60.min(n)), // inside one word + (60.min(n - 1), n.min(70)), // across one word boundary + (64.min(n), n), // aligned lo, unaligned-or-whole hi + (0, n), // whole population + ]; + if n > 700 { + v.push((1, 700)); // crosses more than one 512-row tile + v.push((130, 1100)); + } + v.retain(|&(lo, hi)| lo <= hi && hi <= n); + v +} + +/// FAILS IF: the fold arm and the materialize arm disagree with each other or +/// with the scalar oracle for any range × plane shape — including an empty +/// range, a single row, word-straddling edges, a sub-64-row tail and the +/// whole population. +#[test] +fn fold_and_keep_arms_agree_with_the_oracle() { + let mut seed = 0x5eed_u64; + let mut cases = 0; + for n in [37usize, 64, 1024, 1317] { + let mut shapes: Vec<(&str, Vec)> = vec![ + ("zero", plane(n, |_| false)), + ("one", plane(n, |_| true)), + ("clustered", plane(n, |r| (r / 97) % 3 == 1)), + ]; + let scattered: Vec = (0..n).map(|_| lcg(&mut seed) % 5 == 0).collect(); + shapes.push(("scattered", plane(n, |r| scattered[r]))); + for (name, p) in &shapes { + let masks: [&[u64]; 1] = [p]; + let planes = Planes { + n_rows: n, + masks: &masks, + lanes: &[], + }; + for (lo, hi) in ranges(n as u32) { + let want = oracle(p, lo as usize, hi as usize); + let gate = Some(Operand::Plane(0)); + let count = range_program( + lo, + hi, + gate, + Terminal::Count { + mask: Operand::Scratch(0), + }, + ); + let any = range_program( + lo, + hi, + gate, + Terminal::Any { + mask: Operand::Scratch(0), + }, + ); + assert_eq!( + count.fused_terminal().map(|f| f.fold), + Some(FusedFold::Count) + ); + assert_eq!(any.fused_terminal().map(|f| f.fold), Some(FusedFold::Any)); + let kept = keep(lo, hi, &planes); + let kept_count: usize = kept.iter().map(|w| w.count_ones() as usize).sum(); + let ctx = format!("{name} n={n} [{lo},{hi})"); + assert_eq!(fold(&count, &planes), Value::Count(want), "count {ctx}"); + assert_eq!(kept_count, want, "keep→popcount {ctx}"); + assert_eq!(fold(&any, &planes), Value::Bool(want > 0), "any {ctx}"); + assert_eq!(kept.iter().any(|&w| w != 0), want > 0, "keep→any {ctx}"); + cases += 1; + } + } + } + // Anti-vacuity: the grid really covers the named shapes. + assert!(cases >= 100, "only {cases} cases ran"); +} + +/// FAILS IF: a bare range (no gate) is not folded to `hi - lo` / `lo < hi` +/// without a scratch — the W2a case, where not even the plane is read. +#[test] +fn a_bare_range_folds_to_its_own_width() { + let n = 1317usize; + let p = plane(n, |_| false); + let masks: [&[u64]; 1] = [&p]; + let planes = Planes { + n_rows: n, + masks: &masks, + lanes: &[], + }; + for (lo, hi) in ranges(n as u32) { + let count = range_program( + lo, + hi, + None, + Terminal::Count { + mask: Operand::Scratch(0), + }, + ); + let any = range_program( + lo, + hi, + None, + Terminal::Any { + mask: Operand::Scratch(0), + }, + ); + assert!(!count.requires_scratch()); + assert_eq!(fold(&count, &planes), Value::Count((hi - lo) as usize)); + assert_eq!(fold(&any, &planes), Value::Bool(lo < hi)); + } +} + +/// THE LOAD-BEARING GATE. FAILS IF: the fold arm carves a scratch slot or +/// writes a single word into the caller's arena. +/// +/// The caller's buffer is poisoned to `u64::MAX` and handed to +/// `Scratch::over_for_program`, which zero-fills whatever it carves. After +/// construction AND execution the buffer must still be all `u64::MAX`: any +/// carved slot, and any derived membership word, would show up as a changed +/// word. The twin half runs the `Keep` program through the same buffer and +/// proves the probe can see a carve. +#[test] +fn the_fold_arm_writes_no_derived_membership_word() { + let n = 4096usize; + let p = plane(n, |r| r % 3 == 0); + let masks: [&[u64]; 1] = [&p]; + let planes = Planes { + n_rows: n, + masks: &masks, + lanes: &[], + }; + let (lo, hi) = (70u32, 3000u32); + for terminal in [ + Terminal::Count { + mask: Operand::Scratch(0), + }, + Terminal::Any { + mask: Operand::Scratch(0), + }, + ] { + let prog = range_program(lo, hi, Some(Operand::Plane(0)), terminal); + assert!(!prog.requires_scratch(), "a fused fold needs no scratch"); + let mut buf = vec![u64::MAX; 64]; + { + let mut s = Scratch::over_for_program(&mut buf, &prog, n).expect("carve"); + assert_eq!(s.slots(), 0, "no slot is carved for a fold"); + execute_into(&prog, &planes, &Foreign::NONE, &mut s, Out::None).expect("fold"); + } + assert!( + buf.iter().all(|&w| w == u64::MAX), + "the fold arm touched the caller's arena" + ); + } + + // Can-it-fire: the same probe sees the materialize arm's carve. + let keep_prog = range_program( + lo, + hi, + Some(Operand::Plane(0)), + Terminal::Keep { + mask: Operand::Scratch(0), + }, + ); + assert!(keep_prog.requires_scratch()); + let mut buf = vec![u64::MAX; 64]; + { + let mut s = Scratch::over_for_program(&mut buf, &keep_prog, n).expect("carve"); + let mut out = vec![0u64; words_for(n)]; + execute_into( + &keep_prog, + &planes, + &Foreign::NONE, + &mut s, + Out::Mask(&mut out), + ) + .expect("keep"); + assert!(out.iter().any(|&w| w != 0), "Keep materialized the set"); + } + assert!( + buf.iter().any(|&w| w != u64::MAX), + "the probe cannot see a carve, so its silence above proves nothing" + ); +} + +/// FAILS IF: `Scratch::for_program` allocates for a fold, or the counter is +/// inert (the `Keep` half must allocate). +#[test] +fn sizing_a_fold_allocates_nothing() { + let count = range_program( + 5, + 900, + Some(Operand::Plane(0)), + Terminal::Count { + mask: Operand::Scratch(0), + }, + ); + let keep_prog = range_program( + 5, + 900, + Some(Operand::Plane(0)), + Terminal::Keep { + mask: Operand::Scratch(0), + }, + ); + let before = BYTES.load(Ordering::Relaxed); + let s = Scratch::for_program(&count, 1 << 20).expect("scratch"); + let fold_bytes = BYTES.load(Ordering::Relaxed) - before; + drop(s); + let before = BYTES.load(Ordering::Relaxed); + let s = Scratch::for_program(&keep_prog, 1 << 20).expect("scratch"); + let keep_bytes = BYTES.load(Ordering::Relaxed) - before; + drop(s); + assert_eq!(fold_bytes, 0, "a fold carved a scratch arena"); + assert!(keep_bytes > 0, "the counter cannot see an allocation"); +} + +/// FAILS IF: a shape that holds derived membership elsewhere is wrongly +/// fused — a scratch gate, a second op, or a terminal reading another slot. +#[test] +fn only_the_exact_shape_is_fused() { + let count = Terminal::Count { + mask: Operand::Scratch(0), + }; + assert!(range_program(0, 9, Some(Operand::Scratch(1)), count) + .fused_terminal() + .is_none()); + let two_ops = Program::new( + vec![ + MaskOp::Pred { + pred: Pred::Range { lo: 0, hi: 9 }, + under: None, + dst: 0, + }, + MaskOp::And { + a: Operand::Scratch(0), + b: Operand::Plane(0), + dst: 1, + }, + ], + Terminal::Count { + mask: Operand::Scratch(1), + }, + ); + assert!(two_ops.fused_terminal().is_none()); + assert!(two_ops.requires_scratch()); + assert!(range_program( + 0, + 9, + None, + Terminal::All { + mask: Operand::Scratch(0) + } + ) + .fused_terminal() + .is_none()); +} + +/// FAILS IF: the touched-word span departs from the law +/// `lo == hi → 0` else `floor((hi-1)/64) - floor(lo/64) + 1`. +#[test] +fn touched_words_obey_the_law() { + let mut checked = 0; + for lo in [0u32, 1, 63, 64, 65, 127, 128, 500, 511, 512, 513] { + for width in [0u32, 1, 2, 63, 64, 65, 128, 600] { + let hi = lo + width; + let want = if lo == hi { + 0 + } else { + ((hi - 1) / 64 - lo / 64 + 1) as usize + }; + assert_eq!(touched_words(lo, hi).len(), want, "[{lo},{hi})"); + checked += 1; + } + } + assert_eq!(touched_words(65, 65), 0..0); + assert_eq!(touched_words(0, 0), 0..0); + assert!(checked > 80); +} + +/// FAILS IF: validation is skipped on the fused path — an out-of-bounds range +/// or an absent plane must still be refused, not read. +#[test] +fn the_fused_path_still_validates() { + let n = 100usize; + let p = plane(n, |_| true); + let masks: [&[u64]; 1] = [&p]; + let planes = Planes { + n_rows: n, + masks: &masks, + lanes: &[], + }; + let mut s = Scratch::new(0, 0); + let too_far = range_program( + 0, + 101, + Some(Operand::Plane(0)), + Terminal::Count { + mask: Operand::Scratch(0), + }, + ); + assert!(execute_into(&too_far, &planes, &Foreign::NONE, &mut s, Out::None).is_err()); + let no_plane = range_program( + 0, + 10, + Some(Operand::Plane(3)), + Terminal::Count { + mask: Operand::Scratch(0), + }, + ); + assert!(execute_into(&no_plane, &planes, &Foreign::NONE, &mut s, Out::None).is_err()); +} From 1d3e3366c5b4616cc33fc2fbf66460e2a89037c0 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 12:22:17 +0000 Subject: [PATCH 2/4] mask-risc tests: count allocations per thread, not per process The harness runs tests on parallel threads; a process-wide counter picked up 120 stray bytes from a sibling test inside the sizing window. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01GXUahz73MZxtxWcfpHp9dG --- .../tests/fused_terminal.rs | 23 +++++++++++++------ 1 file changed, 16 insertions(+), 7 deletions(-) diff --git a/crates/lance-graph-mask-risc/tests/fused_terminal.rs b/crates/lance-graph-mask-risc/tests/fused_terminal.rs index 901789e67..a4ec3fc1b 100644 --- a/crates/lance-graph-mask-risc/tests/fused_terminal.rs +++ b/crates/lance-graph-mask-risc/tests/fused_terminal.rs @@ -15,7 +15,7 @@ //! counted is still a materialization. use std::alloc::{GlobalAlloc, Layout, System}; -use std::sync::atomic::{AtomicUsize, Ordering}; +use std::cell::Cell; use lance_graph_mask_risc::exec::{execute_into, Scratch}; use lance_graph_mask_risc::{ @@ -24,12 +24,21 @@ use lance_graph_mask_risc::{ struct Counting; -static BYTES: AtomicUsize = AtomicUsize::new(0); +thread_local! { + // Per THREAD, not per process: the test harness runs tests on parallel + // threads, and a process-wide counter picks up their allocations inside + // this test's window (measured: 120 stray bytes on an otherwise clean run). + static BYTES: Cell = const { Cell::new(0) }; +} + +fn bytes() -> usize { + BYTES.with(Cell::get) +} // SAFETY: a pure pass-through to `System`; the counter is the only addition. unsafe impl GlobalAlloc for Counting { unsafe fn alloc(&self, layout: Layout) -> *mut u8 { - BYTES.fetch_add(layout.size(), Ordering::Relaxed); + let _ = BYTES.try_with(|b| b.set(b.get() + layout.size())); // SAFETY: same layout, same contract as the caller's. unsafe { System.alloc(layout) } } @@ -313,13 +322,13 @@ fn sizing_a_fold_allocates_nothing() { mask: Operand::Scratch(0), }, ); - let before = BYTES.load(Ordering::Relaxed); + let before = bytes(); let s = Scratch::for_program(&count, 1 << 20).expect("scratch"); - let fold_bytes = BYTES.load(Ordering::Relaxed) - before; + let fold_bytes = bytes() - before; drop(s); - let before = BYTES.load(Ordering::Relaxed); + let before = bytes(); let s = Scratch::for_program(&keep_prog, 1 << 20).expect("scratch"); - let keep_bytes = BYTES.load(Ordering::Relaxed) - before; + let keep_bytes = bytes() - before; drop(s); assert_eq!(fold_bytes, 0, "a fold carved a scratch arena"); assert!(keep_bytes > 0, "the counter cannot see an allocation"); From 021e6a5b07279df01806ac1ac247d379e1792c93 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 12:24:15 +0000 Subject: [PATCH 3/4] =?UTF-8?q?mask-risc:=20range=5Ffused=5Fprobe=20?= =?UTF-8?q?=E2=80=94=20materialized=20vs=20fused=20Range=20=E2=88=A9=20pla?= =?UTF-8?q?ne=20=E2=86=92=20Count?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Reports latency (median), derived membership words written and plane words read separately; asserts both arms agree on every case. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01GXUahz73MZxtxWcfpHp9dG --- .../examples/range_fused_probe.rs | 145 ++++++++++++++++++ .../tests/fused_terminal.rs | 2 +- 2 files changed, 146 insertions(+), 1 deletion(-) create mode 100644 crates/lance-graph-mask-risc/examples/range_fused_probe.rs diff --git a/crates/lance-graph-mask-risc/examples/range_fused_probe.rs b/crates/lance-graph-mask-risc/examples/range_fused_probe.rs new file mode 100644 index 000000000..8b3756b93 --- /dev/null +++ b/crates/lance-graph-mask-risc/examples/range_fused_probe.rs @@ -0,0 +1,145 @@ +//! Measure `Range[lo, hi) ∩ resident plane → Count` two ways over the SAME +//! membership relation: +//! +//! - MATERIALIZED: `Pred::Range → Scratch(0)`, `And(Scratch(0), Plane(0)) → +//! Scratch(1)`, `Count(Scratch(1))` — the ops write derived membership one +//! tile at a time before the terminal counts it. +//! - FUSED: `Pred::Range` gated by the plane, `Count` — the executor folds the +//! plane's touched words and two register-masked edge words, writing none. +//! +//! Reported separately, never collapsed into one "speedup": latency (median +//! of repeats), derived membership words written, and plane words read. The +//! word counts are DERIVED from the executor's contract (the tiled path +//! writes every tile of every slot; the fused path reads `touched_words`), and +//! the fused count is cross-checked against the materialized one each run. +//! +//! `cargo run --release -p lance-graph-mask-risc --example range_fused_probe` + +use std::time::Instant; + +use lance_graph_mask_risc::exec::{execute_into, Scratch}; +use lance_graph_mask_risc::{ + touched_words, Foreign, MaskOp, Operand, Out, Planes, Pred, Program, Terminal, Value, +}; + +fn lcg(seed: &mut u64) -> u64 { + *seed = seed + .wrapping_mul(6364136223846793005) + .wrapping_add(1442695040888963407); + *seed >> 11 +} + +fn plane(n: usize, set: impl Fn(usize) -> bool) -> Vec { + let mut w = vec![0u64; n.div_ceil(64)]; + for r in 0..n { + if set(r) { + w[r / 64] |= 1u64 << (r % 64); + } + } + w +} + +fn median_ns(mut f: impl FnMut() -> Value, reps: usize) -> (f64, Value) { + let mut times = Vec::with_capacity(reps); + let mut last = Value::Count(0); + for _ in 0..reps { + let t = Instant::now(); + last = std::hint::black_box(f()); + times.push(t.elapsed().as_nanos() as f64); + } + times.sort_by(|a, b| a.total_cmp(b)); + (times[reps / 2], last) +} + +fn main() { + println!( + "{:>8} {:>10} {:>18} {:>12} {:>12} {:>10} {:>10} {:>9} {:>9}", + "N", "range", "plane", "mat_ns", "fused_ns", "mat_wr", "fused_wr", "mat_rd", "fused_rd" + ); + let mut seed = 0xfeed_u64; + for n in [4_096usize, 65_536, 1_048_576] { + let scattered: Vec = (0..n).map(|_| lcg(&mut seed).is_multiple_of(97)).collect(); + let shapes: [(&str, Vec); 3] = [ + ("dense", plane(n, |r| r % 4 != 0)), + ("sparse-clustered", plane(n, |r| (r / 4096) % 50 == 7)), + ("sparse-scattered", plane(n, |r| scattered[r])), + ]; + let nn = n as u32; + let ranges: [(&str, u32, u32); 5] = [ + ("tiny", nn / 2, nn / 2 + 3), + ("one-tile", 1000.min(nn - 600), 1000.min(nn - 600) + 512), + ("1%", nn / 3, nn / 3 + nn / 100), + ("25%", nn / 5, nn / 5 + nn / 4), + ("whole", 0, nn), + ]; + let words = n.div_ceil(64); + for (pname, p) in &shapes { + let masks: [&[u64]; 1] = [p]; + let planes = Planes { + n_rows: n, + masks: &masks, + lanes: &[], + }; + for (rname, lo, hi) in ranges { + let materialized = Program::new( + vec![ + MaskOp::Pred { + pred: Pred::Range { lo, hi }, + under: None, + dst: 0, + }, + MaskOp::And { + a: Operand::Scratch(0), + b: Operand::Plane(0), + dst: 1, + }, + ], + Terminal::Count { + mask: Operand::Scratch(1), + }, + ); + let fused = Program::new( + vec![MaskOp::Pred { + pred: Pred::Range { lo, hi }, + under: Some(Operand::Plane(0)), + dst: 0, + }], + Terminal::Count { + mask: Operand::Scratch(0), + }, + ); + assert!(materialized.fused_terminal().is_none()); + assert!(fused.fused_terminal().is_some()); + let mut ms = Scratch::for_program(&materialized, n).expect("scratch"); + let mut fs = Scratch::for_program(&fused, n).expect("scratch"); + let reps = if n > 100_000 { 31 } else { 201 }; + let (mat_ns, mv) = median_ns( + || { + execute_into(&materialized, &planes, &Foreign::NONE, &mut ms, Out::None) + .expect("materialized") + }, + reps, + ); + let (fused_ns, fv) = median_ns( + || { + execute_into(&fused, &planes, &Foreign::NONE, &mut fs, Out::None) + .expect("fused") + }, + reps, + ); + assert_eq!(mv, fv, "the two arms disagree: {pname} {rname} N={n}"); + // Materialized: the range slot and the And slot are each + // written over every tile; the plane is read once over the + // whole population. Fused: nothing written; the plane is read + // over the touched span only. + let (mat_wr, mat_rd) = (2 * words, words); + let fused_rd = touched_words(lo, hi).len(); + println!( + "{n:>8} {rname:>10} {pname:>18} {mat_ns:>12.0} {fused_ns:>12.0} \ + {mat_wr:>10} {:>10} {mat_rd:>9} {fused_rd:>9}", + 0 + ); + } + } + } +} diff --git a/crates/lance-graph-mask-risc/tests/fused_terminal.rs b/crates/lance-graph-mask-risc/tests/fused_terminal.rs index a4ec3fc1b..48b0edb38 100644 --- a/crates/lance-graph-mask-risc/tests/fused_terminal.rs +++ b/crates/lance-graph-mask-risc/tests/fused_terminal.rs @@ -147,7 +147,7 @@ fn fold_and_keep_arms_agree_with_the_oracle() { ("one", plane(n, |_| true)), ("clustered", plane(n, |r| (r / 97) % 3 == 1)), ]; - let scattered: Vec = (0..n).map(|_| lcg(&mut seed) % 5 == 0).collect(); + let scattered: Vec = (0..n).map(|_| lcg(&mut seed).is_multiple_of(5)).collect(); shapes.push(("scattered", plane(n, |r| scattered[r]))); for (name, p) in &shapes { let masks: [&[u64]; 1] = [p]; From daa1585682a8515006adc99270dd43332ab669b3 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 12:25:30 +0000 Subject: [PATCH 4/4] =?UTF-8?q?board:=20terminal=20elects=20materializatio?= =?UTF-8?q?n=20=E2=80=94=20W2b-A=20shipped,=20T1-FUSED=E2=80=B2=20refuted?= =?UTF-8?q?=20for=20Range=20=E2=88=A9=20plane?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit New entry records the Wave 0 re-audit (tiled scratch made the population-sized claims stale; derived tile writes remained), the fused lowering, measurements, disable-verified falsifiers, and the WORKING-MODEL classification of the remaining ops. Status cells updated for D-WFL-W2b‴ and D-WFL-T1-FUSED′ only. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01GXUahz73MZxtxWcfpHp9dG --- .claude/board/STATUS_BOARD.md | 4 +- ...inal-elects-materialization-range-plane.md | 95 +++++++++++++++++++ .claude/board/entries/README.md | 3 +- 3 files changed, 99 insertions(+), 3 deletions(-) create mode 100644 .claude/board/entries/2026-09-23-terminal-elects-materialization-range-plane.md diff --git a/.claude/board/STATUS_BOARD.md b/.claude/board/STATUS_BOARD.md index ff22b9903..d9d86d3fc 100644 --- a/.claude/board/STATUS_BOARD.md +++ b/.claude/board/STATUS_BOARD.md @@ -52,7 +52,7 @@ materialization, and that is the ideal Layer-0 operation, not a compromise. |---|---|---|---| | D-WFL-AXIS | **Three independent axes, not one binary:** OPERATORS (fold · mask/ternlog · project · rotate · neighbour · reduce) × CARRIERS (canonical lane · range · descriptor · resident mask · cached mask) × MATERIALIZATION CHOICE (fused vs materialized bitmap). The BBB question is not *fold or mask?* but **is this membership relation transient algebra, or has it been PROMOTED to a mask carrier?** | Queued | every plan must record the promotion, not the operator choice. Falsified if a plan can promote a membership relation to a carrier without that appearing in it | | D-WFL-ENTROPY | **Representation entropy should follow ANSWER entropy.** *"Do these two million-row regions intersect?"* ≈ 1 bit; building 125 KB of mask to find it is the obscenity — and 125 KB is Seam B's MEASURED number at N=1M, not rhetoric. *"How many overlap?"* = 32–64 bits; also fused. *"Give me the overlap, six thoughts will manipulate it"* justifies the bitmap, which is then low-entropy **relative to its future workload** | Queued | the ratio `materialized bytes : answer bytes` reported per operation (§6's `R_info`), with the downstream workload named whenever it exceeds 1 | -| D-WFL-W2b‴ | ⊘ supersedes W2b″'s "fold arm vs mask arm" — **both arms mask.** W2b-A: `Range × resident → FUSED masking → Count/Any`, no result mask. W2b-B: `→ masking → a MATERIALIZED bounded mask`. Same masking semantics, different result carrier | Queued | differential across arms and against the oracle. **W2b-B carries a burden W2b-A does not: it must NAME and MEASURE the downstream reuse justifying the carrier — a materialization with no demonstrated consumer FAILS the arm.** That is what deliberate promotion costs | +| D-WFL-W2b‴ | ⊘ supersedes W2b″'s "fold arm vs mask arm" — **both arms mask.** W2b-A: `Range × resident → FUSED masking → Count/Any`, no result mask. W2b-B: `→ masking → a MATERIALIZED bounded mask`. Same masking semantics, different result carrier | In PR — W2b-A shipped (fused `Range ∩ plane → Count/Any`, 0 derived words; `entries/2026-09-23-terminal-elects-materialization-range-plane.md`); W2b-B reuse burden OPEN | differential across arms and against the oracle. **W2b-B carries a burden W2b-A does not: it must NAME and MEASURE the downstream reuse justifying the carrier — a materialization with no demonstrated consumer FAILS the arm.** That is what deliberate promotion costs | **Shortest form:** fold the datasets, mask the folds, materialize only when the mask itself is worth keeping. @@ -69,7 +69,7 @@ defines what a fold IS — never what the machine may do. | D-WFL-SIBLING | **The rule governs the TRANSITION, not the bytes:** crossing from fold-native to mask-native execution must be deliberate and visible at the T2 planning membrane. Once MASK is elected, behaving like a mask engine (AND → TERNLOG → shift → cache) is legitimate. Forbidden only: the planner believes it is folding, a helper silently allocates `words_for(N)`, and nobody made the decision | Queued | the BBB question must be answerable for every plan: *who elected the mask, on what basis?* Falsified if a plan can become mask-native without an election appearing in it | | D-WFL-SEAMB′ | ⊘ **restates Seam B more precisely than every earlier framing** (performance complaint · T1 conformance failure · fold-law violation — all circling this). The defect in `Pred::Range` is NOT that it writes a mask. It is that the planner can neither elect nor decline: there is exactly ONE path, so **the choice does not exist**. Seam B is an ABSENT DECISION, not a present mask | Queued | fixed when both paths exist and the plan records which was taken — not when the mask disappears | | D-WFL-W2b″ | ⊘ **supersedes D-WFL-W2b′'s "must not write".** W2b demonstrates BOTH legal paths over identical semantics: FOLD-NATIVE (`Range ∩ resident → Count/Any`, no second mask) and MASK-NATIVE (`→ a bounded/cached mask` because a consumer reuses it). Pipeline vs materialize | Queued | the two arms differentially checked against each other AND the oracle — identical row sets, identical Count/Any. The earlier "zero derived buffers, asserted by counter" gate now scopes to the FOLD arm only | -| D-WFL-T1-FUSED′ | ⊘ upgrade from optimization to **enabler**: without a fused `popcount(a & b)` over a span there is no intermediate-buffer-free path, so **the fold-native arm does not exist at all**. The primitive CREATES the choice — which is exactly why Seam B had no decision in it | Queued | unchanged differential gate vs `mask_and` + `popcount_batch_u64`; the framing change raises its priority from nice-to-have to W2b-blocking | +| D-WFL-T1-FUSED′ | ⊘ upgrade from optimization to **enabler**: without a fused `popcount(a & b)` over a span there is no intermediate-buffer-free path, so **the fold-native arm does not exist at all**. The primitive CREATES the choice — which is exactly why Seam B had no decision in it | Refuted for `Range ∩ plane` (existing `popcount_batch_u64`/`mask_any` over the borrowed span suffice); plane∩plane `and_popcount` still OPEN | unchanged differential gate vs `mask_and` + `popcount_batch_u64`; the framing change raises its priority from nice-to-have to W2b-blocking | | D-WFL-ELECT | the election rule, static first: `terminal Count → FOLD`; `one AND then Count → probably FOLD`; `reuse_count > 1 → consider MASK`; `shared cached result → MASK`; `Wabe frontier reused → maybe MASK`; `~11 ns cached mask → almost certainly MASK`. DuckDB-style dynamic costing later | Queued | static rules must be inspectable in the plan. Falsified if the rule set fires the same way on every program (it would carry no information — cf. the can-it-stay-silent twin) | ## D-WFL — the cache scoping (2026-09-19): frozen is fine, marching is the disaster diff --git a/.claude/board/entries/2026-09-23-terminal-elects-materialization-range-plane.md b/.claude/board/entries/2026-09-23-terminal-elects-materialization-range-plane.md new file mode 100644 index 000000000..1812ee3d2 --- /dev/null +++ b/.claude/board/entries/2026-09-23-terminal-elects-materialization-range-plane.md @@ -0,0 +1,95 @@ +# 2026-09-23 — The terminal elects materialization: fused `Range ∩ plane → Count/Any` + +**Status:** MEASURED (flagship W2b-A) · OPEN (W2b-B reuse burden; plane∩plane and +compare→Count fusion; execution extent) +**D-ids:** D-WFL-W2b‴ (arm A shipped), D-WFL-T1-FUSED′ (refuted for this shape), +D-WFL-MASKOP / D-WFL-EXPR / D-WFL-SEAMB″ (corrected, still queued), D-WFL-FUSE (confirmed) + +## What was actually wrong (re-audited on main 33df0710) +- **Stale:** D-WFL-MASKOP's "`exec.rs:566` forces every slot to `words_for(n_rows)`" and the + plan's "a `Pred::Range` at 1M rows writes 125 KB". Since #1266, scratch is tiled + (`TILE_WORDS = 8`, `tile_words_for`), so a slot is at most 8 words. +- **Still true, and stricter:** `Range ∩ resident plane → Count` wrote derived membership + on every tile. Quack lowers to `Pred{Range, under: gate}`. The executor then ran + `mask_set_range` and `mask_and_assign` into the tile slot, and `Count` read it back with + `popcount_batch_u64`. The work was also not bounded by the touched span: every tile of the + population was walked, however narrow the range. + +## What changed +In `crates/lance-graph-mask-risc`: +- `Program::fused_terminal()` recognises `[Pred{Range, under: None | Plane}]` followed by + `Count` or `Any` of that slot. `Program::requires_scratch()` is DERIVED from it, and + `touched_words(lo, hi)` is the one spelling of the span. +- `execute_into` folds a fused program straight from the resident plane's touched words + plus two register-masked edge words. Validation stays total, using a local bookkeeping word. +- `Scratch::for_program` and `over_for_program` carve **zero** slots for such a program. +- `Keep` is never fused: it is the explicit election of a bitmap. + +No second evaluator, no new IR and no ndarray change. + +## What materialization disappeared, and what remains deliberate +- **Gone:** on the fold arm, derived words written = 0 and scratch slots carved = 0. Gated + by a poisoned caller arena that must stay all-`u64::MAX`; a twin test proves the same + probe sees `Keep`'s carve. +- **Allocation:** `Scratch::for_program` allocates 0 bytes for a fold, counted per thread. +- **Deliberate:** `Keep` still writes the demanded `Out::Mask`. + +## D-WFL-T1-FUSED′ refuted for this shape +No new primitive was needed. A range's interior mask is all ones, so the relation is the +plane's own words over the span. The existing `popcount_batch_u64` and `mask_any` are +enough, called on the borrowed slice and on two one-word register temporaries. This +confirms D-WFL-FUSE: it was a lowering rule, not a primitive. The general slice∩slice +`popcount(a & b)` still lacks a buffer-free primitive (see the classification below). + +## Measurements +`cargo run --release -p lance-graph-mask-risc --example range_fused_probe`. The two arms +agree on every case (asserted). Median ns: + +| N | range | plane | materialized | fused | derived words written, mat / fused | plane words read, mat / fused | +|---|---|---|---|---|---|---| +| 4,096 | tiny | dense | 653 | 54 | 128 / 0 | 64 / 1 | +| 65,536 | 25% | dense | 9,002 | 119 | 2,048 / 0 | 1,024 / 257 | +| 1,048,576 | tiny | dense | 138,620 | 56 | 32,768 / 0 | 16,384 / 1 | +| 1,048,576 | 25% | dense | 141,688 | 1,072 | 32,768 / 0 | 16,384 / 4,097 | +| 1,048,576 | whole | dense | 148,396 | 4,435 | 32,768 / 0 | 16,384 / 16,384 | + +- The fused latency is flat in N for a tiny range (54 → 56 ns from 4K to 1M rows) and + scales with the touched span, not the population. +- The word counts are derived from the executor's contract, not from instrumentation: + - Tiled path: two slots written over every tile. + - Fused path: reads exactly `touched_words`. + - The fused-arm zero is the one the gate test enforces. +- Sparse-scattered rows at 1M showed run-to-run noise, up to 297 µs on the materialized arm. + +## Falsifiers (`tests/fused_terminal.rs`, disable-verified) +| gate | disable | result | +|---|---|---| +| fold == Keep→popcount/any == scalar oracle, over 4 row counts × 4 plane shapes × the named edges (empty `[65,65)` / `[0,0)`, single row, aligned/unaligned, inside one word, across a word, across tiles, sub-64 tail, whole) | tail edge word not counted | red: `count one n=1024 [60,70)` | +| poisoned arena untouched + zero slots carved | always carve the declared slots | red | +| `for_program` allocates 0 bytes for a fold | same | red | +| fusion recognised at all | `fused_terminal` → `None` | 4 of 7 red | + +The allocation gate first flaked (120 stray bytes): the counter was process-wide while the +harness runs tests in parallel. It is now per-thread. + +## Classification of the remaining ops (WORKING-MODEL, not measured) +| op → scalar terminal | class | +|---|---| +| `Range`, bare or gated by a resident plane | **fused** (this entry) | +| `Not(a)` → Count | fusion rule: `n − popcount(a)`; no primitive | +| `And` / `AndNot(plane, plane)` → Count | needs a T1 primitive: slice-slice `and_popcount` (ndarray has `U64x8::xor_popcount` as the precedent, but no AND form) | +| `Or` → Count | fusion rule once `and_popcount` exists: `\|a\| + \|b\| − \|a∧b\|` | +| `Xor` → Count | needs a u64-slice XOR popcount (the fused `hamming_distance_raw` exists on bytes only) | +| `Ternlog` → Count | needs `ternlog_popcount`, which generalises the three above | +| lane compares (`Gt`/`Eq`/… `_to_mask_under`) → Count | needs compare-count primitives; today a tile-local write remains | +| `Gather`, `Keep`, `ScatterOr` | deliberate materialization | + +The desired direction is FEWER concepts: one `ternlog_popcount` would subsume `and`, `or`, +`xor` and `andnot` counting. Not built here; the flagship needed none of it. + +## Still open +- **W2b-B**: the Keep arm is tested, but its downstream-reuse burden (name and measure the + consumer that justifies the carrier) is not met. +- **Execution extent** (PR C): `execute_into` has no ranged entry point. +- **D-WFL-MASKOP**: `MaskOp` still reads as an assignment. Only the terminal-side lowering + moved. diff --git a/.claude/board/entries/README.md b/.claude/board/entries/README.md index c0579dff9..77c7f905b 100644 --- a/.claude/board/entries/README.md +++ b/.claude/board/entries/README.md @@ -25,10 +25,11 @@ index row, (3) no duplicate entry id. Checks 1 and 2 are deliberately opposite directions; the stranding this convention prevents shows up in exactly one of them, never both. -149 entries, 2026-08-06 .. 2026-09-23. +150 entries, 2026-08-06 .. 2026-09-23. | date | entry id | finding | file | |---|---|---|---| +| 2026-09-23 | `terminal-elects-materialization-range-plane` | | [2026-09-23-terminal-elects-materialization-range-plane.md](2026-09-23-terminal-elects-materialization-range-plane.md) | | 2026-09-23 | `quack-having-sym-sum-presence-mask` | | [2026-09-23-quack-having-sym-sum-presence-mask.md](2026-09-23-quack-having-sym-sum-presence-mask.md) | | 2026-09-23 | `cubecl-llvm-boundary-and-audit-regrade` | | [2026-09-23-cubecl-llvm-boundary-and-audit-regrade.md](2026-09-23-cubecl-llvm-boundary-and-audit-regrade.md) | | 2026-09-22 | `quack-duckdb-parity-t0-keyed-reduction` | | [2026-09-22-quack-duckdb-parity-t0-keyed-reduction.md](2026-09-22-quack-duckdb-parity-t0-keyed-reduction.md) |