From 7695fa685bf40594c68b4604163c64be8e2aa10f Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 15:19:26 +0000 Subject: [PATCH] =?UTF-8?q?simd:=20mask=5Fternlog=5Fpopcount=20/=20mask=5F?= =?UTF-8?q?ternlog=5Fany=20=E2=80=94=20a=203-input=20Boolean=20membership?= =?UTF-8?q?=20ends=20in=20Count/Any=20with=20no=20mask=20written?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Built only from U64x8 methods every realization already carries (ternlog, popcnt, +, |, reduce_sum), so no backend code is added: the missing piece was the slice loop, not an ISA primitive. Word-level contract identical to mask_ternlog + popcount_batch_u64 / mask_any; register padding is never counted (an odd table evaluates padding lanes to all-ones). Probe examples/ternlog_fold_probe.rs: Count 1.3-1.9x, Any 2.6-6.2x over the materializing pair on avx2/avx512. Tests cover all 256 tables x 14 lengths x dense/sparse against the pair they replace; parity crate extended, green on v4, v3, neon-qemu, wasm, wasm-scalar. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01GXUahz73MZxtxWcfpHp9dG --- .claude/blackboard.md | 17 ++ Cargo.toml | 4 + crates/simd-masking-parity/src/lib.rs | 29 ++-- examples/ternlog_fold_probe.rs | 166 +++++++++++++++++++ src/simd.rs | 2 + src/simd_masking_ops.rs | 227 ++++++++++++++++++++++++++ 6 files changed, 435 insertions(+), 10 deletions(-) create mode 100644 examples/ternlog_fold_probe.rs diff --git a/.claude/blackboard.md b/.claude/blackboard.md index d32d099e..eb3aded7 100644 --- a/.claude/blackboard.md +++ b/.claude/blackboard.md @@ -1,3 +1,20 @@ +## 2026-09-23 (21) — ternlog → Count/Any without a mask: a slice loop, NOT a new ISA primitive + +Question asked (lance-graph #1270 prompt): what is the smallest T1 operation that lets an arbitrary 2/3-input Boolean membership END in Count/Any without writing a mask — and does existing `U64x8` composition already do it register-only? + +**Answer (MEASURED): the composition exists on every realization; only the slice loop was missing.** `U64x8::{ternlog::, popcnt, +, |, reduce_sum}` are present on avx512 / avx2-polyfill / scalar / neon / wasm, so no backend code is added. Added two slice functions to `simd_masking_ops.rs` + the facade, built only from those methods: + +- `mask_ternlog_popcount::(a, b, c) -> u64` — `Σ popcount(ternlog(a,b,c))`, lane-wise accumulate, one `reduce_sum`. +- `mask_ternlog_any::(a, b, c) -> bool` — OR-accumulate, horizontal test once per block of 8 chunks. + +Word-level contract (both): every bit of every word counts, exactly as `popcount_batch_u64`/`mask_any` over the materialized `mask_ternlog` result would; an odd `IMM` sets the last word's dead tail bits and they count (caller masks the last word). Register PADDING never counts — the tail runs through the packed op and only the live lanes are read, because `ternlog(0,0,0)` is all-ones for an odd table (pinned by `mask_ternlog_folds_never_count_register_padding`, NOR3 over 9 words). + +Evidence: `examples/ternlog_fold_probe.rs` (M = materialize+reduce, R = register fold, S = scalar fused), `AND2_OR`. Count M/R: avx2 1.27–1.50×, avx512 1.90–1.94×. Any M/R (all-zero worst case): avx2 3.2–6.2×, avx512 2.6–4.4×. A per-chunk Any test LOST to M on avx2 at 16K words (0.91×) — hence the block. S ≈ R on avx2 and at the memory-bound avx512 size: the win is not writing the mask, not SIMD per se. + +Tests: `mask_ternlog_folds_match_the_materializing_pair_for_all_256_tables` (all 256 tables × 14 lengths × dense/sparse, against the exact pair they replace), padding test, two length-mismatch panics. Parity: `slice_ternlog!` in `crates/simd-masking-parity` now also checks both folds (`0x69x` count, `0x65x` any); green on native v4, native v3, neon-qemu, wasm, wasm-scalar. + +Consumer: lance-graph-mask-risc fused terminal `MaskOp::{And,Or,Xor,AndNot,Ternlog} → Count/Any`. + ## 2026-09-21 (20) — six index-addressed mask primitives (0xDxx): gather / scatter-or / keyed group-sum (direct + via index) / indexed equality / ORDERED key-run distinct fold All in `simd_masking_ops.rs` + the `simd::` facade, documented in ADDRESS terms only (`index` / `table` / `keys`; no join, foreign-key, semijoin, table-name or ERP vocabulary — T1 does not know what a consumer means by an index lane). All are deliberately scalar bit-walks: permutations/scatters indexed by data, not a fixed stride, so none of this crate's backends can vector-load them (same shape as `masked_strided_group_sum`, which is NOT a keyed group-by — it sums one record's own byte-groups into a scalar; zero callers of it are affected). diff --git a/Cargo.toml b/Cargo.toml index 6881d86d..652c2d0c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -59,6 +59,10 @@ required-features = ["std"] name = "ternlog_amortization_probe" required-features = ["std"] +[[example]] +name = "ternlog_fold_probe" +required-features = ["std"] + [[example]] name = "r2il_column_scan_probe" required-features = ["std"] diff --git a/crates/simd-masking-parity/src/lib.rs b/crates/simd-masking-parity/src/lib.rs index 9fe7303f..96ee9614 100644 --- a/crates/simd-masking-parity/src/lib.rs +++ b/crates/simd-masking-parity/src/lib.rs @@ -38,12 +38,12 @@ use ndarray::simd::{ le_i32_to_mask_under, le_u64_to_mask, le_u8_to_mask, lt_i32_to_mask, lt_i32_to_mask_under, lt_u64_to_mask, lt_u8_to_mask, mask_all, mask_and, mask_and_assign, mask_andnot, mask_andnot_assign, mask_any, mask_gather_u32, mask_not, mask_not_assign, mask_or, mask_or_assign, mask_scatter_or_u32, mask_set_range, mask_shift_morton, - mask_ternlog, mask_ternlog_assign, mask_xor, mask_xor_assign, masked_group_sum_i32, masked_group_sum_i32_via, - masked_key_run_count_u32, masked_max_i32, masked_min_i32, masked_strided_group_sum, masked_sum_i32, - masked_sum_wrapping_add_i32, ne_i32_to_mask, - ne_i32_to_mask_under, ne_u32_to_mask, ne_u32_to_mask_under, ne_u64_to_mask, ne_u8_to_mask, - ternary_match_strided_to_mask, ternary_match_u32_to_mask, ternary_match_u32_to_mask_under, - ternary_match_u64_to_mask, ternary_match_u64_to_mask_under, ternlog, I32x16, KeyRunCarry, MortonDir, U32x16, U64x8, + mask_ternlog, mask_ternlog_any, mask_ternlog_assign, mask_ternlog_popcount, mask_xor, mask_xor_assign, + masked_group_sum_i32, masked_group_sum_i32_via, masked_key_run_count_u32, masked_max_i32, masked_min_i32, + masked_strided_group_sum, masked_sum_i32, masked_sum_wrapping_add_i32, ne_i32_to_mask, ne_i32_to_mask_under, + ne_u32_to_mask, ne_u32_to_mask_under, ne_u64_to_mask, ne_u8_to_mask, ternary_match_strided_to_mask, + ternary_match_u32_to_mask, ternary_match_u32_to_mask_under, ternary_match_u64_to_mask, + ternary_match_u64_to_mask_under, ternlog, I32x16, KeyRunCarry, MortonDir, U32x16, U64x8, }; /// Number of check groups [`run`] executes (for the log line only). @@ -642,6 +642,14 @@ fn check_mask_algebra() -> Result<(), u32> { if t != dst { return Err($code | 0x8); } + // The no-mask folds equal the materializing pair they replace. + let want: u64 = dst.iter().map(|w| u64::from(w.count_ones())).sum(); + if mask_ternlog_popcount::(&a, &b, &c) != want { + return Err($code | 0x80); + } + if mask_ternlog_any::(&a, &b, &c) != mask_any(&dst) { + return Err($code | 0x40); + } }}; } slice_ternlog!(ternlog::AND2_OR, 0x610); @@ -788,9 +796,7 @@ fn check_masked_reductions() -> Result<(), u32> { } let want_wrapping_add = (0..n) .filter(|&i| (m[i / 64] >> (i % 64)) & 1 == 1) - .fold(0i64, |acc, i| { - acc.wrapping_add(vals[i].wrapping_add(rhs[i]) as i64) - }); + .fold(0i64, |acc, i| acc.wrapping_add(vals[i].wrapping_add(rhs[i]) as i64)); if masked_sum_wrapping_add_i32(&vals, &rhs, m) != want_wrapping_add { return Err(0x850 | k); } @@ -1410,7 +1416,10 @@ fn check_gather_scatter_group() -> Result<(), u32> { if n >= 2 && keys[0] < keys[n - 1] { let mut bad = keys[1..].to_vec(); bad.push(keys[0]); - let mut c = KeyRunCarry { key: Some(bad[0]), hit: true }; + let mut c = KeyRunCarry { + key: Some(bad[0]), + hit: true, + }; let before = c; if masked_key_run_count_u32(&bad, &sel_bits, &mut c).is_some() { return Err(0xD52); diff --git a/examples/ternlog_fold_probe.rs b/examples/ternlog_fold_probe.rs new file mode 100644 index 00000000..a6806149 --- /dev/null +++ b/examples/ternlog_fold_probe.rs @@ -0,0 +1,166 @@ +//! Can a 2/3-input Boolean membership END in Count/Any without writing a mask? +//! +//! The question this probe answers is narrower than "is a fused primitive +//! faster". It is: **does the existing `U64x8` surface already compose the +//! fold register-only**, or is a new T1 primitive required? Every realization +//! of `U64x8` (avx512 / avx2-polyfill / scalar / neon / wasm) carries +//! `ternlog::`, `popcnt`, `+` and `reduce_sum`, so the composition +//! `ternlog → popcnt → accumulate → reduce_sum` type-checks everywhere. What +//! the probe measures is whether that composition is a real win over the +//! materializing path, on each backend this workspace builds for. +//! +//! Three arms compute the IDENTICAL result for `(a & b) | c` (`AND2_OR`, the +//! mask-risc flagship immediate); a correctness gate aborts on disagreement: +//! +//! | arm | shape | derived words written | +//! |---|---|---| +//! | M | `mask_ternlog::` into a buffer, then `popcount_batch_u64` | n | +//! | R | chunked `U64x8::ternlog → popcnt`, lane-wise accumulate, one `reduce_sum` | 0 | +//! | S | plain scalar `((a & b) \| c).count_ones()` over zipped slices | 0 | +//! +//! Any is measured the same way: M = materialize + `mask_any`; R = +//! OR-accumulate with a test once per block of 8 chunks; S = scalar `any`. Any +//! is timed on an all-zero result (the worst case: no early exit is possible). +//! A first R arm that tested every chunk LOST to M on avx2 at 16K words (0.91×) +//! — the per-chunk horizontal test cost more than the ternlog it guarded. +//! +//! Measured 2026-09-23 (median ns, avx2 = `config-v3`, avx512 = `config-v4`, +//! a host without `avx512vpopcntdq`, so avx512 `popcnt` is the LUT path): +//! +//! | backend | words | Count M/R | Any M/R | +//! |---|---|---|---| +//! | avx2 | 16 384 | 1.27 | 3.24 | +//! | avx2 | 262 144 | 1.50 | 6.18 | +//! | avx512 | 16 384 | 1.90 | 2.59 | +//! | avx512 | 262 144 | 1.94 | 4.39 | +//! +//! The scalar fused arm S is ~equal to R on avx2 and at the memory-bound size +//! on avx512 — the win here is NOT writing the mask, not SIMD per se. +//! +//! Usage (the backend is compile-time and the default config is +//! `target-cpu=native`, so pin the tier and read the `backend:` line): +//! +//! ```text +//! env -u RUSTFLAGS cargo --config .cargo/config-v3.toml run --release --example ternlog_fold_probe +//! env -u RUSTFLAGS cargo --config .cargo/config-v4.toml run --release --example ternlog_fold_probe +//! ``` + +use std::time::Instant; + +use ndarray::simd::ternlog::AND2_OR; +use ndarray::simd::{mask_any, mask_ternlog, popcount_batch_u64, U64x8}; + +const L: usize = U64x8::LANES; + +fn lcg(s: &mut u64) -> u64 { + *s = s + .wrapping_mul(6364136223846793005) + .wrapping_add(1442695040888963407); + *s +} + +/// Arm R: the register-only composition over existing `U64x8` methods. +fn fold_count(a: &[u64], b: &[u64], c: &[u64]) -> u64 { + let (ca, ta) = a.as_chunks::(); + let (cb, tb) = b.as_chunks::(); + let (cc, tc) = c.as_chunks::(); + let mut acc = U64x8::splat(0); + for ((x, y), z) in ca.iter().zip(cb).zip(cc) { + let t = U64x8::from_array(*x).ternlog::(U64x8::from_array(*y), U64x8::from_array(*z)); + acc += t.popcnt(); + } + let mut tail = 0u64; + for ((x, y), z) in ta.iter().zip(tb).zip(tc) { + tail += ((x & y) | z).count_ones() as u64; + } + acc.reduce_sum() + tail +} + +fn fold_any(a: &[u64], b: &[u64], c: &[u64]) -> bool { + let (ca, ta) = a.as_chunks::(); + let (cb, tb) = b.as_chunks::(); + let (cc, tc) = c.as_chunks::(); + // OR-accumulate in a register and test once per block of chunks: a + // per-chunk horizontal test costs more than the ternlog it guards. + const BLOCK: usize = 8; + let mut acc = U64x8::splat(0); + for (i, ((x, y), z)) in ca.iter().zip(cb).zip(cc).enumerate() { + acc |= U64x8::from_array(*x).ternlog::(U64x8::from_array(*y), U64x8::from_array(*z)); + if i % BLOCK == BLOCK - 1 && acc.to_array().iter().any(|&w| w != 0) { + return true; + } + } + if acc.to_array().iter().any(|&w| w != 0) { + return true; + } + ta.iter() + .zip(tb) + .zip(tc) + .any(|((x, y), z)| (x & y) | z != 0) +} + +fn median(reps: usize, mut f: impl FnMut() -> T) -> (f64, T) { + let mut ts = Vec::with_capacity(reps); + let mut last = None; + for _ in 0..reps { + let t = Instant::now(); + last = Some(std::hint::black_box(f())); + ts.push(t.elapsed().as_nanos() as f64); + } + ts.sort_by(|x, y| x.total_cmp(y)); + (ts[reps / 2], last.expect("reps > 0")) +} + +fn main() { + let backend = if cfg!(target_feature = "avx512f") { + "avx512" + } else if cfg!(target_feature = "avx2") { + "avx2-polyfill" + } else { + "scalar" + }; + println!("backend: {backend}"); + println!( + "{:>8} {:>6} {:>12} {:>12} {:>12} {:>8} {:>8}", + "words", "term", "M_ns", "R_ns", "S_ns", "M/R", "S/R" + ); + let mut seed = 0x7E4_u64; + for &n in &[8usize, 128, 16_384, 262_144] { + let a: Vec = (0..n).map(|_| lcg(&mut seed)).collect(); + let b: Vec = (0..n).map(|_| lcg(&mut seed)).collect(); + let c: Vec = (0..n).map(|_| lcg(&mut seed) & lcg(&mut seed)).collect(); + let zero = vec![0u64; n]; + let mut dst = vec![0u64; n]; + let reps = if n > 100_000 { 41 } else { 2001 }; + + let (m, vm) = median(reps, || { + mask_ternlog::(&a, &b, &c, &mut dst); + popcount_batch_u64(&dst) + }); + let (r, vr) = median(reps, || fold_count(&a, &b, &c)); + let (s, vs) = median(reps, || { + a.iter() + .zip(&b) + .zip(&c) + .map(|((x, y), z)| ((x & y) | z).count_ones() as u64) + .sum::() + }); + assert!(vm == vr && vr == vs, "count disagreement at n={n}: {vm} {vr} {vs}"); + println!("{n:>8} {:>6} {m:>12.0} {r:>12.0} {s:>12.0} {:>8.2} {:>8.2}", "Count", m / r, s / r); + + // Any over an all-zero result: a = b = c = 0, so no early exit. + let (m, vm) = median(reps, || { + mask_ternlog::(&zero, &zero, &zero, &mut dst); + mask_any(&dst) + }); + let (r, vr) = median(reps, || fold_any(&zero, &zero, &zero)); + let (s, vs) = median(reps, || { + zero.iter() + .zip(&zero) + .zip(&zero) + .any(|((x, y), z)| (x & y) | z != 0) + }); + assert!(!vm && !vr && !vs, "any disagreement at n={n}"); + println!("{n:>8} {:>6} {m:>12.0} {r:>12.0} {s:>12.0} {:>8.2} {:>8.2}", "Any", m / r, s / r); + } +} diff --git a/src/simd.rs b/src/simd.rs index eefdcbaa..01842d27 100644 --- a/src/simd.rs +++ b/src/simd.rs @@ -823,7 +823,9 @@ pub use crate::simd_masking_ops::{ mask_set_range, mask_shift_morton, mask_ternlog, + mask_ternlog_any, mask_ternlog_assign, + mask_ternlog_popcount, mask_xor, mask_xor_assign, masked_group_count_u32, diff --git a/src/simd_masking_ops.rs b/src/simd_masking_ops.rs index 8a2c4945..ff7b2591 100644 --- a/src/simd_masking_ops.rs +++ b/src/simd_masking_ops.rs @@ -630,6 +630,139 @@ pub fn mask_ternlog_assign(a: &mut [u64], b: &[u64], c: &[u64]) } } +/// `Σ popcount(ternlog::(a, b, c))` over `u64` mask words — the count +/// of a 3-input Boolean membership, with **no mask written**. +/// +/// The fold [`mask_ternlog`] followed by [`popcount_batch_u64`](crate::bitwise::popcount_batch_u64) would compute, +/// minus the intermediate buffer: each 8-word chunk is combined in a register +/// (`U64x8::ternlog`), popcounted in the register (`U64x8::popcnt`) and added +/// lane-wise into an accumulator that is reduced once at the end. It is built +/// only from `U64x8` methods every realization carries, so it adds no backend +/// code and no `cfg`. +/// +/// Two-input functions are the same call with `c` ignored by the table +/// (`ternlog::AND2` is `a & b`, whatever `c` holds); pass any same-length +/// slice for `c`, e.g. `a` again. +/// +/// # Word-level contract (tail bits) +/// +/// Every bit of every word is counted, exactly as [`popcount_batch_u64`](crate::bitwise::popcount_batch_u64) over +/// the materialized result would count it. For an even `IMM` and conforming +/// inputs (tail bits zero) that is the row count; for an **odd** `IMM` (true of +/// all-zero inputs) the dead tail bits of the last word are set by the table +/// and ARE counted — mask them out of the last word yourself, exactly as +/// [`mask_ternlog`] documents for its `dst`. Register padding past the end of +/// the slices is never counted. +/// +/// # Panics +/// +/// Panics unless `a.len() == b.len() == c.len()`. +/// +/// # Examples +/// +/// ``` +/// use ndarray::simd::{mask_ternlog_popcount, ternlog}; +/// +/// let a = [0b1100u64, u64::MAX]; +/// let b = [0b1010u64, 0]; +/// let c = [0b0001u64, 0b111]; +/// // (a & b) | c: word 0 -> 0b1001 (2 bits), word 1 -> 0b111 (3 bits) +/// assert_eq!(mask_ternlog_popcount::<{ ternlog::AND2_OR }>(&a, &b, &c), 5); +/// ``` +#[inline] +pub fn mask_ternlog_popcount(a: &[u64], b: &[u64], c: &[u64]) -> u64 { + assert_eq!(a.len(), b.len(), "mask_ternlog_popcount: a/b length mismatch"); + assert_eq!(a.len(), c.len(), "mask_ternlog_popcount: a/c length mismatch"); + const L: usize = crate::simd::U64x8::LANES; + let (ca, ta) = a.as_chunks::(); + let (cb, tb) = b.as_chunks::(); + let (cc, tc) = c.as_chunks::(); + let mut acc = crate::simd::U64x8::splat(0); + for ((x, y), z) in ca.iter().zip(cb).zip(cc) { + let va = crate::simd::U64x8::from_array(*x); + let vb = crate::simd::U64x8::from_array(*y); + let vc = crate::simd::U64x8::from_array(*z); + acc += va.ternlog::(vb, vc).popcnt(); + } + let mut total = acc.reduce_sum(); + if !ta.is_empty() { + let va = crate::simd::U64x8::from_array(pad_tail(ta)); + let vb = crate::simd::U64x8::from_array(pad_tail(tb)); + let vc = crate::simd::U64x8::from_array(pad_tail(tc)); + // Only the live lanes: a padding lane holds ternlog(0,0,0), which is + // all-ones for an odd IMM and must not be counted. + let t = va.ternlog::(vb, vc).to_array(); + total += t[..ta.len()] + .iter() + .map(|w| u64::from(w.count_ones())) + .sum::(); + } + total +} + +/// `true` iff any bit of `ternlog::(a, b, c)` is set, over `u64` mask +/// words — a 3-input Boolean membership tested for non-emptiness with **no +/// mask written**. +/// +/// The fold [`mask_ternlog`] followed by [`mask_any`] would compute, minus the +/// intermediate buffer. Each 8-word chunk is combined in a register and +/// OR-accumulated; the accumulator is tested once per block of chunks rather +/// than per chunk (a per-chunk horizontal test costs more than the ternlog it +/// guards), so a hit returns within one block of where it occurs. Built only +/// from `U64x8` methods every realization carries. +/// +/// Tail bits follow [`mask_ternlog_popcount`]'s word-level contract: for an odd +/// `IMM` the last word's dead tail bits are set by the table and make this +/// `true`; register padding past the end of the slices never does. +/// +/// # Panics +/// +/// Panics unless `a.len() == b.len() == c.len()`. +/// +/// # Examples +/// +/// ``` +/// use ndarray::simd::{mask_ternlog_any, ternlog}; +/// +/// let a = [0b1100u64; 3]; +/// let b = [0b0011u64; 3]; +/// let c = [0u64; 3]; +/// assert!(!mask_ternlog_any::<{ ternlog::AND3 }>(&a, &b, &c)); // disjoint +/// assert!(mask_ternlog_any::<{ ternlog::OR3 }>(&a, &b, &c)); +/// ``` +#[inline] +pub fn mask_ternlog_any(a: &[u64], b: &[u64], c: &[u64]) -> bool { + assert_eq!(a.len(), b.len(), "mask_ternlog_any: a/b length mismatch"); + assert_eq!(a.len(), c.len(), "mask_ternlog_any: a/c length mismatch"); + const L: usize = crate::simd::U64x8::LANES; + /// Chunks OR-accumulated between two horizontal tests. + const BLOCK: usize = 8; + let (ca, ta) = a.as_chunks::(); + let (cb, tb) = b.as_chunks::(); + let (cc, tc) = c.as_chunks::(); + let mut acc = crate::simd::U64x8::splat(0); + for (i, ((x, y), z)) in ca.iter().zip(cb).zip(cc).enumerate() { + let va = crate::simd::U64x8::from_array(*x); + let vb = crate::simd::U64x8::from_array(*y); + let vc = crate::simd::U64x8::from_array(*z); + acc |= va.ternlog::(vb, vc); + if i % BLOCK == BLOCK - 1 && acc.to_array().iter().any(|&w| w != 0) { + return true; + } + } + if acc.to_array().iter().any(|&w| w != 0) { + return true; + } + if !ta.is_empty() { + let va = crate::simd::U64x8::from_array(pad_tail(ta)); + let vb = crate::simd::U64x8::from_array(pad_tail(tb)); + let vc = crate::simd::U64x8::from_array(pad_tail(tc)); + let t = va.ternlog::(vb, vc).to_array(); + return t[..ta.len()].iter().any(|&w| w != 0); + } + false +} + /// Sum of `values[i]` where mask bit `i` is set, widened to `i64`. /// /// Bit order is the module convention: element `i` is bit `i % 64` of @@ -4508,6 +4641,100 @@ mod tests { assert_eq!(d[1] & TAIL_MASK, 0, "AND3 tail follows a's tail"); } + /// The ternlog→Count/Any folds against the MATERIALIZING spelling + /// (`mask_ternlog` + `popcount_batch_u64` / `mask_any`) — the exact pair + /// they replace — over the family's length set, dense and sparse. + fn check_ternlog_fold_imm() { + for &len in &[0usize, 1, 2, 7, 8, 9, 15, 16, 17, 63, 64, 65, 100, 129] { + for sparse in [false, true] { + let mut seed = 0xF01D_0000_0000_0001 ^ (IMM as u64) ^ (len as u64) << 8 ^ u64::from(sparse); + let mut w = || { + let x = splitmix64(&mut seed); + if sparse { + x & splitmix64(&mut seed) & splitmix64(&mut seed) & splitmix64(&mut seed) + } else { + x + } + }; + let a: Vec = (0..len).map(|_| w()).collect(); + let b: Vec = (0..len).map(|_| w()).collect(); + let c: Vec = (0..len).map(|_| w()).collect(); + let mut dst = vec![0u64; len]; + mask_ternlog::(&a, &b, &c, &mut dst); + assert_eq!( + mask_ternlog_popcount::(&a, &b, &c), + crate::bitwise::popcount_batch_u64(&dst), + "count imm={IMM:#04x} len={len} sparse={sparse}" + ); + assert_eq!( + mask_ternlog_any::(&a, &b, &c), + mask_any(&dst), + "any imm={IMM:#04x} len={len} sparse={sparse}" + ); + } + } + } + + #[test] + fn mask_ternlog_folds_match_the_materializing_pair_for_all_256_tables() { + macro_rules! all_imms { + ($($imm:literal),* $(,)?) => { $( check_ternlog_fold_imm::<$imm>(); )* }; + } + all_imms!( + 0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0x10, 0x11, + 0x12, 0x13, 0x14, 0x15, 0x16, 0x17, 0x18, 0x19, 0x1A, 0x1B, 0x1C, 0x1D, 0x1E, 0x1F, 0x20, 0x21, 0x22, 0x23, + 0x24, 0x25, 0x26, 0x27, 0x28, 0x29, 0x2A, 0x2B, 0x2C, 0x2D, 0x2E, 0x2F, 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, + 0x36, 0x37, 0x38, 0x39, 0x3A, 0x3B, 0x3C, 0x3D, 0x3E, 0x3F, 0x40, 0x41, 0x42, 0x43, 0x44, 0x45, 0x46, 0x47, + 0x48, 0x49, 0x4A, 0x4B, 0x4C, 0x4D, 0x4E, 0x4F, 0x50, 0x51, 0x52, 0x53, 0x54, 0x55, 0x56, 0x57, 0x58, 0x59, + 0x5A, 0x5B, 0x5C, 0x5D, 0x5E, 0x5F, 0x60, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, 0x67, 0x68, 0x69, 0x6A, 0x6B, + 0x6C, 0x6D, 0x6E, 0x6F, 0x70, 0x71, 0x72, 0x73, 0x74, 0x75, 0x76, 0x77, 0x78, 0x79, 0x7A, 0x7B, 0x7C, 0x7D, + 0x7E, 0x7F, 0x80, 0x81, 0x82, 0x83, 0x84, 0x85, 0x86, 0x87, 0x88, 0x89, 0x8A, 0x8B, 0x8C, 0x8D, 0x8E, 0x8F, + 0x90, 0x91, 0x92, 0x93, 0x94, 0x95, 0x96, 0x97, 0x98, 0x99, 0x9A, 0x9B, 0x9C, 0x9D, 0x9E, 0x9F, 0xA0, 0xA1, + 0xA2, 0xA3, 0xA4, 0xA5, 0xA6, 0xA7, 0xA8, 0xA9, 0xAA, 0xAB, 0xAC, 0xAD, 0xAE, 0xAF, 0xB0, 0xB1, 0xB2, 0xB3, + 0xB4, 0xB5, 0xB6, 0xB7, 0xB8, 0xB9, 0xBA, 0xBB, 0xBC, 0xBD, 0xBE, 0xBF, 0xC0, 0xC1, 0xC2, 0xC3, 0xC4, 0xC5, + 0xC6, 0xC7, 0xC8, 0xC9, 0xCA, 0xCB, 0xCC, 0xCD, 0xCE, 0xCF, 0xD0, 0xD1, 0xD2, 0xD3, 0xD4, 0xD5, 0xD6, 0xD7, + 0xD8, 0xD9, 0xDA, 0xDB, 0xDC, 0xDD, 0xDE, 0xDF, 0xE0, 0xE1, 0xE2, 0xE3, 0xE4, 0xE5, 0xE6, 0xE7, 0xE8, 0xE9, + 0xEA, 0xEB, 0xEC, 0xED, 0xEE, 0xEF, 0xF0, 0xF1, 0xF2, 0xF3, 0xF4, 0xF5, 0xF6, 0xF7, 0xF8, 0xF9, 0xFA, 0xFB, + 0xFC, 0xFD, 0xFE, 0xFF, + ); + } + + #[test] + fn mask_ternlog_folds_never_count_register_padding() { + // NOR3 (0x01) is true of all-zero inputs, so a zero-padded register + // lane evaluates to all-ones. 9 words = one full chunk + a 1-word + // tail, leaving 7 padding lanes that would add 448 if counted. + let z = [0u64; 9]; + assert_eq!(mask_ternlog_popcount::<0x01>(&z, &z, &z), 9 * 64); + // A single live lane is enough to decide Any, but with every live + // word forced to zero the answer must come from live lanes only. + let ones = [u64::MAX; 9]; + assert_eq!(mask_ternlog_popcount::<0x01>(&ones, &ones, &ones), 0); + assert!(!mask_ternlog_any::<0x01>(&ones, &ones, &ones), "padding must not make Any true"); + // Can-fire half: a hit in the tail alone is found. + let mut a = [0u64; 9]; + a[8] = 1 << 63; + assert!(mask_ternlog_any::<{ crate::simd::ternlog::OR3 }>(&a, &z, &z)); + // ...and one past the first block (chunk 8 = words 64..72), found too. + let mut far = vec![0u64; 200]; + far[70] = 1; + let zz = vec![0u64; 200]; + assert!(mask_ternlog_any::<{ crate::simd::ternlog::OR3 }>(&far, &zz, &zz)); + assert_eq!(mask_ternlog_popcount::<{ crate::simd::ternlog::OR3 }>(&far, &zz, &zz), 1); + } + + #[test] + #[should_panic(expected = "length mismatch")] + fn mask_ternlog_popcount_rejects_length_mismatch() { + mask_ternlog_popcount::<0x80>(&[0u64; 4], &[0u64; 3], &[0u64; 4]); + } + + #[test] + #[should_panic(expected = "length mismatch")] + fn mask_ternlog_any_rejects_length_mismatch() { + mask_ternlog_any::<0x80>(&[0u64; 4], &[0u64; 4], &[0u64; 5]); + } + #[test] #[should_panic(expected = "length mismatch")] fn mask_ternlog_rejects_length_mismatch() {