@@ -1673,39 +1673,30 @@ fn resolve_rowstore_and_hop_masks(
16731673/// `n_rows` as a `u64` BEFORE any `as usize` cast, so an out-of-range
16741674/// `u64` target can never reach an indexing operation.
16751675///
1676- /// **Shape: gather, not sweep.** The hop reads ONLY the rows `src`
1677- /// names. For each such row it walks its participating facets in place
1678- /// out of that row's own 512 bytes — classid compare, payload decode and
1679- /// scatter together, one row's cache lines at a time.
1680- ///
1681- /// It did not always. Two earlier shapes swept the WHOLE population and
1682- /// then intersected with `src`: first one full-width
1683- /// `simd_rowstore_classid_mask` pass PER FACET (32 passes), then one
1684- /// `simd_rowstore_facet_match` pass answering all 32 at once. Both were
1685- /// replaced on measurement, not taste
1686- /// (`ISS-LGJ-HOP-SWEEPS-FULL-POPULATION`).
1687- ///
1688- /// The crossover a sweep would need in order to win does not exist.
1689- /// `examples/hop_gather_vs_sweep.rs` measured both shapes over four
1690- /// populations (1 024 … 262 144) × twelve frontier densities, asserting
1691- /// byte-identical output at every point: gather wins EVERY configuration,
1692- /// by 2 612× at the sparsest and still **1.7×** at 100 % density. The
1693- /// mechanism is why it holds even when every row is in the frontier — a
1694- /// sweep MATERIALISES an `n`-element per-row intermediate that each row
1695- /// reads exactly once, so the cost is never amortised, while the gather
1696- /// computes the same answer inline. A sweep is strictly more work at
1697- /// every density, not merely more work at sparse ones.
1698- ///
1699- /// Consequence worth stating plainly: this scalar gather beats a
1700- /// vectorised `ndarray::simd` sweep. The win is in NOT DOING THE WORK,
1701- /// not in the vector width — so no SIMD primitive is called here, and
1702- /// none is missing. (`simd_rowstore_facet_match` remains the kernel
1703- /// behind [`lgj_row_facet_match`]; it is not orphaned.)
1704- ///
1705- /// The one shape that could still favour a precomputed mask is REUSE —
1706- /// memoising the per-row answer across many hops on the same
1707- /// `(store, classid)`. That is a caching design with its own
1708- /// invalidation questions, and is deliberately not this function's.
1676+ /// **Shape: selection is mask algebra.** For each participating facet the
1677+ /// hop computes two whole-population predicates through the layout's own
1678+ /// lane geometry — `class_f` (classid == `edge_classid`, facet base +0) and
1679+ /// `struct_f` (`payload_hi32 == 0`, facet base +12), both the same strided
1680+ /// equality primitive — and conjoins them with `src` in ONE truth-table
1681+ /// pass (`mask_ternlog_assign::<AND3>`). No row is examined to decide
1682+ /// whether it participates. The only walk left is the scatter, which
1683+ /// EMITS from the selected set: the destination index is decoded from the
1684+ /// selected row's payload, the operand of a permutation rather than a
1685+ /// membership decision.
1686+ ///
1687+ /// It did not always. A gather (walk `src`'s set bits, read each row's
1688+ /// facets in place) measured faster on the AoS store and shipped briefly;
1689+ /// the operator ruled it out — walking a population's set bits is a
1690+ /// serialization of a population that is already there, whether or not it
1691+ /// allocates — and R1 restored the mask shape at a measured 19× cost on
1692+ /// AoS. The facet-major columnar store (ABI minor 10) is what makes the
1693+ /// lawful shape fast: every predicate becomes a contiguous pass, and the
1694+ /// hop runs 3.3–4.8× over AoS at every frontier arm. The two-AND spelling
1695+ /// of the conjunction was the last rank-1 residue of that arc.
1696+ ///
1697+ /// The next rung is a semiring product (`dst = src ⊗ A`) over an adjacency
1698+ /// operand this ABI does not yet carry — the V3 `EdgeBlock` in the row KEY
1699+ /// (decode modes 1..=3, RESERVED).
17091700///
17101701/// `docs/abi.md` §13 is the full normative statement.
17111702#[ no_mangle]
@@ -1762,15 +1753,17 @@ pub extern "C" fn lgj_hop(
17621753
17631754 // SELECTION IS MASK ALGEBRA.
17641755 //
1765- // selected_f = src ∧ class_f ∧ struct_f
1756+ // selected_f = ternlog<AND3>( class_f, src, struct_f)
17661757 // dst = ⋁_{f ∈ participation} scatter(selected_f)
17671758 //
17681759 // Both predicates are the SAME strided-equality primitive
17691760 // (`simd_rowstore_u32_eq_mask`) at two offsets into the facet — the
1770- // classid at +0, the structured-edge gate at +12 — and the two ANDs
1771- // are word-parallel over 64 rows at a time. No row is examined to
1772- // decide whether it participates; participation is computed for the
1773- // whole population and intersected.
1761+ // classid at +0, the structured-edge gate at +12 — and their
1762+ // conjunction with `src` is ONE 3-input truth-table pass
1763+ // (`mask_ternlog_assign::<AND3>`, one VPTERNLOGQ per 512 bits), not
1764+ // two ANDs through a scratch write. No row is examined to decide
1765+ // whether it participates; participation is computed for the whole
1766+ // population and intersected.
17741767 //
17751768 // Three shapes preceded this one and each traded the algebra for
17761769 // arithmetic:
@@ -1814,12 +1807,17 @@ pub extern "C" fn lgj_hop(
18141807 edge_classid,
18151808 & mut selected,
18161809 ) ;
1817- // ∧ src — narrow to the frontier.
1818- kernels:: simd_mask_and_assign ( & mut selected, & src_snapshot) ;
18191810 // struct_f — payload_hi32 == 0 marks a structured edge.
18201811 kernels:: simd_rowstore_u32_eq_mask ( bytes, h_off, h_stride, n, 0 , & mut structured) ;
1821- // ∧ — the gate that used to be an `if`.
1822- kernels:: simd_mask_and_assign ( & mut selected, & structured) ;
1812+ // ∧ src ∧ struct_f — ONE truth-table pass. AND is the rank-1
1813+ // spelling of a mask op: `selected & src & struct_f` is the
1814+ // 3-input table AND3 (0x80), and spelling it as two `mask_and`
1815+ // passes wrote every word twice to say it once.
1816+ kernels:: simd_mask_ternlog_assign :: < { kernels:: ternlog:: AND3 } > (
1817+ & mut selected,
1818+ & src_snapshot,
1819+ & structured,
1820+ ) ;
18231821
18241822 // Emit from the SELECTED set. Every row reached here has already
18251823 // satisfied all three predicates; the walk decides nothing.
0 commit comments